<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Cell. Infect. Microbiol.</journal-id>
<journal-title>Frontiers in Cellular and Infection Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Cell. Infect. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">2235-2988</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fcimb.2024.1384809</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Cellular and Infection Microbiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>FAIR compliant database development for human microbiome data samples</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Dorst</surname>
<given-names>Mathieu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Zeevenhooven</surname>
<given-names>Nathan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wilding</surname>
<given-names>Rory</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mende</surname>
<given-names>Daniel</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/224969"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Brandt</surname>
<given-names>Bernd W.</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/139628"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zaura</surname>
<given-names>Egija</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/96392"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hoekstra</surname>
<given-names>Alfons</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/401413"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Sheraton</surname>
<given-names>Vivek M.</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2654184"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Informatics Institute, University of Amsterdam</institution>, <addr-line>Amsterdam</addr-line>, <country>Netherlands</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Supabase Limited Liability Company (LLC)</institution>, <addr-line>San Francisco, CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Amsterdam Institute of Infection and Immunity, Amsterdam University Medical Center</institution>, <addr-line>Amsterdam</addr-line>, <country>Netherlands</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Preventive Dentistry, Academic Centre for Dentistry Amsterdam, Vrije Universiteit Amsterdam and University of Amsterdam</institution>, <addr-line>Amsterdam</addr-line>, <country>Netherlands</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Computational Science Lab, Informatics Institute, University of Amsterdam</institution>, <addr-line>Amsterdam</addr-line>, <country>Netherlands</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Thuy Do, University of Leeds, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Gelio Alves, National Institutes of Health (NIH), United States</p>
<p>Jaroslaw Mazuryk, Universit&#xe9; Catholique de Louvain, Belgium</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Vivek M. Sheraton, <email xlink:href="mailto:v.s.muniraj@uva.nl">v.s.muniraj@uva.nl</email>
</p>
</fn>
<fn fn-type="equal" id="fn003">
<p>&#x2020;These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>07</day>
<month>05</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>14</volume>
<elocation-id>1384809</elocation-id>
<history>
<date date-type="received">
<day>10</day>
<month>02</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>17</day>
<month>04</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Dorst, Zeevenhooven, Wilding, Mende, Brandt, Zaura, Hoekstra and Sheraton</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Dorst, Zeevenhooven, Wilding, Mende, Brandt, Zaura, Hoekstra and Sheraton</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Sharing microbiome data among researchers fosters new innovations and reduces cost for research. Practically, this means that the (meta)data will have to be standardized, transparent and readily available for researchers. The microbiome data and associated metadata will then be described with regards to composition and origin, in order to maximize the possibilities for application in various contexts of research. Here, we propose a set of tools and protocols to develop a real-time FAIR (Findable. Accessible, Interoperable and Reusable) compliant database for the handling and storage of human microbiome and host-associated data.</p>
</sec>
<sec>
<title>Methods</title>
<p>The conflicts arising from privacy laws with respect to metadata, possible human genome sequences in the metagenome shotgun data and FAIR implementations are discussed. Alternate pathways for achieving compliance in such conflicts are analyzed. Sample traceable and sensitive microbiome data, such as DNA sequences or geolocalized metadata are identified, and the role of the GDPR (General Data Protection Regulation) data regulations are considered. For the construction of the database, procedures have been realized to make data FAIR compliant, while preserving privacy of the participants providing the data.</p>
</sec>
<sec>
<title>Results and discussion</title>
<p>An open-source development platform, Supabase, was used to implement the microbiome database. Researchers can deploy this real-time database to access, upload, download and interact with human microbiome data in a FAIR complaint manner. In addition, a large language model (LLM) powered by ChatGPT is developed and deployed to enable knowledge dissemination and non-expert usage of the database.</p>
</sec>
</abstract>
<abstract abstract-type="graphical">
<title>Graphical Abstract</title>
<p>
<graphic xlink:href="fcimb-14-1384809-g006.tif" position="anchor"/>
</p>
</abstract>
<kwd-group>
<kwd>database</kwd>
<kwd>fair principles</kwd>
<kwd>general data protection regulation (GDPR)</kwd>
<kwd>(meta)data</kwd>
<kwd>microbiome</kwd>
<kwd>pseudonymize</kwd>
<kwd>real-time</kwd>
</kwd-group>
<contract-num rid="cn001">NWA.1389.20.080</contract-num>
<contract-sponsor id="cn001">Nederlandse Organisatie voor Wetenschappelijk Onderzoek<named-content content-type="fundref-id">10.13039/501100003246</named-content>
</contract-sponsor>
<counts>
<fig-count count="5"/>
<table-count count="0"/>
<equation-count count="0"/>
<ref-count count="32"/>
<page-count count="11"/>
<word-count count="4689"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Extra-intestinal Microbiome</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<sec id="s1_1">
<label>1.1</label>
<title>Data and data sharing</title>
<p>Data sharing facilitates collaboration between researchers and therefore could lead to new findings as various disciplines and research groups utilize the data differently (<xref ref-type="bibr" rid="B32">Yoong et&#xa0;al., 2022</xref>). The FAIR principles are standards by which data can be more easily exchanged, preserved, and curated for research purposes (<xref ref-type="bibr" rid="B9">Chue Hong et&#xa0;al., 2021</xref>). The downstream reuse of data would often run into significant issues, such as incomplete metadata, absence of raw data, or incompatibility of software, leading to datasets being practically unusable outside of the primary research case for which they were conceived (<xref ref-type="bibr" rid="B22">Roche et&#xa0;al., 2015</xref>). Crucially, in development of machine learning techniques, the efficiency of the algorithms relies heavily on the quality of the labels (<xref ref-type="bibr" rid="B30">Willemink et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B28">Wilding et&#xa0;al., 2022</xref>). In such cases, standardizing the metadata would enable the generation of high quality labels and help systematize the label generation process. Machines can then more easily query the database for the right data and perform operations on it, while researchers, for example, can easily combine datasets to get new insights. This removes the friction that exists with different data formats and gives space for more efficient and faster data handling, which results in new insights to evolve more rapidly (<xref ref-type="bibr" rid="B2">Abuimara et&#xa0;al., 2022</xref>). Large numerical datasets, such as data originating from multiscale simulation studies (<xref ref-type="bibr" rid="B24">Sheraton et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B5">B&#xe9;quignon et&#xa0;al., 2023</xref>), -omics (<xref ref-type="bibr" rid="B25">Subramanian et&#xa0;al., 2020</xref>) or imaging studies (<xref ref-type="bibr" rid="B7">Bray et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B4">Baglamis et&#xa0;al., 2023</xref>), should be properly categorized along with clearly described provenance. Additionally, the origin and composition of the data will have to be described, to enable data reconstructions in such a way that it cannot be misinterpreted. To overcome these obstacles of non-standardized or flawed use of data and to ensure proper data sharing, the Findability, Accessibility, Interoperability and Reusability (FAIR) principles were introduced (<xref ref-type="bibr" rid="B29">Wilkinson et&#xa0;al., 2016</xref>). A FAIR compliant database makes it very easy for all types of data to be discovered, since the metadata has been standardized (<xref ref-type="bibr" rid="B10">Da Silva Santos et&#xa0;al., 2023</xref>). Every FAIR object (image, table, text, etc.) or dataset should have a unique identifier assigned, which should then be described with rich metadata. In a study evaluating the FAIR4HEALTH initiative, the implementation of FAIR principles has been proven to save researchers on average approximately 56% of their time in data gathering and compilation activities and approximately 16,800 euros per month in institution funding, when conducting health research efforts (<xref ref-type="bibr" rid="B20">Mart&#xed;nez-Garc&#xed;a et&#xa0;al., 2023</xref>). There are additional contextual advantages such as avoidance of the risk of repetition of research and expedited literature gathering (<xref ref-type="bibr" rid="B14">Garabedian et&#xa0;al., 2022</xref>). FAIR compliance or open-data availability requirements for scientific research is gradually becoming crucial. There are requirements set out by funding agencies and journals to make data open source. Advanced planning for FAIR set-up is crucial to keep costs and implementation durations at minimal. Retrospective FAIRification processes, such as in pharmaceutical research and development (R&amp;D) departments, have been show to entail significant costs (<xref ref-type="bibr" rid="B3">Alharbi et&#xa0;al., 2021</xref>). Assuming a 2.5% cost of total project budget for FAIR implementation, it has been shown to save around &#x20ac;2.6 billion per year for EU Horizon 2020 projects (<xref ref-type="bibr" rid="B13">European Commission, 2018</xref>). It is therefore imperative to incentivize and initiate FAIR deployment at the early stages of a project rather than towards the end of life of a project or after publication of research.</p>
</sec>
<sec id="s1_2">
<label>1.2</label>
<title>FAIR principles in the context of human microbiome data</title>
<p>Here, the focus is on application of the FAIR principles to human microbiome research. &#x201c;Microbiome data&#x201d; refers to the genetic material that is collected and analyzed from a community of microorganisms, such as bacteria, viruses, fungi, and other microbes, that live in a specific environment. Techniques such as microbiome shotgun sequencing is one of the methods to generate microbiome data. This involves breaking down the genetic material from all the microorganisms in a sample into small fragments, or reads. By comparing these reads to reference databases, researchers can identify and quantify the different microorganisms present in the sample, as well as determine the functions of the genes that are present. When looking at microbiome data, for which, in this case, the microbiome will be defined according to <xref ref-type="bibr" rid="B6">Berg et&#xa0;al. (2020)</xref> as &#x201c;all of the microbial components in a given ecosystem or plant, animal or human system&#x201d;, a few issues arise with FAIR compliance. There are microbiome databases such as National Microbiome Data Collaborative (NMDC) (<xref ref-type="bibr" rid="B12">The National Microbiome Data Collaborative Data Portal: an integrated multi-omics microbiome data resource, 2022</xref>) that focus on providing a FAIR-adherent platform for storage of microbiome meta data. Microbiome data from non-human systems, such as plant-associated or aquatic ecosystems, which are not bound by intellectual property (IP) and ethical considerations, could be readily uploaded to these platforms with minimal filtering or data clean-up. However, the microbiome (meta)data derived from humans, contains highly confidential information such as fragments of human DNA. Consequently, it is not permissible to disclose this data to the general public (<xref ref-type="bibr" rid="B18">Irving and Clarke, 2019</xref>). If this pseudonymized or anonymized data were to be made public and combined with the metadata of the dataset, which includes information regarding for example the age, sex and (approximate) location of the donor, it could lead to an intrusion on the privacy of the donor, thus violating the GDPR (<xref ref-type="bibr" rid="B15">G&#xfc;rsoy et&#xa0;al., 2022</xref>). To enable open data publications, human (host) DNA data scrubbing tools have been proposed. However, such tools do not remove all human host DNA. Additionally, during this process, the tools may incorrectly remove some non-host DNA data. Bad actors could potentially use this partly scrubbed datasets and their derivatives to orchestrate serious privacy violations. For example, the identity of the donor could be derivable, and private medical information could be traceable, such as possible health risks and genetic mutations (<xref ref-type="bibr" rid="B18">Irving and Clarke, 2019</xref>).</p>
<p>Development of a FAIR database for human microbiome data begins with establishment of unique identifiers. An object&#x2019;s unique identifier in a microbiome dataset could be a pseudonymized identifier of a participant or a sample collection location, so that longitudinal data can be traced in the metadata fields. These unique identifiers are essential and can therefore not be redacted. <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> summarizes the FAIR requirements for microbiome data and practical difficulties associated with implementing such a system with complete data transparency. A general question when comparing the suggestions for data transparency, as proposed by the FAIR principles, and the current GDPR privacy and data legislation, is whether the two can coexist in a world where reuse and transparency of scientific data is fully optimized, and the privacy of donors is guaranteed. This means that the GDPR must be satisfied prior to implementing FAIR principles. Here, we report on development and use of computational and data management tools for striking a balance on implementing the various FAIR principles without violating the privacy regulations.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>A summary of FAIR principles with respect to microbiome data. The end-dotted connections to microbiome data indicate the FAIR requirements and the flathead connections indicate the practical considerations inhibiting FAIR deployment.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1384809-g001.tif"/>
</fig>
</sec>
</sec>
<sec id="s2">
<label>2</label>
<title>Methods and Protocols</title>
<p>This section presents the protocols and tools we have designed for use in the development of FAIR-complaint database for human microbiome data. The protocols, codes and algorithms are available through GitHub at <ext-link ext-link-type="uri" xlink:href="https://github.com/SheratonMV/FAIRDatabase">https://github.com/SheratonMV/FAIRDatabase</ext-link>.</p>
<sec id="s2_1">
<label>2.1</label>
<title>Database development</title>
<p>To successfully apply the FAIR principles to a database, the type of database used is crucial. Microbiome data collection and storage is a continuous process. This necessitates the database to be real-time, so the data could be synchronized at all times. Any new additions or changes done at sample collection level or processing level would therefore immediately be reflected in the database, thus ensuring accurate data for all users at all times. Next, to be FAIR-compliant the database system itself should be open source. We rely on Supabase, an open-source, real-time, relational database, which suits our microbiome data well.</p>
<p>Since we handle sensitive (identifiable) information, the database was deployed locally as to handle data in accordance with GDPR guidelines. Additionally, we created a user interface to access the database and upload new data. The database is built with Python programming language (v3.10) and the corresponding supabase_py (v0.02) module (<xref ref-type="bibr" rid="B26">Supabase Inc,</xref>). Supabase offers different authentication options for database access. For our microbiome database, users can register with an email address and password and later login with same credentials and additional security measures such as two factor authorization or Single Sign-On (SSO). The login and registration are handled by Supabase functions, which are linked to dedicated authentication for the tables within the database. After signing up, the user does not get access to all functionalities (e.g. uploading is not possible for a standard user, and data collected by an uploader may not be available for other uploaders), and access rights can be modified by the database moderator. Such restriction ensures that raw data entry is carried out only at the sample collection point and prohibits data corruption or manipulation by an intermediate user or the uploader themself.</p>
<p>Unfortunately, Supabase currently neither offers table creation outside of their own SQL-editor nor direct SQL querying. To enable users to upload data in the database, the Python script for the user interface had to be linked to a local SQL editor, which can communicate with the database via SQL scripts directly. For this connection, the open source Psycopg2 (version 2.9.6) package was used (<xref ref-type="bibr" rid="B27">Varrazzo,</xref>). This package offers direct connection to the database and can execute SQL queries. The file uploaded is checked for the right format, and then stored into the database. If the format does not comply with FAIR principles, the file is first converted to complaint formats for data and metadata. Currently, the table creation is limited by a soft lock on the Supabase database, to1664 columns per table. In our database, 500 columns will be set as a limit to an uploaded table. The remaining available columns could later be used to add foreign keys to establish relationship between various tables. If the table consists of more columns, the table will be split up in multiple tables. The uploaded tables can be previewed via the user interface. This preview shows the first 15 rows and 10 columns to give the user a visual idea of what the table contains, as well as an overview of the metadata of the table contents. The entire table or associated datasets can be downloaded in CSV format, in accordance with the accessibility principle. To find, filter and carry out privacy metric calculations on the data, based on fields (column names) or metadata, the user can also query the database on certain columns (i.e. DNA sequences) and find corresponding datasets with the specified (pseudo- or anonymized) data.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Data pipeline</title>
<p>
<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> shows the complete ten-step data pipeline, starting from the sample collection from the donors and ending at the end-user. The donor dataset containing microbiome data (metagenomics shotgun data) is first filtered to remove all identifiable human (DNA) data using host contamination removal tool such as HoCoRT (<xref ref-type="bibr" rid="B23">Rumbavicius et&#xa0;al., 2023</xref>). The samples are then pseudo- or anonymized, depending on the data handling requirements, followed by removal of non-unique and low-quality reads (<xref ref-type="bibr" rid="B23">Rumbavicius et&#xa0;al., 2023</xref>). This process is described in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> at Step 3. Once this is complete, the data is imported from the environment of the provider (shown as the top rounded square) into our system (shown as the bottom rounded square). Upon entering our environment, and prior to data insertion into the database, a second round of quality control is performed. Here, as seen in Step 6, the data is filtered, with the aim of preventing any sensitive reads accidentally entering our database including checks with the data privacy module. It is also pseudo- or anonymized for a second time, so that, in the case of a leak at our provider, there is no connection to our version of the data, or vice versa. Step 6 is crucial to guaranteeing compliance with GDPR data regulations. Once this step has been completed, the data is inserted into the database, where it is stored and integrated with existing data entries, based on the unique subject identifier, which will be used as the primary key (step 7).</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Human microbiome data collection workflow and data transfer pipeline from the donors to end users. The raw data at step 2 includes human shotgun data, metadata and host-associated data such as age, ethnicity, health status.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1384809-g002.tif"/>
</fig>
<p>Once the data has been entered into the database, it can be queried by the database users, subject to permissions received based on their authorization level. Once signed in, a user can query and view data, as shown in Step 8 and 9 of <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>. Users have varying abilities to retrieve and view data based on their authorization levels, with specific examples being demonstrated in steps 9 and 10 using authorization levels A and B respectively. Thus, in the developed pipeline, microbiome data could be anonymized at three stages, 1. Filtering and removing donor&#x2019;s personal data (association) from the sequencing data with standardized keys at data provider site 2. Removal of human DNA data from metagenomics shotgun data using host contamination removal tool. And 3. Secondary anonymization or pseudonymization at the FAIR database environment (site).</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Large language model deployment</title>
<p>Utilizing Supabase&#x2019;s inbuilt postgres vector database and AI toolkit (<xref ref-type="bibr" rid="B26">Supabase Inc,</xref>), we developed an interface to interact with the database relying on a large language model (LLM). The current implementation relies on scaffolding an edge function to forward query to ChatGPT&#x2019;s API access. The LLM vendor or software can be switched by changing the API. Thus, the LLM could be locally hosted and queried from the database, in the future, provided sufficient computing resources are available.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Throughput results</title>
<p>The database developed in this work was tested for its throughput capabilities. To carry out the tests, we generated CSV files containing values with various number of rows and columns. The synthetic data was generated to closely resemble the count matrices in microbiome data. The data upload results show that uploading a file to the database is mostly a linear relation with regards to the number of rows and columns (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>).</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Table upload speeds for varying number of columns and rows, <bold>(A)</bold> for different fixed number of rows and <bold>(B)</bold> for different fixed number of columns. The colors refer to the fixed number of columns and rows respectively.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1384809-g003.tif"/>
</fig>
<p>As observed in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>, the time to upload a file roughly doubles with doubling of the column size, for total number of columns less than 10,000. However, for tables containing more than 10,000 columns, the upload time fluctuates depending on what the total size is. As stated above, in Supabase, there is maximum of 1664 columns per table. For practical purposes, for Postgres tables, such as in Supabase, a maximum of 2 (<xref ref-type="bibr" rid="B17">Huttenhower et&#xa0;al., 2023</xref>) (approximately 4 billion) rows could be stored in a table. Tables with more than 1664 columns were split into multiple sub-tables. These sub-tables can then be queried as single large table by establishing a common key, in our case sample donor id, for establishing the relation between the tables. For this reason, the column splitting limit per table was set to 500 for sub-table creation and leaving space for columns to be added later to the table. We did not observe significant upload time deviations based on rows and columns count combinations, for instance 3000 rows and 4000 columns table upload speed took almost equal time as 4000 rows and 3000 columns table upload, approximately 73 seconds.</p>
<p>
<xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4A, B</bold>
</xref> show the speed of retrieving data from the tables. To estimate this retrieval performance, a table and its related tables were looked up in the database and all rows corresponding to the tables were retrieved with an SQL query and later downloaded as csv file. The retrieval times show an approximately linear trend between the number of rows or columns queried and time for retrieval.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Table retrieval speeds for varying number of columns and rows, <bold>(A)</bold> for different fixed number of rows and <bold>(B)</bold> for different fixed number of columns. The colors refer to the fixed number of columns and rows respectively.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1384809-g004.tif"/>
</fig>
</sec>
</sec>
<sec id="s3" sec-type="discussion">
<label>3</label>
<title>Discussions</title>
<p>In this article, we designed and built a FAIR-compliant database for storage of microbiome data. The FAIR principles were applied to the database platform itself as well as the data. Due to privacy-centric sensitivities associated with microbiome data, it is not possible to implement the FAIR principles in a strict manner, as that would require total transparency of data. As this database could include human data (DNA) and metadata, unfiltered publication of this data will both violate the privacy of the sample donors, and breach GDPR. The database was built using the open source Supabase platform. If authorized, users of the database can access, upload and download datasets. As shown in the results section, the linear scaling relation for data transfer speeds suggests that the database is capable of handling large volumes of data efficiently and proportionally to the underlying computing resources (computational power and storage). Two or more different data entries which were taken from the same donor by the same entity, can be either linked in relational datasets, or appended to one another. At the moment, due to Supabase functionality limits, appending data would first require the manual addition of the new columns to the existing table, before the data can be inserted. The main requirement for this functionality is for the providing entity to standardize its anonymization key (<xref ref-type="bibr" rid="B1">Abouelmehdi et&#xa0;al., 2018</xref>), thus giving each individual donor a unique identifier. This identifier can then be used across multiple sampling instances and multiple datasets, throughout time, to identify and group data entries per individual sample. The probability of a single sample having multiple measurements in our database is minimized. The only way this could be bypassed is if at a later date, the database would receive datasets from multiple uploading entities providing data from a single donor, then the donor would have had samples taken at more than one of those entities as duplicates. In that case, the donor would get a different unique identifier at each entity, leading to them not matching when all the data is combined in the database.</p>
<p>It can be concluded that the implementation of the FAIR principles in microbiome data research is possible to some extent, however, not completely. Other obstacles that we encountered were on the extent of the database implementation protocols as modules and the side effects that accompany it. Implementing these compliance principles can take a lot of time and thus can be expensive in terms of personnel hours. This comes hand in hand with technical expertise required to handle microbiome data, as the principles require all (meta)data to be standardized with Controlled Vocabulary Terms (CVT) from appropriate MIxS standards (<xref ref-type="bibr" rid="B31">Yilmaz et&#xa0;al., 2011</xref>) (minimum information about any (x) sequence), formatted and managed with regards to access and privacy. These obstacles have meant that the FAIR principles could not be fully implemented in their original essence, however, the remaining guidelines were still of great influence on the design of database. To overcome the above discussed constraints, we have provided three major modules in this work to adhere to the mandatory privacy principles and facilitate FAIR compliance.</p>
<sec id="s3_1">
<label>3.1</label>
<title>FAIR data format compliance module</title>
<p>Microbiome shotgun data file formats, such as sequence files, FASTA or FASTQ, or count data, being plain-text format files, are not inherently FAIR compliant by themselves. It is therefore necessary to make them FAIR compliant by converting to universal Column Separated Values (CSV) format. The converter module converts such files into FAIR-complaint &#x2018;.csv&#x2019; files containing appropriate data tables and a separate metadata &#x2018;.txt&#x2019; files. During the data upload process, the database automatically scans the file types uploaded and raises appropriate error messages if the upload file is not strictly FAIR adherent. By creating comprehensive metadata about (or from) the dataset and strictly following the CSV format for data structure, both the metadata and data were made interoperable. This system allows the researchers to effortlessly input new data and revise current records in real-time and enables seamless querying, downloading, exhibition, and merging of the datasets with other standardized external sources.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Data privacy module</title>
<p>The data privacy module in our database functions in two distinct manners, (i) generation of privacy metrics and (ii) integration pipeline for federated learning. In cases of datasets containing social or demographic data of the microbiome data donors, the privacy module can be deployed to calculate privacy metrics, such as the entropy measure (to calculate the information predictability of data) or L-diversity score (<xref ref-type="bibr" rid="B19">Machanavajjhala et&#xa0;al., 2007</xref>). This way uploaded data that is personally identifiable can be easily identified. This can be later rectified by methods such as k-anonymity (<xref ref-type="bibr" rid="B21">Mayer et&#xa0;al., 2023</xref>) and &#x3b5;- differential privacy (<xref ref-type="bibr" rid="B11">Dong et&#xa0;al., 2022</xref>). In our privacy module, we have provided an implementation of k-anonymity measure on the data. Additional functions can be easily added to this module at the data filtering step, if needed.</p>
<p>There may arise situations where access to sensitive data may be necessary for completion of research objective. In such cases, federated learning could potentially mitigate the missing data situation. In federated learning, no sensitive data or metadata is shared with an end-user but rather all the computations required by the third-party or end user are carried out locally, this process is visually represented in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5A</bold>
</xref>. Only the results from this analysis will be shared with the end-user. A couple of steps are required to implement such a system to ensure that the results shared with the end-user are devoid of any unintended data leak. First, the output structure should be known beforehand, even before running any sensitive data analysis. Second, ensure that sensitive data cannot be reconstructed from the results. One way to pre-generate output against an end-user shared algorithm or program is to use synthetic microbiome data (<xref ref-type="bibr" rid="B16">Hittmeir et&#xa0;al., 2022</xref>). This enables an initial mock-run of the analysis to check against output structure and the results or graphs generated. Some outlier data points such as minority demographics could still be traceable in the generated results. To alleviate any further issues arising from identifiable data being present from the processed outputs, it is necessary to analyze the results before issuing them to the end-user. This can either be done by filtering the input with our privacy metric calculation module and/or by investigating only with algorithms compliant with privacy preserving analysis techniques (<xref ref-type="bibr" rid="B15">G&#xfc;rsoy et&#xa0;al., 2022</xref>). To ensure data security, we provide a modular pipeline that enables the application of delegated learning techniques without revealing the confidential information to the user. Our system integrates this process directly into the database.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>
<bold>(A)</bold> Pipeline of federated learning from two different databases that are not connected with each other. <bold>(B)</bold> Deployment of Large Language Model with Supabase via edge functions.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcimb-14-1384809-g005.tif"/>
</fig>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Accessibility enrichment module</title>
<p>Making scientific research findings accessible and understandable to the general public is a crucial component of FAIR data availability. Large language models can play a key role in facilitating this process by transforming complex scientific information into more relatable formats. These models can assist in creating educational materials, generating reports for policymakers, or answering questions from journalists and the media, ultimately improving knowledge dissemination on various levels within society. To ensure such accessibility at all levels of research, in this study, we have deployed a LLM (ChatGPT-based) via Supabase edge functions (<xref ref-type="bibr" rid="B8">Cao et&#xa0;al., 2020</xref>) to interpret data or findings (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5B</bold>
</xref>). Such an LLM, even if trained only against published literature from a study would enable broader knowledge dissemination to people of all strata, for instance could help answer a high school student&#x2019;s question. This approach ensures FAIR is not restricted to scientific researcher, but could be widely used by non-researchers alike.</p>
<p>It is necessary to acknowledge the current lack of direct incentives to promote FAIR among researchers. <xref ref-type="bibr" rid="B17">Huttenhower et&#xa0;al. (2023)</xref> have provided some excellent pointers to incentivize FAIRification of data. Some include financial incentives such as prioritizing research proposals with FAIR compliance and penalizing others could promote FAIR culture among researchers. Even though they propose full deposition for publications, it may not be straight forward in countries where data privacy laws stifle FAIR principles. Workarounds such as the ones described in this manuscript should be considered to promote egalitarian scientific research irrespective of legal privacy considerations. Finally, the current study is limited to dealing with common datatypes and descriptors found in microbiome data. In presence of additional datatypes or descriptors, it may be necessary to implement additional privacy preservation metrics to ensure the data is non-identifiable. This is also true for validating outputs generated from Federated learning approaches. Here, the outputs could vary vastly based on the algorithms used. Even if the outputs themselves do not make the data identifiable, combining them could render it identifiable. Therefore, necessary algorithm screening techniques will need to be developed and implemented.</p>
</sec>
</sec>
<sec id="s4" sec-type="conclusions">
<label>4</label>
<title>Conclusions</title>
<p>Our considerations and resulting procedures open opportunities for future thinking and research on conflicts between FAIR and privacy laws (GDPR). One concept that could use exploration, is the sharing of anonymization keys between healthcare data providers. This would mean that data from two different entities on the same patient could be merged, possibly creating new connections, trends and scientific findings. Naturally, this also brings along new concerns for privacy and data security, however, seeing as patient data is already being shared between the world&#x2019;s healthcare providers, it could be a logical step to share patient&#x2019;s anonymized aliases as well. A second field of interest is how the research in this manuscript relates to the rest of the world. In Europe, on which this work is based, GDPR strictly govern data privacy and data use. In other parts of the world, rules and regulations are different, meaning that if this research had been conducted somewhere else, it could have led to different procedures. The need for a harmonized global view on data sharing in healthcare research is underscored by the upcoming efforts in this field in Asia and Africa. Sharing healthcare data beyond national and continental borders will greatly advance global healthcare research, especially in the new territory of microbiome and genomic data research. An exploration of these global cooperative efforts, including the digital and legal infrastructure needed to support them, could be a catalyst for new scientific advancements and findings. In future, collaborations between healthcare professionals, computational modelers, data scientists and AI experts can be crucial in maximizing the potential of data and technology in healthcare. Therefore, it is important to promote interdisciplinary education and training programs that bring together healthcare professionals, data scientists, and AI experts. The user friendly nature of the framework developed in this manuscript should help overcome technical difficulties associated with using FAIR microbiome data. However, it is still crucial to ensure that future physicians are better prepared to leverage these tools in delivering patient care.</p>
</sec>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material. Further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>MD: Investigation, Methodology, Software, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. NZ: Data curation, Investigation, Methodology, Software, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. RW: Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. DM: Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. BB: Methodology, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. EZ: Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. AH: Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. VS: Conceptualization, Data curation, Investigation, Methodology, Software, Supervision, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing.</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This publication is part of the project &#x201c;METAHEALTH: Health in a microbial, sociocultural and care context in the first 1000 days of life&#x201d; (project number NWA.1389.20.080) of the research program Dutch Research Agenda - Research along Routes by Consortia (NWA-ORC) 2020/21 which is (partly) financed by the Dutch Research Council (NWO).</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>Author RW is employed by the company Supabase LLC.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<title>Abbreviations</title>
<fn fn-type="abbr">
<p>AI, Artificial Intelligence; API, Application Programming Interface; CSV, Comma Separated Values; CVT, Controlled Vocabulary Terms; DNA, Deoxy Ribonucleic Acid; FAIR, Findable. Accessible, Interoperable and Reusable; GDPR, General Data Protection Regulation; IP, Intellectual Property; LLM, Large Language Model; NMDC, National Microbiome Data Collaborative; MIxS, Minimum Information about any (x) Sequence; SQL, Structured Query Language; SSO, Single Sign-On.</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abouelmehdi</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Beni-Hessane</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Khaloufi</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Big healthcare data: preserving security and privacy</article-title>. <source>J. Big Data</source> <volume>5</volume>, <fpage>1</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s40537-017-0110-7</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abuimara</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Hobson</surname> <given-names>B. W.</given-names>
</name>
<name>
<surname>Gunay</surname> <given-names>B.</given-names>
</name>
<name>
<surname>O&#x2019;Brien</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A data-driven workflow to improve energy efficient operation of commercial buildings: A review with real-world examples</article-title>. <source>Building Serv. Eng. Res. Technol.</source> <volume>43</volume>, <fpage>517</fpage>&#x2013;<lpage>534</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1177/01436244211069655</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alharbi</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Skeva</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Juty</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Jay</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Goble</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Exploring the current practices, costs and benefits of FAIR implementation in pharmaceutical research and development: A qualitative interview study</article-title>. <source>Data Intell.</source> <volume>3</volume>, <fpage>507</fpage>&#x2013;<lpage>527</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1162/dint_a_00109</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Baglamis</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Saha</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Heijden der van</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Miedema</surname> <given-names>D. M.</given-names>
</name>
<name>
<surname>Gent van</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Krawczyk</surname> <given-names>P. M.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). &#x201c;<article-title>A novel high-throughput framework to quantify spatio-temporal tumor clonal dynamics</article-title>,&#x201d; in <source>Computational science &#x2013; ICCS 2023</source>. Ed. <person-group person-group-type="editor">
<name>
<surname>Miky&#x161;ka</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group> (<publisher-name>Springer Nature Switzerland</publisher-name>, <publisher-loc>Cham</publisher-loc>), <fpage>10475 345</fpage>&#x2013;<lpage>359</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>B&#xe9;quignon</surname> <given-names>O. J. M.</given-names>
</name>
<name>
<surname>Bongers</surname> <given-names>B.J.</given-names>
</name>
<name>
<surname>Jespers</surname> <given-names>W.</given-names>
</name>
<name>
<surname>IJzerman</surname> <given-names>A. P.</given-names>
</name>
<name>
<surname>Water der van</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Westen van</surname> <given-names>G.J.P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Papyrus: a large-scale curated dataset aimed at bioactivity predictions</article-title>. <source>J. Cheminform</source> <volume>15</volume>, <fpage>3</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13321-022-00672-x</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Berg</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Rybakova</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Fischer</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Cernava</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Verg&#xe8;s Champomier</surname> <given-names>M.-C.</given-names>
</name>
<name>
<surname>Charles</surname> <given-names>T.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Microbiome definition re-visited: old concepts and new challenges</article-title>. <source>Microbiome</source> <volume>8</volume>, <fpage>103</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s40168-020-00875-0</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bray</surname> <given-names>M.-A.</given-names>
</name>
<name>
<surname>Gustafsdottir</surname> <given-names>S. M.</given-names>
</name>
<name>
<surname>Rohban</surname> <given-names>M. H.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ljosa</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Sokolnicki</surname> <given-names>K. L.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). <article-title>A dataset of images and morphological profiles of 30 000 small-molecule treatments using the Cell Painting assay</article-title>. <source>GigaScience</source> <volume>6</volume>, <elocation-id>giw014</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/gigascience/giw014</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Meng</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>An overview on edge computing research</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>85714</fpage>&#x2013;<lpage>85728</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/Access.6287639</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chue Hong</surname> <given-names>N. P.</given-names>
</name>
<name>
<surname>Katz</surname> <given-names>D. S.</given-names>
</name>
<name>
<surname>Barker</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Lamprecht</surname> <given-names>A.-L.</given-names>
</name>
<name>
<surname>Martinez</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Psomopoulos</surname> <given-names>F. E.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <source>FAIR principles for research software (FAIR4RS principles)</source> (<publisher-loc>Research Data Alliance (RDA)</publisher-loc>: <publisher-name>United Kingdom</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.15497/RDA00068</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Da Silva Santos</surname> <given-names>L. O. B.</given-names>
</name>
<name>
<surname>Burger</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Kaliyaperumal</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Wilkinson</surname> <given-names>M. D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>FAIR data point: A FAIR-oriented approach for metadata publication</article-title>. <source>Data Intell.</source> <volume>5</volume>, <fpage>163</fpage>&#x2013;<lpage>183</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1162/dint_a_00160</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Roth</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Su</surname> <given-names>W. J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Gaussian differential privacy</article-title>. <source>J. R. Stat. Soc. Ser. B: Stat. Method.</source> <volume>84</volume>, <fpage>3</fpage>&#x2013;<lpage>37</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/rssb.12454</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eloe-Fadrosh</surname> <given-names>E. A.</given-names>
</name>
<name>
<surname>Ahmed</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Anubhav, Babinski</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Baumes</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Borkum</surname> <given-names>M.</given-names>
</name>
<etal/></person-group> (<year>2022</year>). <article-title>The National Microbiome Data Collaborative Data Portal: an integrated multi-omics microbiome data resource</article-title>. <source>nat</source>. <volume>50</volume> (<issue>D1</issue>), <fpage>D828</fpage>&#x2013;<lpage>D836</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkab990</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="book">
<person-group person-group-type="author">
<collab>European Commission</collab>
</person-group> (<year>2018</year>). &#x201c;<article-title>Directorate general for research and innovation. &amp; PwC EU services</article-title>,&#x201d; in <source>Cost-benefit analysis for FAIR research data: cost of not having FAIR research data</source> (<publisher-loc>Luxemburg</publisher-loc>: <publisher-name>Publications Office, LU</publisher-name>).</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Garabedian</surname> <given-names>N. T.</given-names>
</name>
<name>
<surname>Schreiber</surname> <given-names>P. J.</given-names>
</name>
<name>
<surname>Brandt</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Zschumme</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Blatter</surname> <given-names>I. L.</given-names>
</name>
<name>
<surname>Dollmann</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Generating FAIR research data in experimental tribology</article-title>. <source>Sci. Data</source> <volume>9</volume>, <fpage>315</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41597-022-01429-9</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>G&#xfc;rsoy</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ni</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Brannon</surname> <given-names>C. M.</given-names>
</name>
<name>
<surname>Gerstein</surname> <given-names>M. B.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Functional genomics data: privacy risk assessment and technological mitigation</article-title>. <source>Nat. Rev. Genet.</source> <volume>23</volume>, <fpage>245</fpage>&#x2013;<lpage>258</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41576-021-00428-7</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hittmeir</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Mayer</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Ekelhart</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Utility and privacy assessment of synthetic microbiome data</article-title>,&#x201d; in <source>Data and applications security and privacy XXXVI</source>. Eds. <person-group person-group-type="editor">
<name>
<surname>Sural</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>H.</given-names>
</name>
</person-group> (<publisher-name>Springer International Publishing</publisher-name>, <publisher-loc>Cham</publisher-loc>), <volume>13383</volume>, <fpage>15</fpage>&#x2013;<lpage>27</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huttenhower</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Finn</surname> <given-names>R. D.</given-names>
</name>
<name>
<surname>McHardy</surname> <given-names>A. C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Challenges and opportunities in sharing microbiome data and analyses</article-title>. <source>Nat. Microbiol.</source> <volume>8</volume>, <fpage>1960</fpage>&#x2013;<lpage>1970</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41564-023-01484-x</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Irving</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Clarke</surname> <given-names>A. J.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Ethical and social issues in clinical genetics</article-title>,&#x201d; in <source>Emery and rimoin&#x2019;s principles and practice of medical genetics and genomics</source> (<publisher-loc>Netherlands</publisher-loc>: <publisher-name>Elsevier</publisher-name>), <fpage>327</fpage>&#x2013;<lpage>354</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/B978-0-12-812536-6.00013-4</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Machanavajjhala</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Kifer</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Gehrke</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Venkitasubramaniam</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>
<italic>L</italic> -diversity: privacy beyond</article-title>. <source>k -anonymity. ACM Trans. Knowl. Discovery Data</source> <volume>1</volume>, <fpage>3</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/1217299.1217302</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mart&#xed;nez-Garc&#xed;a</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Alvarez-Romero</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Rom&#xe1;n-Villar&#xe1;n</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Bernabeu-Wittel</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Luis Parra-Calder&#xf3;n</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>FAIR principles to improve the impact on health research management outcomes</article-title>. <source>Heliyon</source> <volume>9</volume>, <elocation-id>e15733</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.heliyon.2023.e15733</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Mayer</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Karlowicz</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Hittmeir</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>k-anonymity on metagenomic features in microbiome databases</article-title>,&#x201d; in <conf-name>ARES '23: Proceedings of the 18th International Conference on Availability, Reliability and Security</conf-name> (<publisher-loc>Benevento Italy</publisher-loc>: <publisher-name>ACM</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1145/3600160.3600178</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roche</surname> <given-names>D. G.</given-names>
</name>
<name>
<surname>Kruuk</surname> <given-names>L. E. B.</given-names>
</name>
<name>
<surname>Lanfear</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Binning</surname> <given-names>S. A.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Public data archiving in ecology and evolution: how well are we doing</article-title>? <source>PloS Biol.</source> <volume>13</volume>, <elocation-id>e1002295</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pbio.1002295</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rumbavicius</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Rounge</surname> <given-names>T. B.</given-names>
</name>
<name>
<surname>Rognes</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>HoCoRT: host contamination removal tool</article-title>. <source>BMC Bioinf.</source> <volume>24</volume>, <fpage>371</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12859-023-05492-w</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sheraton</surname> <given-names>M. V.</given-names>
</name>
<name>
<surname>Melnikov</surname> <given-names>V. R.</given-names>
</name>
<name>
<surname>Sloot</surname> <given-names>P. M. A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Prediction and quantification of bacterial biofilm detachment using Glazier&#x2013;Graner&#x2013;Hogeweg method based model simulations</article-title>. <source>J. Theor. Biol.</source> <volume>482</volume>, <fpage>109994</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jtbi.2019.109994</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Subramanian</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Verma</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Jere</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Anamika</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Multi-omics data integration, interpretation, and its application</article-title>. <source>Bioinform. Biol. Insights</source> <volume>14</volume>, <fpage>117793221989905</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1177/1177932219899051</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="web">
<person-group person-group-type="author">
<collab>Supabase Inc</collab>
</person-group> <article-title>Supabase vector database and AI toolkit</article-title>. Available online at: <uri xlink:href="https://github.com/supabase/supabase/tree/master/apps/docs/pages/guides/ai">https://github.com/supabase/supabase/tree/master/apps/docs/pages/guides/ai</uri>.</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Varrazzo</surname> <given-names>D</given-names>
</name>
</person-group>. <source>Psycopg &#x2013; PostgreSQL database adapter for Python</source>. Available at: <uri xlink:href="https://www.psycopg.org/docs/">https://www.psycopg.org/docs/</uri>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wilding</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Sheraton</surname> <given-names>V. M.</given-names>
</name>
<name>
<surname>Soto</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Chotai</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>E. Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Deep learning applied to breast imaging classification and segmentation with human expert intervention</article-title>. <source>J. Ultrasound</source> <volume>25</volume>, <fpage>659</fpage>&#x2013;<lpage>666</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s40477-021-00642-3</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wilkinson</surname> <given-names>M. D.</given-names>
</name>
<name>
<surname>Dumontier</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Aalbersberg</surname> <given-names>I. J.</given-names>
</name>
<name>
<surname>Appleton</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Axton</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Baak</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2016</year>). <article-title>The FAIR Guiding Principles for scientific data management and stewardship</article-title>. <source>Sci. Data</source> <volume>3</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>9</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/sdata.2016.18</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Willemink</surname> <given-names>M. J.</given-names>
</name>
<name>
<surname>Koszek</surname> <given-names>W. A.</given-names>
</name>
<name>
<surname>Hardell</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Fleischmann</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Harvey</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Preparing medical imaging data for machine learning</article-title>. <source>Radiology</source> <volume>295</volume>, <fpage>4</fpage>&#x2013;<lpage>15</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1148/radiol.2020192224</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yilmaz</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Kottmann</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Field</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Knight</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Cole</surname> <given-names>J. R.</given-names>
</name>
<name>
<surname>Amaral-Zettler</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2011</year>). <article-title>Minimum information about a marker gene sequence (MIMARKS) and minimum information about any (x) sequence (MIxS) specifications</article-title>. <source>Nat. Biotechnol.</source> <volume>29</volume>, <fpage>415</fpage>&#x2013;<lpage>420</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nbt.1823</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yoong</surname> <given-names>S. L.</given-names>
</name>
<name>
<surname>Turon</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Grady</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Hodder</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Wolfenden</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>The benefits of data sharing and ensuring open sources of systematic review data</article-title>. <source>J. Public Health</source> <volume>44</volume>, <fpage>e582</fpage>&#x2013;<lpage>e587</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/pubmed/fdac031</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>