<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="review-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Drug Discov.</journal-id>
<journal-title>Frontiers in Drug Discovery</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Drug Discov.</abbrev-journal-title>
<issn pub-type="epub">2674-0338</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1222655</article-id>
<article-id pub-id-type="doi">10.3389/fddsv.2023.1222655</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Drug Discovery</subject>
<subj-group>
<subject>Review</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Yin-yang in drug discovery: rethinking <italic>de novo</italic> design and development of predictive models</article-title>
<alt-title alt-title-type="left-running-head">Ch&#xe1;vez-Hern&#xe1;ndez et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fddsv.2023.1222655">10.3389/fddsv.2023.1222655</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Ch&#xe1;vez-Hern&#xe1;ndez</surname>
<given-names>Ana L.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1962271/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>L&#xf3;pez-L&#xf3;pez</surname>
<given-names>Edgar</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1573790/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Medina-Franco</surname>
<given-names>Jos&#xe9; L.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/415613/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Pharmacy</institution>, <institution>DIFACQUIM Research Group</institution>, <institution>School of Chemistry</institution>, <institution>Universidad Nacional Aut&#xf3;noma de M&#xe9;xico</institution>, <institution>Avenida Universidad</institution>, <addr-line>Mexico City</addr-line>, <country>Mexico</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Chemistry and Graduate Program in Pharmacology</institution>, <institution>Center for Research and Advanced Studies of the National Polytechnic Institute</institution>, <addr-line>Mexico City</addr-line>, <country>Mexico</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/106271/overview">Ho Leung Ng</ext-link>, Atomwise Inc., United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/530935/overview">Rodolpho C. Braga</ext-link>, InsilicAll, Brazil</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1746083/overview">Jeremy J. Yang</ext-link>, University of New Mexico, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Jos&#xe9; L. Medina-Franco, <email>medinajl@unam.mx</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>21</day>
<month>06</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>3</volume>
<elocation-id>1222655</elocation-id>
<history>
<date date-type="received">
<day>15</day>
<month>05</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>12</day>
<month>06</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Ch&#xe1;vez-Hern&#xe1;ndez, L&#xf3;pez-L&#xf3;pez and Medina-Franco.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Ch&#xe1;vez-Hern&#xe1;ndez, L&#xf3;pez-L&#xf3;pez and Medina-Franco</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Chemical and biological data are the cornerstone of modern drug discovery programs. Finding qualitative yet better quantitative relationships between chemical structures and biological activity has been long pursued in medicinal chemistry and drug discovery. With the rapid increase and deployment of the predictive machine and deep learning methods, as well as the renewed interest in the <italic>de novo</italic> design of compound libraries to enlarge the medicinally relevant chemical space, the balance between quantity and quality of data are becoming a central point in the discussion of the type of data sets needed. Although there is a general notion that the more data, the better, it is also true that its quality is crucial despite the size of the data itself. Furthermore, the active versus inactive compounds ratio balance is also a major consideration. This review discusses the most common public data sets currently used as benchmarks to develop predictive and classification models used in <italic>de novo</italic> design. We point out the need to continue disclosing inactive compounds and negative data in peer-reviewed publications and public repositories and promote the balance between the positive (Yang) and negative (Yin) bioactivity data. We emphasize the importance of reconsidering drug discovery initiatives regarding both the utilization and classification of data.</p>
</abstract>
<kwd-group>
<kwd>big data</kwd>
<kwd>chemoinformatics</kwd>
<kwd>chemical libraries</kwd>
<kwd>data quality</kwd>
<kwd>
<italic>de novo</italic> design</kwd>
<kwd>drug discovery</kwd>
<kwd>machine learning</kwd>
<kwd>negative results</kwd>
</kwd-group>
<contract-sponsor id="cn001">Direcci&#xf3;n General de Asuntos del Personal Acad&#xe9;mico, Universidad Nacional Aut&#xf3;noma de M&#xe9;xico<named-content content-type="fundref-id">10.13039/501100006087</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>In silico Methods and Artificial Intelligence for Drug Discovery</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Data and the increasing role of predictive models, including machine and deep learning (<xref ref-type="bibr" rid="B77">Mouchlis et al., 2021</xref>; <xref ref-type="bibr" rid="B5">Bajorath et al., 2022</xref>), are the cornerstone of modern drug discovery programs (<xref ref-type="bibr" rid="B138">Zhang et al., 2022</xref>). The increasing use of computational methods that recently included deep learning is reducing the time and financial costs of finding drug candidates (<xref ref-type="bibr" rid="B138">Zhang et al., 2022</xref>). For instance, computer-aided drug design (CADD) has led to the discovery of more than seventy approved drugs (<xref ref-type="bibr" rid="B98">Sabe et al., 2021</xref>) including remdesivir as an emergency treatment against SARS-CoV-2 in 2021 (<xref ref-type="bibr" rid="B27">Dos Santos Nascimento et al., 2021</xref>).</p>
<p>CADD methods are typically divided into two main categories, structure-based drug design (SBDD) and ligand-based drug design (LBDD) that rely on the three-dimensional (3D) structure data available for one or more molecular targets, or the structure-activity data of ligands, respectively. Examples of deep learning applications in SBDD include AlphaFold to assist in homology modeling, and DiffDock in molecular docking. AlphaFold predicts 3D protein structures according to their amino acid sequences (<xref ref-type="bibr" rid="B47">Jumper et al., 2021</xref>), and DiffDock predicts the binding mode between the ligand and specific protein target (<xref ref-type="bibr" rid="B24">Corso et al., 2022</xref>). One of the most notable approaches in LBDD are quantitative structure-activity relationships (QSAR) (<xref ref-type="bibr" rid="B27">Dos Santos Nascimento et al., 2021</xref>). Current QSAR methods use machine learning and deep learning (<xref ref-type="bibr" rid="B114">Soares et al., 2022</xref>) that can be divided into linear methods and nonlinear methods (<xref ref-type="bibr" rid="B89">Patel et al., 2014</xref>; <xref ref-type="bibr" rid="B36">Greener et al., 2022</xref>). Linear methods include linear regression, multiple linear regression, partial least squares, and principal component analysis (<xref ref-type="bibr" rid="B89">Patel et al., 2014</xref>). Nonlinear methods include artificial neural networks, k-nearest neighbors, and Bayesian neural nets, to name a few examples (<xref ref-type="bibr" rid="B89">Patel et al., 2014</xref>; <xref ref-type="bibr" rid="B36">Greener et al., 2022</xref>).</p>
<p>Advances in deep learning models have a significant progress in molecule generation, representing a big step forward in bridging the gap between chemical entities and drug-like properties (<xref ref-type="bibr" rid="B56">Krishnan et al., 2021</xref>). Deep learning algorithms are currently used in the renewed interest in the <italic>de novo</italic> design of chemical libraries. In 2020, the successful application of deep learning in drug discovery, that included the <italic>de novo</italic> design using deep learning, was selected by the Massachusetts Institute of Technology Technology Review as one of the top ten breakthrough technologies (<xref ref-type="bibr" rid="B48">Juskalian et al., 2023</xref>).</p>
<p>
<italic>De novo</italic> design is aimed at generating new chemical entities (NCE) with desired properties (<xref ref-type="bibr" rid="B87">Palazzesi and Pozzan, 2022</xref>). <italic>De novo</italic> design based on deep learning algorithms (<xref ref-type="bibr" rid="B87">Palazzesi and Pozzan, 2022</xref>) requires a large number of compounds that may demand significant computational resources. However, bioactivity data for a biological endpoint is not always sufficient. The lack of data has led to the development of new methods for compound selection and applications for deep learning algorithms are being developed (<xref ref-type="bibr" rid="B40">Guo M et al., 2021</xref>).</p>
<p>Knowledge-based drug design frequently involves quality data (<xref ref-type="bibr" rid="B90">Perron et al., 2022b</xref>) to develop models with useful predictions (<xref ref-type="bibr" rid="B107">Schneider et al., 2020</xref>). To this end, rethinking the methodologies used for drug discovery and development campaigns is crucial. The quality of data sets, decoy data sets and inactive compounds used in predictive models, and <italic>de novo</italic> design models need to be reviewed and discussed.</p>
<p>The main purpose of this manuscript is discussing the importance of quality data, decoy data sets, and the balance needed between inactive (i.e., &#x201c;Yin&#x201d;) and active (&#x201c;Yang&#x201d;) compounds currently employed in <italic>de novo</italic> design and developing predictive models of biological activity to generate NCE. Following up on previous studies (<xref ref-type="bibr" rid="B107">Schneider et al., 2020</xref>; <xref ref-type="bibr" rid="B5">Bajorath et al., 2022</xref>; <xref ref-type="bibr" rid="B23">Cherkasov, 2023</xref>), we comment on the need to rethink the way to drug design and develop campaigns. The manuscript is organized into four main sections. After this Introduction, <xref ref-type="sec" rid="s2">Section 2</xref> presents an overview of <italic>de novo</italic> design. <xref ref-type="sec" rid="s3">Section 3</xref> discusses the main public data sources used to develop predictive models. <xref ref-type="sec" rid="s4">Section 4</xref> discusses criteria to generate quality data sets. The last section presents a summary of conclusions and perspectives.</p>
</sec>
<sec id="s2">
<title>2 <italic>De novo</italic> design overview</title>
<p>
<italic>De novo</italic> design aims to generate new chemical structures from scratch with desired predicted properties, e.g., absorption, distribution, metabolism, excretion, toxicity (ADMET), other drug-likeness properties, and biological activities (<xref ref-type="bibr" rid="B87">Palazzesi and Pozzan, 2022</xref>). The two main strategies for <italic>de novo</italic> design can be classified into SBDD and LBDD (<italic>vide supra</italic>) (<xref ref-type="bibr" rid="B138">Zhang et al., 2022</xref>). A recent example of a structured-based <italic>de novo</italic> design is the RELATION model that learns from the desired geometric features of protein-ligand complexes to generate new molecules (<xref ref-type="bibr" rid="B124">Wang et al., 2022</xref>). The generation process applies a fragment-based strategy given an initial chemical scaffold embedded in the binding site of the target protein. The pre-trained model generates molecules iteratively by sequentially adding, deleting, inserting, or replacing and linking fragments (<xref ref-type="bibr" rid="B138">Zhang et al., 2022</xref>).</p>
<p>In contrast, ligand-oriented <italic>de novo</italic> design focuses on the ligands themselves, thereby generating compounds with new chemical structures with novel scaffolds from active compounds while optimizing the desired properties (<xref ref-type="bibr" rid="B134">Xie et al., 2022</xref>). A general workflow is schematically summarized in <xref ref-type="fig" rid="F1">Figure 1</xref> which has seven main steps (<xref ref-type="bibr" rid="B56">Krishnan et al., 2021</xref>; <xref ref-type="bibr" rid="B138">Zhang et al., 2022</xref>): 1) Selecting compound data sets from public or in-house sources (further discussed in <xref ref-type="sec" rid="s3">Section 3</xref>); 2) Filtering molecular data sets with desired properties such as drug-likeness. In the example of <xref ref-type="fig" rid="F1">Figure 1</xref> a data set with three subsets of compounds is represented with a star, triangle, and circle, respectively. The compounds represented with a star have drug-like properties (<xref ref-type="bibr" rid="B62">Lipinski et al., 2001</xref>; <xref ref-type="bibr" rid="B122">Veber et al., 2002</xref>); those represented with triangles comply with some of the drug-likeness properties, and those represented with circles are not compliant. Other approaches to select compounds from the data sets use molecular fingerprints (<xref ref-type="bibr" rid="B49">Kadurin et al., 2017</xref>) or filter compounds directly via similarity-based virtual screening instead of designing NCE from scratch (<xref ref-type="bibr" rid="B117">Tong et al., 2021</xref>). 3) Selecting the molecular representation as a basis to learn and represent the structures and properties of molecules, e.g., SMILES (<xref ref-type="bibr" rid="B127">Weininger, 1988</xref>), SELFIES (<xref ref-type="bibr" rid="B55">Krenn et al., 2020</xref>) or molecular graphs (<xref ref-type="bibr" rid="B111">Simonovsky and Komodakis, 2018</xref>). 4) Developing and validating the model for molecule generation using metrics such as the operating characteristic curve. 5) Optimizing the model by combining reinforcement learning and property prediction (<xref ref-type="bibr" rid="B84">Olivecrona et al., 2017</xref>). 6) Generating molecules <italic>de novo</italic>, 7) Assessing the biological activity of the compounds designed in relevant <italic>in vitro</italic> or <italic>in vivo</italic> models.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Overview of ligand-based <italic>de novo</italic> design. 1) Selecting data sets. 2) Filtering molecular data sets with desired properties such as drug-likeness. In this example, compounds represented with stars comply with drug-likeness properties (<xref ref-type="bibr" rid="B62">Lipinski et al., 2001</xref>; <xref ref-type="bibr" rid="B122">Veber et al., 2002</xref>). 3) Choosing a molecular representation. 4) Selecting a <italic>de novo</italic> design model. 5) Developing, validating and optimizing the model. 6) Generating molecules <italic>de novo</italic>. 7) Testing the compounds in a relevant biological experiment.</p>
</caption>
<graphic xlink:href="fddsv-03-1222655-g001.tif"/>
</fig>
<p>Deep learning, currently used in ligand-based <italic>de novo</italic> design, learns the probability distribution of molecular data and generates continuous or discrete latent representations for molecules with property optimization (<xref ref-type="bibr" rid="B33">G&#xf3;mez-Bombarelli et al., 2018</xref>). The algorithms map the learned probability distribution and molecule representation into novel molecules while optimizing molecular properties (<xref ref-type="bibr" rid="B11">Bilodeau et al., 2022</xref>) through the tuning of hyperparameters (<xref ref-type="bibr" rid="B91">Perron et al., 2022a</xref>; <xref ref-type="bibr" rid="B10">Bender et al., 2022</xref>). Advances in deep learning are significantly advancing molecule generation, representing a big step forward in bridging the gap between chemical entities and drug-like properties (<xref ref-type="bibr" rid="B56">Krishnan et al., 2021</xref>).</p>
<p>Ligand&#xb4;s properties can be optimized in two steps: 1) property-based generation, wherein models would learn the chemical space of molecules with desirable properties; and 2) novel molecules are generated within a desired property space (<xref ref-type="bibr" rid="B11">Bilodeau et al., 2022</xref>). Examples of ligand-based <italic>de novo</italic> design are deep neural networks (DNN), recurrent neural networks (RNNs) (<xref ref-type="bibr" rid="B84">Olivecrona et al., 2017</xref>), and variational autoencoders (VAE) (<xref ref-type="bibr" rid="B33">G&#xf3;mez-Bombarelli et al., 2018</xref>). Olivercroma <italic>et al.</italic> (<xref ref-type="bibr" rid="B84">Olivecrona et al., 2017</xref>) proposed the REIVENT model that uses RNN for <italic>de novo</italic> design. They introduced a reinforcement learning method to fine-tune the pre-trained RNN so the model could generate structures with desirable properties. Recently, Blaschke et al. released REINVENT 2.0 (<xref ref-type="bibr" rid="B12">Blaschke et al., 2020</xref>) making the code freely accessible in Github.</p>
<p>Ligand-based <italic>de novo</italic> design using DNN (<xref ref-type="bibr" rid="B87">Palazzesi and Pozzan, 2022</xref>) requires a large number of compounds that demand more computational resources. The DNN architecture is prone to problems because of fitting numerous parameters. For this reason, a large training data set is needed to reduce the risk of overfitting. However, sufficient bioactivity data for a biological endpoint is not always available (<xref ref-type="bibr" rid="B133">Wu et al., 2018</xref>). The lack of sufficient data has led to using methods for compound selection or the development of new methods for compound selection. Altae-Tran et al<italic>.</italic> (<xref ref-type="bibr" rid="B1">Altae-Tran et al., 2017</xref>) demonstrated how the one-shot learning paradigm can be used to address the overfitting problem; they used DNN to transform small molecules into embedding vectors in a continuous feature space whose similarity measures are then iteratively learned. They showed that this DNN architecture offers convincing performance in many activity prediction tasks given limited amounts of training. On the other hand, computer scientists advise using algorithms that can detect meaningful patterns in small data sets, which is a typical case in the early stage of drug discovery (<xref ref-type="bibr" rid="B104">Schneider and Clark, 2019</xref>). For instance, an initial approach to <italic>de novo</italic> design is to start from small data sets of compounds with diverse structures and diverse properties of pharmaceutical relevance (<xref ref-type="bibr" rid="B16">Ch&#xe1;vez-Hern&#xe1;ndez and Medina-Franco, 2023</xref>).</p>
<p>The availability of gold standard datasets as well as independently generated data sets are valuable in generating well-performing models (<xref ref-type="bibr" rid="B121">Vamathevan et al., 2019</xref>). Dissimilarity-based compound selection could be improved if one focused the selection on a structural diverse dataset (for instance derived from natural products). Some approaches proposed suggest using quality data sets using a dissimilarity-based compound selection method such as the MaxMin or MaxSum algorithms (Leach and Gilleteds, 2007). Recently, we reported the use of the MaxMin algorithm for the selection of natural product subsets (<xref ref-type="bibr" rid="B16">Ch&#xe1;vez-Hern&#xe1;ndez and Medina-Franco, 2023</xref>) using the Universal Natural Product Database (UNPD) (<xref ref-type="bibr" rid="B38">Gu et al., 2013</xref>). In that study, the natural product subsets generated had the most diverse chemical structures with physicochemical properties of pharmaceutical interest similar to the original data set. Chemical structures in the natural product subsets were represented with SMILES encoding chirality, an important feature of natural products.</p>
</sec>
<sec id="s3">
<title>3 Main sources of data sets used to develop generative and predictive models</title>
<sec id="s3-1">
<title>3.1 Current status of reference and benchmark datasets</title>
<p>The first step in <italic>de novo</italic> design is to select, from the vast chemical space, the appropriate subset of all possible molecules for a desired biological activity (<xref ref-type="bibr" rid="B105">Schneider et al., 2000</xref>). To have an idea, the size of the chemical space has been estimated at around 10<sup>60</sup> small molecules and between 10<sup>20</sup>&#x2013;10<sup>24</sup> for all molecules up to 30 atoms that comply with Lipinski&#x2019;s rule-of-five (<xref ref-type="bibr" rid="B96">Reymond, 2015</xref>). According to Yang <italic>et al.</italic> compound data sets can be classified into on-demand databases, collections containing bioactivity data, compounds databases commercially available, and natural products databases (<xref ref-type="bibr" rid="B135">Yang et al., 2019</xref>). Herein, we include benchmark, decoy and inactive compounds data sets as others categories as illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>. In this figure, on-demand databases are further divided into commercially available (e.g., Enamine-REAL, CHEMriya and Freedom Space) (<xref ref-type="bibr" rid="B20">Chemspace, 2023</xref>) and in-house (e.g., Pfizer and AstraZeneca). The figure shows examples of compound databases in other categories which are discussed in the remainder of this section.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Classification of compound databases and representative examples of each one. For the discussion of this manuscript, databases are split into six main categories: on-demand, commercial availability, bioactivity, natural products, benchmark and decoys.</p>
</caption>
<graphic xlink:href="fddsv-03-1222655-g002.tif"/>
</fig>
<p>Among the different types of chemical databases, <italic>de novo</italic> design employs libraries from different categories outlined in <xref ref-type="fig" rid="F2">Figure 2</xref>. Specific examples are ChEMBL (<xref ref-type="bibr" rid="B26">Davies et al., 2015</xref>; <xref ref-type="bibr" rid="B75">Mendez et al., 2019</xref>), PubChem (<xref ref-type="bibr" rid="B50">Kim et al., 2023</xref>), DrugBank (<xref ref-type="bibr" rid="B130">Wishart et al., 2006</xref>; <xref ref-type="bibr" rid="B129">Wishart et al., 2008</xref>; <xref ref-type="bibr" rid="B128">Wishart et al., 2018</xref>), Enamine&#xb4;s REadily AccessibLe (REAL) (<xref ref-type="bibr" rid="B29">Enamine, 2023</xref>), CHEMriya (<xref ref-type="bibr" rid="B19">CHEMriya, 2023</xref>), Freedom Space (<xref ref-type="bibr" rid="B20">Chemspace, 2023</xref>), ZINC-22 (<xref ref-type="bibr" rid="B116">Tingle et al., 2023</xref>), and MoleculeNet (<xref ref-type="bibr" rid="B133">Wu et al., 2018</xref>) which more details for each one are provided in <xref ref-type="table" rid="T1">Table 1</xref> and further commented in the next sections.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Main sources of public molecular data sets used in <italic>de novo</italic> design.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Data sets</th>
<th align="left">Category</th>
<th align="left">Description</th>
<th align="left">Ref.</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">ChEMBL</td>
<td align="left">Bioactivity</td>
<td align="left">Database with 2,354,965 bioactive drug-like small molecules with 2D structures and calculated properties.</td>
<td align="center">
<xref ref-type="bibr" rid="B26">Davies et al. (2015),</xref> <xref ref-type="bibr" rid="B75">Mendez et al. (2019)</xref>
</td>
</tr>
<tr>
<td align="left">PubChem</td>
<td align="left">Bioactivity</td>
<td align="left">Database at the US National Institutes of Health with 115 million compounds. It includes names, molecular formulas, structures, physical properties, and biological activities.</td>
<td align="center">
<xref ref-type="bibr" rid="B50">Kim et al. (2023)</xref>
</td>
</tr>
<tr>
<td align="left">DrugBank</td>
<td align="left">Bioactivity</td>
<td align="left">Version 5.1.10 contains 15,448 drug entries including 2,740 approved small molecule drugs.</td>
<td align="center">
<xref ref-type="bibr" rid="B130">Wishart et al. (2006)</xref>
</td>
</tr>
<tr>
<td align="left">ZINC-22</td>
<td align="left">Commercial</td>
<td align="left">Database with over 37 billion enumerated, searchable, commercially available compounds in 2D.</td>
<td align="center">
<xref ref-type="bibr" rid="B116">Tingle et al. (2023)</xref>
</td>
</tr>
<tr>
<td align="left">CHEMriya</td>
<td align="left">On-demand</td>
<td align="left">Database with 12 billion novel and synthetically feasible small molecules.</td>
<td align="center">
<xref ref-type="bibr" rid="B19">CHEMriya (2023)</xref>
</td>
</tr>
<tr>
<td align="left">Freedom Space (Chemspace)</td>
<td align="left">On-demand</td>
<td align="left">Database with 201 million molecules; 73% of its compounds comply with drug-likeness properties.</td>
<td align="center">
<xref ref-type="bibr" rid="B20">Chemspace (2023)</xref>
</td>
</tr>
<tr>
<td align="left">Enamine-REAL</td>
<td align="left">On-demand</td>
<td align="left">Database with 6 billion synthetic compounds that comply with drug-likeness properties.</td>
<td align="center">
<xref ref-type="bibr" rid="B29">Enamine (2023)</xref>
</td>
</tr>
<tr>
<td align="left">MoleculeNet</td>
<td align="left">Benchmark</td>
<td align="left">Compilation of 17 datasets with over 700,000 compounds in total used for comparison of different machine learning algorithms.</td>
<td align="center">
<xref ref-type="bibr" rid="B133">Wu et al. (2018)</xref>
</td>
</tr>
<tr>
<td align="left">MOSES</td>
<td align="left">Benchmark</td>
<td align="left">Dataset with 1,936,962 molecules from ZINC Clean Lead suitable for hit identification and ADMET optimization. It does have metrics to detect common issues in generative models such as overfitting or if the model does not limit to producing only a few typical molecules.</td>
<td align="center">
<xref ref-type="bibr" rid="B94">Polykovskiy et al. (2020)</xref>
</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-2">
<title>3.2 On-demand databases</title>
<p>Early approaches to ligand-based <italic>de novo</italic> design involved fragment compounds into unique building blocks which could be recombined to make new molecules. A number of commercial suppliers of chemical samples offer large make-on-demand collections that can be reliably synthesized because the building blocks are available as well as the synthetic routes and methods (<xref ref-type="bibr" rid="B125">Warr et al., 2022</xref>; <xref ref-type="bibr" rid="B53">Korn et al., 2023</xref>). There are also large collections of fragments or building blocks commercially available. Examples of on-demand compound databases and suppliers are REAL (Enamine) (<xref ref-type="bibr" rid="B29">Enamine, 2023</xref>), CHEMriya (OTAVA) (<xref ref-type="bibr" rid="B19">CHEMriya, 2023</xref>), and Freedom Space (Chemspace) (<xref ref-type="bibr" rid="B20">Chemspace, 2023</xref>) (<xref ref-type="table" rid="T1">Table 1</xref>). REAL database (<xref ref-type="bibr" rid="B29">Enamine, 2023</xref>) comprises over 6 billion molecules that comply with the traditional drug-likeness criteria. CHEMriya (<xref ref-type="bibr" rid="B19">CHEMriya, 2023</xref>) contains 12 billion novel and synthetically feasible small molecules whose molecules are not explicitly listed in the public domain. Freedom Space (<xref ref-type="bibr" rid="B20">Chemspace, 2023</xref>) contains 201 million molecules and 73% of its compounds are drug-like (as assessed with the &#x201c;rule of five&#x201d;). Examples of on-demand in-house databases from the pharmaceutical industry are 10<sup>15</sup> compounds of AZ Space (AstraZeneca) (<xref ref-type="bibr" rid="B35">Grebner, 2022</xref>), 10<sup>19</sup> compounds of JFS (Johnson &#x26; Johnson) (<xref ref-type="bibr" rid="B126">Warr, 2021</xref>), 10<sup>18</sup> compounds of PGVL (Pfizer) (<xref ref-type="bibr" rid="B43">Hu et al., 2012</xref>), 10<sup>17</sup> compounds BICLAIM (Boehringer Ingelheim) (<xref ref-type="bibr" rid="B53">Korn et al., 2023</xref>), and 10<sup>20</sup> compounds MASSIV (Merck/EMD) (<xref ref-type="bibr" rid="B53">Korn et al., 2023</xref>).</p>
</sec>
<sec id="s3-3">
<title>3.3 Commercially available databases</title>
<p>One of the largest and long-standing compendiums of commercially available compounds in ZINC. The most recent version, ZINC-22 (<xref ref-type="bibr" rid="B116">Tingle et al., 2023</xref>) contains over 37 billion enumerated, searchable, commercially available compounds in 2D. Over 4.5 billion have been built in biologically relevant ready-to-dock 3D formats (<xref ref-type="bibr" rid="B116">Tingle et al., 2023</xref>). Some examples of <italic>de novo</italic> design using ZINC include the design of inhibitors of DDR1 (discoidin domain receptor 1, a kinase target implicated in fibrosis and other diseases) (<xref ref-type="bibr" rid="B139">Zhavoronkov et al., 2019</xref>) and compounds with activity towards the dopamine receptor D2 (<xref ref-type="bibr" rid="B63">Liu et al., 2019</xref>; <xref ref-type="bibr" rid="B69">Maziarka et al., 2020</xref>).</p>
</sec>
<sec id="s3-4">
<title>3.4 Bioactivity databases</title>
<p>
<italic>De novo</italic> design based on deep learning algorithms frequently use PubChem, ChEMBL, and DrugBank to select subsets of compounds focused on a biological target or biological endpoint as the design of ligands (<xref ref-type="bibr" rid="B60">Li et al., 2018</xref>; <xref ref-type="bibr" rid="B59">Li et al., 2022</xref>; <xref ref-type="bibr" rid="B63">Liu et al., 2019</xref>). PubChem (<xref ref-type="bibr" rid="B50">Kim et al., 2023</xref>) is a freely accessible database from the US National Institutes of Health (NIH) with over 115 million compounds. At the time of writing, the most recent version release of ChEMBL is 32 (<xref ref-type="bibr" rid="B26">Davies et al., 2015</xref>; <xref ref-type="bibr" rid="B75">Mendez et al., 2019</xref>) and contains 2,354,965 compounds bioactive drug-like small molecules with 2D structures and calculated properties. DrugBank (<xref ref-type="bibr" rid="B130">Wishart et al., 2006</xref>; <xref ref-type="bibr" rid="B129">Wishart et al., 2008</xref>; <xref ref-type="bibr" rid="B128">Wishart et al., 2018</xref>) version 5.1.10 (released 2023-01-04) contains 15,448 drug entries including 2,740 approved small molecule drugs, 1,577 approved biologics (proteins, peptides, vaccines, and allergens), 134 nutraceuticals and over 6,717 experimental (discovery-phase) drugs. Some applications include the <italic>de novo</italic> design of SARS-CoV-2 Mpro inhibitors (<xref ref-type="bibr" rid="B59">Li et al., 2022</xref>), the design of ligands against the adenosine receptor (A<sub>2A</sub>R) (<xref ref-type="bibr" rid="B63">Liu et al., 2019</xref>), and the generation of compounds analogs to celecoxib (used to manage symptoms of various types of arthritis pain and reduce precancerous polyps in the colon) (<xref ref-type="bibr" rid="B60">Li et al., 2018</xref>; <xref ref-type="bibr" rid="B28">DRUGBANK, 2023</xref>).</p>
</sec>
<sec id="s3-5">
<title>3.5 Natural product databases</title>
<p>Natural product databases (<xref ref-type="bibr" rid="B34">G&#xf3;mez-Garc&#xed;a and Medina-Franco, 2022</xref>; <xref ref-type="bibr" rid="B99">Sald&#xed;var-Gonz&#xe1;lez et al., 2022</xref>) are important in drug discovery. From drugs approved by 2020 about 23% are natural products or derivatives (<xref ref-type="bibr" rid="B79">Newman and Cragg, 2020</xref>). Natural products have a diversity of privileged scaffolds (<xref ref-type="bibr" rid="B3">Atanasov et al., 2021</xref>; <xref ref-type="bibr" rid="B37">Grigalunas et al., 2022</xref>) and molecular fragments (<xref ref-type="bibr" rid="B17">Ch&#xe1;vez-Hern&#xe1;ndez et al., 2020a</xref>; <xref ref-type="bibr" rid="B18">Ch&#xe1;vez-Hern&#xe1;ndez et al., 2020b</xref>) that depend on the particular source (<xref ref-type="bibr" rid="B70">Medina-Franco et al., 2022b</xref>); a diversity of chiral centers; and a larger fraction of sp<sup>3</sup> carbon atoms and functional groups (<xref ref-type="bibr" rid="B3">Atanasov et al., 2021</xref>; <xref ref-type="bibr" rid="B37">Grigalunas et al., 2022</xref>).</p>
<p>Privileged structures were defined by Evans et al. (<xref ref-type="bibr" rid="B30">Evans et al., 1988</xref>) as <italic>chemical structures capable of providing useful ligands for more than one receptor judicious modification of such structures could be a viable alternative in the search for new receptor agonists and antagonists</italic>. <xref ref-type="bibr" rid="B106">Schneider and Schneider (2017)</xref> define a privileged structure as a chemical structure that may be considered to possess geometries suitable for decoration with side chains, such that the resulting products bind to different target proteins or a ligand that potently interacts with one (selective binder) or many target receptors (promiscuous binder). To this end, natural products are used in the development of pseudo-natural products, compounds that are generated through a <italic>de novo</italic> combination of natural product fragments, allowing the exploration of uncharted areas of biologically relevant chemical space that are different from the chemical space covered by the compounds from which they are derived (<xref ref-type="bibr" rid="B37">Grigalunas et al., 2022</xref>).</p>
<p>Representative natural product datasets that can be used in <italic>de novo</italic> design are Collection of Open NatUral ProdUcTs (COCONUT) (<xref ref-type="bibr" rid="B115">Sorokina et al., 2021</xref>), SuperNatural 3.0 (<xref ref-type="bibr" rid="B32">Gallo et al., 2023</xref>), UNPD (<xref ref-type="bibr" rid="B38">Gu et al., 2013</xref>), NuBBE<sub>DB</sub> (<xref ref-type="bibr" rid="B92">Pilon et al., 2017</xref>; <xref ref-type="bibr" rid="B101">Sald&#xed;var-Gonz&#xe1;lez et al., 2019</xref>), SistematX (<xref ref-type="bibr" rid="B108">Scotti et al., 2018</xref>; <xref ref-type="bibr" rid="B25">Costa et al., 2021</xref>), CIFPMA (<xref ref-type="bibr" rid="B86">Olmedo et al., 2017</xref>; <xref ref-type="bibr" rid="B85">Olmedo and Medina-Franco, 2020</xref>), PeruNPDB (<xref ref-type="bibr" rid="B7">Barazorda-Ccahuana et al., 2023</xref>), BIOFACQUIM (<xref ref-type="bibr" rid="B93">Pil&#xf3;n-Jim&#xe9;nez et al., 2019</xref>; <xref ref-type="bibr" rid="B102">S&#xe1;nchez-Cruz et al., 2019</xref>), UNIIQUIM(<xref ref-type="bibr" rid="B119">UNIIQUIM, 2015</xref>), and are summarized in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Examples of natural product databases in the public domain.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Data sets</th>
<th align="left">Description</th>
<th align="left">Ref.</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">COCONUT</td>
<td align="left">Extensive database with 406,076 unique structures.</td>
<td align="center">
<xref ref-type="bibr" rid="B115">Sorokina et al. (2021)</xref>
</td>
</tr>
<tr>
<td align="left">SuperNatural 3.0</td>
<td align="left">A database with 449 058 natural compounds and derivatives. It includes chemical structure, physicochemical information, information on pathways, mechanism of action, toxicity, vendor information if available, drug-like chemical space prediction for several diseases such as antiviral, antibacterial, antimalarial, anticancer, and target-specific cells.</td>
<td align="center">
<xref ref-type="bibr" rid="B32">Gallo et al. (2023)</xref>
</td>
</tr>
<tr>
<td align="left">UNPD</td>
<td align="left">Second-largest database with around 229,000 natural products that contain chirality information.</td>
<td align="center">
<xref ref-type="bibr" rid="B38">Gu et al. (2013)</xref>
</td>
</tr>
<tr>
<td align="left">TCM Database@Taiwan</td>
<td align="left">Database with more than 20,000 pure compounds isolated from 453 TCM ingredients.</td>
<td align="center">
<xref ref-type="bibr" rid="B21">Chen (2011)</xref>
</td>
</tr>
<tr>
<td align="left">IMPPAT</td>
<td align="left">Database of 9,596 phytochemicals from 1,742 Indian medicinal plants.</td>
<td align="center">
<xref ref-type="bibr" rid="B76">Mohanraj et al. (2018)</xref>
</td>
</tr>
<tr>
<td align="left">AfroDB</td>
<td align="left">Compound collection with more than 1,000 compounds from African medicinal plants.</td>
<td align="center">
<xref ref-type="bibr" rid="B83">Ntie-Kang et al. (2013)</xref>
</td>
</tr>
<tr>
<td align="left">NuBBE<sub>DB</sub>
</td>
<td align="left">Brazilian database with 2,223 natural products encoding as SMILES, InChI, and InChIKey strings, Ro5 and Veber descriptors, source, therapeutic effect, and reference.</td>
<td align="center">
<xref ref-type="bibr" rid="B120">Valli et al. (2013),</xref> <xref ref-type="bibr" rid="B92">Pilon et al. (2017),</xref> <xref ref-type="bibr" rid="B101">Sald&#xed;var-Gonz&#xe1;lez et al. (2019)</xref>
</td>
</tr>
<tr>
<td align="left">SistematX</td>
<td align="left">Brazilian database with 9,514 unique secondary metabolites encoding as SMILES, InChI, and InChIKey strings, and include physicochemical drug-like descriptors, predicted biological activities, and reference.</td>
<td align="center">
<xref ref-type="bibr" rid="B108">Scotti et al. (2018),</xref> <xref ref-type="bibr" rid="B25">Costa et al. (2021)</xref>
</td>
</tr>
<tr>
<td align="left">CIFPMA</td>
<td align="left">Database developed at the University of Panama. It contains natural products that have been tested in over 25 <italic>in vitro</italic> and <italic>in vivo</italic> bioassays, for different therapeutic targets.</td>
<td align="center">
<xref ref-type="bibr" rid="B86">Olmedo et al. (2017),</xref> <xref ref-type="bibr" rid="B85">Olmedo and Medina-Franco (2020)</xref>
</td>
</tr>
<tr>
<td align="left">PeruNPDB</td>
<td align="left">Peru database developed at the Catholic University of Santa Maria. The current version has 280 natural products from animals and plants.</td>
<td align="center">
<xref ref-type="bibr" rid="B7">Barazorda-Ccahuana et al. (2023)</xref>
</td>
</tr>
<tr>
<td align="left">BIOFACQUIM</td>
<td align="left">Mexican database with structures of 531 natural products isolated and characterized at UNAM and other Mexican institutions.</td>
<td align="center">
<xref ref-type="bibr" rid="B93">Pil&#xf3;n-Jim&#xe9;nez et al. (2019),</xref> <xref ref-type="bibr" rid="B102">S&#xe1;nchez-Cruz et al. (2019)</xref>
</td>
</tr>
<tr>
<td align="left">UNIIQUIM</td>
<td align="left">Mexican database with 1,112 plant natural products mostly isolated and characterized at the Institute of Chemistry of the UNAM.</td>
<td align="center">
<xref ref-type="bibr" rid="B119">
<italic>UNIIQUIM</italic> (2015)</xref>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Other libraries of natural products with an emphasis on commercial availability are listed on the NIH website (<xref ref-type="bibr" rid="B80">NIH, 2023</xref>).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>SuperNatural 3.0, COCONUT and UNPD are the most extensive natural product databases. SuperNatural 3.0 (<xref ref-type="bibr" rid="B32">Gallo et al., 2023</xref>) is arguably the most extensive natural product database with 449,058 natural compounds and derivatives; followed by COCONUT (<xref ref-type="bibr" rid="B115">Sorokina et al., 2021</xref>) with 406,076 unique structures (no encoding stereochemistry) and UNPD (<xref ref-type="bibr" rid="B38">Gu et al., 2013</xref>) with 197,201 natural products that contain chirality information.</p>
<p>Several public natural products databases compile the compounds isolated and characterized from a geographical region or the country of origin as China, India and Africa. For instance, Chinese Traditional Medicine (TCM) Database@Taiwan (<xref ref-type="bibr" rid="B21">Chen, 2011</xref>) is a non-commercial TCM database with more than 20,000 pure compounds isolated from 453 TCM ingredients; A curated database of Indian Medicinal Plants, Phytochemistry And Therapeutics (IMPPAT) (<xref ref-type="bibr" rid="B76">Mohanraj et al., 2018</xref>) is a manually curated database of 9,596 phytochemicals from 1,742 Indian medicinal plants; and AfroDB (<xref ref-type="bibr" rid="B83">Ntie-Kang et al., 2013</xref>) with more than 1,000 small and structural diversity compounds from African medicinal plants.</p>
<p>Representative Latin American databases (<xref ref-type="bibr" rid="B34">G&#xf3;mez-Garc&#xed;a and Medina-Franco, 2022</xref>) are NuBBE<sub>DB</sub> (<xref ref-type="bibr" rid="B92">Pilon et al., 2017</xref>; <xref ref-type="bibr" rid="B101">Sald&#xed;var-Gonz&#xe1;lez et al., 2019</xref>), SistematX (<xref ref-type="bibr" rid="B108">Scotti et al., 2018</xref>; <xref ref-type="bibr" rid="B25">Costa et al., 2021</xref>) from Brazil; CIFPMA (<xref ref-type="bibr" rid="B86">Olmedo et al., 2017</xref>; <xref ref-type="bibr" rid="B85">Olmedo and Medina-Franco, 2020</xref>) from Panama; PeruNPDB (<xref ref-type="bibr" rid="B7">Barazorda-Ccahuana et al., 2023</xref>) from Peru; BIOFACQUIM (<xref ref-type="bibr" rid="B93">Pil&#xf3;n-Jim&#xe9;nez et al., 2019</xref>; <xref ref-type="bibr" rid="B102">S&#xe1;nchez-Cruz et al., 2019</xref>) and UNIIQUIM (<xref ref-type="bibr" rid="B119">
<italic>UNIIQUIM</italic>, 2015</xref>) from Mexico. The current version of NuBBE<sub>DB</sub> (<xref ref-type="bibr" rid="B92">Pilon et al., 2017</xref>; <xref ref-type="bibr" rid="B101">Sald&#xed;var-Gonz&#xe1;lez et al., 2019</xref>) contains 2,223 natural products encoding as linear notations as SMILES. SistematX (<xref ref-type="bibr" rid="B108">Scotti et al., 2018</xref>; <xref ref-type="bibr" rid="B25">Costa et al., 2021</xref>) has 9,514 unique secondary metabolites arising from 20,934 botanical occurrences across five families. Other natural product collections from Latin America are CIFPMA, the Natural Products Database from the University of Panama, Republic of Panama (<xref ref-type="bibr" rid="B86">Olmedo et al., 2017</xref>; <xref ref-type="bibr" rid="B85">Olmedo and Medina-Franco, 2020</xref>)with 354 compounds. CIFPMA molecules have the potential to show target selectivity in biochemical assays and are useful molecules to identify reference compounds for virtual screening campaigns (<xref ref-type="bibr" rid="B86">Olmedo et al., 2017</xref>; <xref ref-type="bibr" rid="B85">Olmedo and Medina-Franco, 2020</xref>). The first version of the Peruvian Natural Products Database (PeruNPDB) had 280 natural products isolated from plants and animal sources (<xref ref-type="bibr" rid="B7">Barazorda-Ccahuana et al., 2023</xref>). BIOFACQUIM (<xref ref-type="bibr" rid="B93">Pil&#xf3;n-Jim&#xe9;nez et al., 2019</xref>; <xref ref-type="bibr" rid="B102">S&#xe1;nchez-Cruz et al., 2019</xref>) contains 531 natural products isolated and characterized at the School of Chemistry of the National Autonomous University of Mexico (UNAM) and other Mexican institutions. UNIIQUIM (<xref ref-type="bibr" rid="B119">UNIIQUIM, 2015</xref>) with 1,112 plant natural products mostly isolated and characterized at the Institute of Chemistry of the UNAM.</p>
</sec>
<sec id="s3-6">
<title>3.6 Benchmark databases</title>
<p>The development of reliable machine learning algorithms has been limited due to the lack of standard benchmark datasets to compare the efficacy of the methods proposed (<xref ref-type="bibr" rid="B46">Jain and Nicholls, 2008</xref>). Furthermore, machine learning in chemistry compared with other areas such as computer speech and vision has a main disadvantage, the data recovery (<xref ref-type="bibr" rid="B133">Wu et al., 2018</xref>; <xref ref-type="bibr" rid="B41">Guo et al., 2022</xref>), because of measuring chemical properties often requires specialized instruments; as a result, datasets with experimentally determined results are small and often not sufficiently large to cover the high-demanding needs of machine-learning tasks (<xref ref-type="bibr" rid="B133">Wu et al., 2018</xref>). Another challenge is data splitting (the way in which datasets are split into training data and testing data). Some are random selection and rational selection. The former is randomly extracting a compound&#x2019;s fraction from the data set. In contrast to rational selection, training and testing are selected from the same clusters of compounds. Random selection is common in machine learning but is often not correct for chemical data (<xref ref-type="bibr" rid="B109">Sheridan, 2013</xref>). In response to these challenges, standard benchmark data sets are being developed to evaluate <italic>de novo</italic> design protocols [(<xref ref-type="bibr" rid="B133">Wu et al., 2018</xref>; <xref ref-type="bibr" rid="B13">Brown et al., 2019</xref>; <xref ref-type="bibr" rid="B94">Polykovskiy et al., 2020</xref>). One example is MoleculeNet (<xref ref-type="bibr" rid="B133">Wu et al., 2018</xref>), a large-scale data set built upon multiple public databases. MoleculeNet is organized into regression and classification datasets and has over 700,000 compounds tested on a range of different properties subdivided into four categories (quantum mechanics, physical chemistry, biophysics, and physiology). Another example is the Molecular Sets (MOSES) (<xref ref-type="bibr" rid="B94">Polykovskiy et al., 2020</xref>) that contains 1,936,962 molecules (split into training, testing and scaffold datasets) and a set of metrics to evaluate the quality and diversity of generated structures. Metrics detect common issues in generative models such as overfitting or if the <italic>de novo</italic> design model just generates fairly common (not novel) structures (<xref ref-type="bibr" rid="B13">Brown et al., 2019</xref>; <xref ref-type="bibr" rid="B94">Polykovskiy et al., 2020</xref>). The developers of MOSES implemented and compared several molecular generation models and suggested using the results as reference points for further advancements in generative chemistry research.</p>
</sec>
<sec id="s3-7">
<title>3.7 Current decoy data sets and inactive compounds</title>
<p>Accuracy of predictive models depends on data quality and quantity. Also, the balance between active and inactive compounds is important, which remains an issue to resolve. Historically, the publication of active compounds in a given assay or with a particular endpoint has been prioritized over inactive molecules. For example, a recent comprehensive analysis of published screening bioactivity data shows that in ChEMBL V.29 (release in 2022) there is a large number of active compounds (<italic>ca. 71</italic>%) with respect to the inactive ones (<italic>ca.</italic> 31%); contrary to what it would be expected (<xref ref-type="bibr" rid="B66">L&#xf3;pez-L&#xf3;pez et al., 2022</xref>). These results highlight the relevance of changing the mindset about the importance and utility of inactive or negative data (keeping in mind that the definition of &#x201c;inactive&#x201d; is subjective as it depends on the particular biological assay and the predefined threshold to deem a compound inactive).</p>
<p>Decoy data sets have been developed in an attempt to reduce the gap between inactive (or negative) and active compounds. Decoy molecules are assumed non-active but have high physicochemical property similarity (but not topologically) to reference compounds (<xref ref-type="bibr" rid="B95">R&#xe9;au et al., 2018</xref>). Decoys are useful to evaluate benchmark models that were assembled in the absence of inactive compounds experimentally measured (<xref ref-type="bibr" rid="B45">Irwin, 2008</xref>) and can be used to enrich <italic>de novo</italic> design models. <xref ref-type="table" rid="T3">Table 3</xref> summarizes examples of large databases of experimentally tested active or inactive compounds, decoy datasets, and tools to generate decoys for specific projects.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Examples of potential inactive and decoy resources for enriching <italic>de novo</italic> design models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Datasets with active and inactive compounds</th>
<th align="center">Criteria to select inactive data</th>
<th align="center">Ref.</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">ChEMBL</td>
<td rowspan="2" align="left">Reported activity data.</td>
<td align="left">
<xref ref-type="bibr" rid="B26">Davies et al. (2015),</xref> <xref ref-type="bibr" rid="B75">Mendez et al. (2019)</xref>
</td>
</tr>
<tr>
<td align="center">PubChem</td>
<td align="center">
<xref ref-type="bibr" rid="B50">Kim et al. (2023)</xref>
</td>
</tr>
<tr>
<td align="center">Binding DB</td>
<td align="left">Reported ligand-receptor affinity.</td>
<td align="center">
<xref ref-type="bibr" rid="B22">Chen et al. (2002)</xref>
</td>
</tr>
<tr>
<td style="background-color:#c6c7c9" align="center">Decoy datasets</td>
<td style="background-color:#c6c7c9" align="left">Common decoy selection criteria</td>
<td style="background-color:#c6c7c9" align="left"/>
</tr>
<tr>
<td align="center">ZINC</td>
<td rowspan="2" align="left">Compounds that share drug-like properties with the reference (active) compounds.</td>
<td align="center">
<xref ref-type="bibr" rid="B116">Tingle et al. (2023)</xref>
</td>
</tr>
<tr>
<td align="center">DUD-E</td>
<td align="center">
<xref ref-type="bibr" rid="B78">Mysinger et al. (2012)</xref>
</td>
</tr>
<tr>
<td align="center">DUD</td>
<td align="left">Database with 2950 annotated ligands and 95,316 property-matched decoys for 40 targets.</td>
<td align="center">
<xref ref-type="bibr" rid="B45">Irwin (2008)</xref>
</td>
</tr>
<tr>
<td align="center">MUV</td>
<td align="left">Compounds that share structural similarity with active reported compounds.</td>
<td align="center">
<xref ref-type="bibr" rid="B97">Rohrer and Baumann (2009)</xref>
</td>
</tr>
<tr>
<td align="center">DEKOIS 2.0</td>
<td align="left">Compounds that share drug-like properties and structural similarity with the reference (active) compounds.</td>
<td align="center">
<xref ref-type="bibr" rid="B8">Bauer et al. (2013)</xref>
</td>
</tr>
<tr>
<td style="background-color:#c6c7c9" align="center">Decoy tools</td>
<td style="background-color:#c6c7c9" align="center">Common decoy compound selection criteria</td>
<td style="background-color:#c6c7c9" align="left"/>
</tr>
<tr>
<td align="center">DecoyFinder</td>
<td align="left">Allows the automatic creation of datasets of compounds with physicochemical similarity and without structural similarity respect to the reference (active) compounds.</td>
<td align="center">
<xref ref-type="bibr" rid="B15">Cereto-Massagu&#xe9; et al. (2012)</xref>
</td>
</tr>
<tr>
<td align="center">RADER</td>
<td align="left">Allows the automatic generation of datasets of compounds with physicochemical and structural similarity with respect to the reference (active) compounds.</td>
<td align="center">
<xref ref-type="bibr" rid="B123">Wang et al. (2017)</xref>
</td>
</tr>
<tr>
<td align="center">ZINC pharmer</td>
<td align="left">Enables the automatic identification of compounds with pharmacophore similarity with respect to the reference (active and inactive) compounds.</td>
<td align="center">
<xref ref-type="bibr" rid="B51">Koes and Camacho (2012)</xref>
</td>
</tr>
<tr>
<td align="center">Decoy Developer</td>
<td align="left">Allows the automatic generation of peptides decoys.</td>
<td align="center">
<xref ref-type="bibr" rid="B110">Shipman et al. (2019)</xref>
</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Decoy compounds have been used to describe, explore, and expand the knowledge of active molecules. For example, rationalizing the physicochemical, chemical, biological, and clinical data of active compounds (<xref ref-type="bibr" rid="B64">L&#xf3;pez-L&#xf3;pez et al., 2021a</xref>). Recently, decoys can be employed in several <italic>de novo</italic> protocols based on ligand or structure as summarized in <xref ref-type="table" rid="T4">Table 4</xref>.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Examples of applications of decoys in <italic>de novo</italic> design.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Approach</th>
<th align="center">Purpose of using decoy sets</th>
<th align="center">Ref.</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4" align="center">Ligand-based</td>
<td align="left">&#x2022; Validation of new protocols and scoring functions based on similarity metrics and 3D shape.</td>
<td rowspan="4" align="center">
<xref ref-type="bibr" rid="B2">(Ar&#xfa;s-Pous et al. (2020)</xref>; <xref ref-type="bibr" rid="B4">Awale and Reymond. (2015)</xref>; <xref ref-type="bibr" rid="B14">Cao et al. (2020)</xref>; <xref ref-type="bibr" rid="B74">Medina-Franco et al. (2019)</xref>; <xref ref-type="bibr" rid="B82">Norinder et al. (2019)</xref>; <xref ref-type="bibr" rid="B88">Papadopoulos et al. (2021)</xref>; <xref ref-type="bibr" rid="B112">Skalic et al. (2019b)</xref>; <xref ref-type="bibr" rid="B113">Skalic et al. (2019a)</xref>; <xref ref-type="bibr" rid="B118">Ullanat (2020)</xref>
</td>
</tr>
<tr>
<td align="left">&#x2022; Improvement of the accuracy of AI-based models.</td>
</tr>
<tr>
<td align="left">&#x2022; Improvement of the accuracy of QSAR models.</td>
</tr>
<tr>
<td align="left">&#x2022; Enrichment of inactive &#x201c;dark regions&#x201d; in chemical space.</td>
</tr>
<tr>
<td rowspan="2" align="center">Structure-based</td>
<td align="left">&#x2022; Validation of new protocols and scoring functions based in docking, molecular dynamics, and pharmacophore modeling.</td>
<td rowspan="2" align="center">
<xref ref-type="bibr" rid="B6">Balius et al. (2013)</xref>; <xref ref-type="bibr" rid="B9">Beato et al. (2013)</xref>; <xref ref-type="bibr" rid="B39">Guo J et al. (2021)</xref>; <xref ref-type="bibr" rid="B68">Ma et al. (2021)</xref>; <xref ref-type="bibr" rid="B81">Niitsu and Sugita (2023)</xref>
</td>
</tr>
<tr>
<td align="left">&#x2022; Peptide and protein design.</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4">
<title>4 Criteria to generate compound datasets with high quality</title>
<p>The quality of a data set is multifaceted. Commonly, it is associated with the experimental reproducibility of each data point and the experimental similarities between the protocols used to derive such data. Another important aspect of data quality is the balance between active and inactive compound. The latter is specially a challenge in public data sets due to the overall lack of published negative data. Finding qualitative yet better quantitative relationships between chemical structures and biological activity has been long pursued in medicinal chemistry and drug discovery. With the rapid increase and deployment of the predictive machine and deep learning methods, as well as the increased interest in the <italic>de novo</italic> design of chemical libraries (<xref ref-type="bibr" rid="B77">Mouchlis et al., 2021</xref>), the quantity and quality of data are becoming a central point in the discussion of the type of data sets needed (<xref ref-type="bibr" rid="B107">Schneider et al., 2020</xref>). While the more data (<xref ref-type="bibr" rid="B23">Cherkasov, 2023</xref>), the better, it is also true that the quality of the data available (that might not be quite large) is also crucial. Furthermore, the balance between active and inactive compounds is also a major consideration (<xref ref-type="bibr" rid="B66">L&#xf3;pez-L&#xf3;pez et al., 2022</xref>). <xref ref-type="table" rid="T5">Table 5</xref> summarizes criteria for generating quality data sets. The list is not exhaustive but covers what the authors consider key points based on experience and what has been discussed extensively in the literature. Each point is supported by the references indicated in the table and further commented in the next subsections.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Overview of suggested general criteria to generate quality datasets useful in <italic>de novo</italic> design.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Criteria</th>
<th align="center">Brief description</th>
<th align="center">Ref.</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Balance</td>
<td align="left">&#x2022; Quality and quantity data allow the exploration of substantial regions of chemical space.</td>
<td align="center">
<xref ref-type="bibr" rid="B103">Scannell et al. (2022)</xref>; <xref ref-type="bibr" rid="B136">Yang et al. (2023)</xref>
</td>
</tr>
<tr>
<td align="center">Quality (confidence) data</td>
<td align="left">&#x2022; The reliability of the activity data (active or inactive) is crucial to develop predictive models. This is the activity data reproducibility.</td>
<td align="center">
<xref ref-type="bibr" rid="B57">Kumar et al. (2022)</xref>
</td>
</tr>
<tr>
<td align="center">Diversity</td>
<td align="left">&#x2022; Datasets with a high chemical and structural diversity improve the generation of novel molecules.</td>
<td align="center">
<xref ref-type="bibr" rid="B100">Sald&#xed;var-Gonz&#xe1;lez and Medina-Franco (2022)</xref>
</td>
</tr>
<tr>
<td align="center">Preparation or curation</td>
<td align="left">&#x2022; Dataset curation must be focused on one or multiple drug targets. Therefore, molecular descriptors and the cut-off threshold used for the curated must be properly selected.<break/>&#x2022; Dataset should be oriented to resolve specific outcomes and avoid Pan-Assay Interference Compounds (PAINS) structures or chemical structures related to side effects.<break/>&#x2022; In small datasets it is very important to have as much accurate data as possible. The maximum observable accuracy of classification models also depends on the experimental uncertainty and the distribution of the measured values. For instance, datasets with large noise are not recommended for the comparison of different models.</td>
<td align="center">
<xref ref-type="bibr" rid="B31">Fourches et al. (2016)</xref>; <xref ref-type="bibr" rid="B54">Kramer and Lewis (2012)</xref>
</td>
</tr>
<tr>
<td align="center">Complete information</td>
<td align="left">&#x2022; According to the main objective of each project, the dataset used must contain reliable data related to the project&#x2019;s objective. For example, structure containing chemical and physicochemical information, bioactivity data for the related biological endpoint, or outcomes from clinical trials, etc.</td>
<td align="center">
<xref ref-type="bibr" rid="B65">L&#xf3;pez-L&#xf3;pez et al. (2021b)</xref>; <xref ref-type="bibr" rid="B67">L&#xf3;pez-L&#xf3;pez and Medina-Franco. (2023)</xref>; <xref ref-type="bibr" rid="B131">Wu et al. (2023a)</xref>; <xref ref-type="bibr" rid="B132">Wu et al. (2023b)</xref>
</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s4-1">
<title>4.1 Balance</title>
<p>As discussed previously, several current data sets in the public domain are unbalanced due to the infrequent practice of reporting inactive compounds and negative data in general. Historically, the negative and inactive data of preclinical compounds has been ignored by most journals that favor the publication of most active compounds and positive results (<xref ref-type="bibr" rid="B72">Medina-Franco and L&#xf3;pez-L&#xf3;pez, 2022</xref>). However, inactive and negative data are essential in drug design and development. For example, the analysis of high-quality inactive and negative data improves clinical success rate, reduces costs associated with drug development, and reduces the side effects rates (<xref ref-type="bibr" rid="B42">Hayes and Hunter, 2012</xref>; <xref ref-type="bibr" rid="B67">L&#xf3;pez-L&#xf3;pez and Medina-Franco, 2023</xref>). Moreover, data mining and AI approaches are largely benefitted from inactive compounds (<xref ref-type="bibr" rid="B137">Yu, 2021</xref>; <xref ref-type="bibr" rid="B66">L&#xf3;pez-L&#xf3;pez et al., 2022</xref>). The use of inactive and negative data allows real data augmentation to develop AI models, improve their accuracy, and reduce the rate of false-positive cases (<xref ref-type="bibr" rid="B52">Korkmaz, 2020</xref>; <xref ref-type="bibr" rid="B44">
<italic>IBM</italic>, 2022</xref>). Also, the inactive and negative data facilitates the generation of QSPRs models that allows the rationalization of basically any property (<xref ref-type="bibr" rid="B54">Kramer and Lewis, 2012</xref>; <xref ref-type="bibr" rid="B82">Norinder et al., 2019</xref>).</p>
</sec>
<sec id="s4-2">
<title>4.2 Confidence of the activity data</title>
<p>An unwritten rule on AI and computational projects in general is "garbage in, garbage out". This perspective has direct implications in drug design (<xref ref-type="bibr" rid="B5">Bajorath et al., 2022</xref>). Recent studies have demonstrated that the use of quality data allows generating of AI models with higher accuracy than the AI models generated from larger datasets but with low-quality.</p>
</sec>
<sec id="s4-3">
<title>4.3 Chemical and structural diversity</title>
<p>In general, a compound dataset with a large or broad applicability domain, as captured by the diversity of the contents, can give rise to predictive models with a large coverage. This is, molecules from diverse chemical structures could be conveniently interpolated in those models. As a comparison in an experimental setting, high-throughput screening of chemical diverse libraries increases the chances to find hit compounds for targets for which no hit compounds have been previously identified.</p>
<p>Due to the rapid expansion of the chemical universe, recently called the &#x2018;Big Bang&#x2019; of the chemical universe (<xref ref-type="bibr" rid="B23">Cherkasov, 2023</xref>) it is relatively easy to have access to large and diverse regions of the chemical space. However, a practical challenge is to manage such large compound data sets computationally while developing and testing new models. A similar practical problem emerged when combinatorial chemistry was at its peak: it was challenging to design rationally novel large and diverse combinatorial libraries. To tackle this problem numerous diversity selection algorithms have been developed (<xref ref-type="bibr" rid="B58">Leach and Gillet, 2007</xref>). We recently applied a dissimilarity-based compound selection method to obtain three diverse subsets of natural products (with 14,994, 7,497, and 4,998 compounds, respectively) from the UNP. The subsets, that are freely available, can be readily used for <italic>the novo</italic> design applications and as benchmarks for similarity/diversity analysis (<xref ref-type="bibr" rid="B16">Ch&#xe1;vez-Hern&#xe1;ndez and Medina-Franco, 2023</xref>).</p>
</sec>
<sec id="s4-4">
<title>4.4 Preparation or curation</title>
<p>A general curation protocol used on drug discovery datasets is to eliminate duplicate structures, canonize their SMILES representation, eliminate salts, and metals. However, according to the main goal of the <italic>de novo</italic> design model, additional steps to prepare a dataset could be taking into account, for example: 1) eliminating compounds with structural PAINS to reduce the rate of false-positive compounds prediction; 2) deleting compounds reported with side effects and/or ADMET deficiencies, to prioritize the generation of safe and optimization compounds.; or 3) making sure to keep in the dataset compounds with high activity confidence to improve the quality of predicted outputs. This list must be adapted according to the main goal of the <italic>de novo</italic> design model. It is also noted the need to develop robust and consistent protocols that take into scout metal-containing compounds as they have a major role in medicinal inorganic chemistry (<xref ref-type="bibr" rid="B71">Medina-Franco et al., 2022a</xref>).</p>
</sec>
<sec id="s4-5">
<title>4.5 Completeness</title>
<p>Chemical structures should contain the required or relevant information for the goals of the study. For instance, compounds should be annotated with stereochemistry information if the 3D structure and conformation is critical; electronic density and quantum chemical data if the reactivity is key point to predict; the type of the biological activity data such as biochemical, cell-based or functional assays; drug-drug interaction data, pharmacogenomics, or post-marketing annotations; should be aligned with the type of outcome to be predicted and later validated experimentally.</p>
</sec>
</sec>
<sec id="s5">
<title>5 Perspectives of <italic>de novo</italic> design</title>
<p>One of the major perspectives of the <italic>de novo</italic> design is using balanced data sets (as much as experimental data is available) to build reliable models. Similar to QSAR predictive models, it is also crucial the validation of <italic>de novo</italic> protocols using standard and well-curated benchmark datasets (discussed in <xref ref-type="sec" rid="s3-6">Section 3.6</xref>). With the increasing data availability to generate and train new models, it is becoming increasingly easy to explore regions of chemical space previously uncharted and continue contributing to the so-called &#x201c;big bang&#x201d; expansion of the chemical space. A major perspective in this direction is to explore biologically relevant compounds but outside the traditional small molecule chemical space (<xref ref-type="bibr" rid="B73">Medina-Franco et al., 2014</xref>). For instance, exploring metallodrugs (<xref ref-type="bibr" rid="B71">Medina-Franco et al., 2022a</xref>), macrocycles (<xref ref-type="bibr" rid="B61">Liang et al., 2022</xref>), peptides, or the combination of commonly explored chemical spaces, e.g., pseudo-natural products (discussed in <xref ref-type="sec" rid="s3-5">Section 3.5</xref>).</p>
</sec>
<sec sec-type="conclusion" id="s6">
<title>6 Conclusion</title>
<p>Among the main types of datasets used in <italic>the novo</italic> design are on-demand collections, compounds annotated with biological activity, commercially available libraries, and natural products. More recently, a large benchmark data set was developed for machine learning applications. Although there is a general agreement in machine learning that the more data, the better, it is becoming more and more evident to consider the reliability and the quality of the data sets as critical features of the data. Part of the quality is associated with the balance between inactive and active compounds (in a rough analogy with the Yin-Yang concept), tasks that are not always feasible due to the general scarcity of negative (inactive compounds). The later point further emphasizes the continued need to publish and disclose negative results. Due to the fact that the experimental data of inactive compounds are not common, the community is using decoy data sets that by themselves are subject to design and refining using rational approaches. Decoy data sets try to fill the void of experimentally determined inactive molecules. Major criteria to take into account to generate compound data sets with high quality include balanced data sets in terms of active and inactive compounds (when the experimental information is available), structural and chemical diversity, curation or preparation according to the goals of the project, and complete information. All these together contribute to the perspectives of <italic>de novo</italic> design that foresees a continued and rapid expansion of molecules with the potential to become drugs.</p>
</sec>
</body>
<back>
<sec id="s7">
<title>Author contributions</title>
<p>All authors listed have made a substantial, direct, and intellectual contribution to the work and approved it for publication.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>Authors are grateful to DGAPA, UNAM, Programa de Apoyo a Proyectos de Investigaci&#xf3;n e Innovaci&#xf3;n Tecnol&#xf3;gica (PAPIIT), grant no. IN201321. We also thank the Direcci&#xf3;n General de C&#xf3;mputo y de Tecnolog&#xed;as de Informaci&#xf3;n y Comunicaci&#xf3;n (DGTIC), UNAM, for the computational resources to use Miztli supercomputer at UNAM under project LANCAD-UNAM-DGTIC-335; and the innovation space UNAM-HUAWEI the computational resources to use their supercomputer under project-7 &#x201c;<italic>Desarrollo y aplicaci&#xf3;n de algoritmos de inteligencia artificial para el dise&#xf1;o de f&#xe1;rmacos aplicables al tratamiento de diabetes mellitus y c&#xe1;ncer</italic>&#x201d;.</p>
</sec>
<ack>
<p>AC-H and EL-L are thankful to CONACyT, Mexico, for the Ph.D. scholarships number 847870 and 894234, respectively.</p>
</ack>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The author JLM-F declared that he was an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11">
<title>Abbreviations</title>
<p>2D/3D, two-dimensional/three-dimensional; ADMET, absorption, distribution, metabolism, excretion, and toxicity; AI, artificial intelligence; CADD, computer-aided drug design; COCONUT, Collection of Open NatUral ProdUcTs; DNN, deep neural networks; HBA, hydrogen bond acceptors; HBD, hydrogen bond donors; IMPPAT, A curated database of Indian Medicinal Plants, Phytochemistry And Therapeutics; MW, molecular weight; LBDD, ligand-based drug design; log P, octanol-water partition coefficient; NCE, new chemical entities; NIH(US), National Institutes of Health; PAINS, pan-assay interference compounds; Peru NPDB, Peruvian Natural Products Database; QSAR, quantitative structure-activity relationships; REAL, Enamine&#x2019;s REadily AccessibLe; RNNs, recurrent neural networks; SBDD, structure-based drug design; TCM, Traditional Chinese Medicine; TPSA, topological surface area; UNPD, Universal Natural Product Database.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altae-Tran</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ramsundar</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Pappu</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Pande</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Low data drug discovery with one-shot learning</article-title>. <source>ACS central Sci.</source> <volume>3</volume> (<issue>4</issue>), <fpage>283</fpage>&#x2013;<lpage>293</lpage>. <pub-id pub-id-type="doi">10.1021/acscentsci.6b00367</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ar&#xfa;s-Pous</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Patronov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bjerrum</surname>
<given-names>E. J.</given-names>
</name>
<name>
<surname>Tyrchan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Reymond</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>SMILES-based deep generative scaffold decorator for de-novo drug design</article-title>. <source>J. cheminformatics</source> <volume>12</volume> (<issue>1</issue>), <fpage>38</fpage>. <pub-id pub-id-type="doi">10.1186/s13321-020-00441-8</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Atanasov</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Zotchev</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Dirsch</surname>
<given-names>V. M.</given-names>
</name>
<name>
<surname>Supuran</surname>
<given-names>C. T.</given-names>
</name>
</person-group>
<collab>International Natural Product Sciences Taskforce</collab> (<year>2021</year>). <article-title>Natural products in drug discovery: Advances and opportunities</article-title>. <source>Nat. Rev. Drug Discov.</source> <volume>20</volume> (<issue>3</issue>), <fpage>200</fpage>&#x2013;<lpage>216</lpage>. <pub-id pub-id-type="doi">10.1038/s41573-020-00114-z</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Awale</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Reymond</surname>
<given-names>J-L.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Similarity mapplet: Interactive visualization of the directory of useful decoys and ChEMBL in high dimensional chemical spaces</article-title>. <source>J. Chem. Inf. Model.</source> <volume>55</volume> (<issue>8</issue>), <fpage>1509</fpage>&#x2013;<lpage>1516</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.5b00182</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bajorath</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ch&#xe1;vez-Hern&#xe1;ndez</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Duran-Frigola</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fern&#xe1;ndez-de Gortari</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Gasteiger</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>L&#xf3;pez-L&#xf3;pez</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Chemoinformatics and artificial intelligence colloquium: Progress and challenges in developing bioactive compounds</article-title>. <source>J. cheminformatics</source> <volume>14</volume> (<issue>1</issue>), <fpage>82</fpage>. <pub-id pub-id-type="doi">10.1186/s13321-022-00661-0</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Balius</surname>
<given-names>T. E.</given-names>
</name>
<name>
<surname>Allen</surname>
<given-names>W. J.</given-names>
</name>
<name>
<surname>Mukherjee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rizzo</surname>
<given-names>R. C.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Grid-based molecular footprint comparison method for docking and de novo design: Application to HIVgp41</article-title>. <source>J. Comput. Chem.</source> <volume>34</volume> (<issue>14</issue>), <fpage>1226</fpage>&#x2013;<lpage>1240</lpage>. <pub-id pub-id-type="doi">10.1002/jcc.23245</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barazorda-Ccahuana</surname>
<given-names>H. L.</given-names>
</name>
<name>
<surname>Ranilla</surname>
<given-names>L. G.</given-names>
</name>
<name>
<surname>Candia-Puma</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>C&#xe1;rcamo-Rodriguez</surname>
<given-names>E. G.</given-names>
</name>
<name>
<surname>Centeno-Lopez</surname>
<given-names>A. E.</given-names>
</name>
<name>
<surname>Davila-Del-Carpio</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>PeruNPDB: The Peruvian natural products database for <italic>in silico</italic> drug screening</article-title>. <source>Sci. Rep.</source> <volume>13</volume> (<issue>1</issue>), <fpage>7577</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-023-34729-0</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bauer</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Ibrahim</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Vogel</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Boeckler</surname>
<given-names>F. M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Evaluation and optimization of virtual screening workflows with DEKOIS 2.0-a public library of challenging docking benchmark sets</article-title>. <source>J. Chem. Inf. Model.</source> <volume>53</volume> (<issue>6</issue>), <fpage>1447</fpage>&#x2013;<lpage>1462</lpage>. <pub-id pub-id-type="doi">10.1021/ci400115b</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beato</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Beccari</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Cavazzoni</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Lorenzi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Costantino</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Use of experimental design to optimize docking performance: The case of LiGenDock, the docking module of LiGen, a new de novo design program</article-title>. <source>J. Chem. Inf. Model.</source> <volume>53</volume> (<issue>6</issue>), <fpage>1503</fpage>&#x2013;<lpage>1517</lpage>. <pub-id pub-id-type="doi">10.1021/ci400079k</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bender</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schneider</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Segler</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Patrick Walters</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Engkvist</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Rodrigues</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Evaluation guidelines for machine learning tools in the chemical sciences</article-title>. <source>Nat. Rev. Chem.</source> <volume>6</volume> (<issue>6</issue>), <fpage>428</fpage>&#x2013;<lpage>442</lpage>. <pub-id pub-id-type="doi">10.1038/s41570-022-00391-9</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bilodeau</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Jaakkola</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Barzilay</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Jensen</surname>
<given-names>K. F.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Generative models for molecular discovery: Recent advances and challenges</article-title>. <source>Comput. Mol. Sci.</source> <volume>12</volume> (<issue>5</issue>), <fpage>e1608</fpage>. <pub-id pub-id-type="doi">10.1002/wcms.1608</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Blaschke</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ar&#xfa;s-Pous</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Margreitter</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tyrchan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Engkvist</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Reinvent 2.0: An AI tool for de novo drug design</article-title>. <source>J. Chem. Inf. Model.</source> <volume>60</volume> (<issue>12</issue>), <fpage>5918</fpage>&#x2013;<lpage>5922</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.0c00915</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brown</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Fiscato</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Segler</surname>
<given-names>M. H. S.</given-names>
</name>
<name>
<surname>Vaucher</surname>
<given-names>A. C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>GuacaMol: Benchmarking models for de Novo molecular design</article-title>. <source>J. Chem. Inf. Model.</source> <volume>59</volume> (<issue>3</issue>), <fpage>1096</fpage>&#x2013;<lpage>1108</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.8b00839</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Goreshnik</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Coventry</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Case</surname>
<given-names>J. B.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kozodoy</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>De novo design of picomolar SARS-CoV-2 miniprotein inhibitors</article-title>. <source>Science</source> <volume>370</volume> (<issue>6515</issue>), <fpage>426</fpage>&#x2013;<lpage>431</lpage>. <pub-id pub-id-type="doi">10.1126/science.abd9909</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cereto-Massagu&#xe9;</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Guasch</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Valls</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mulero</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pujadas</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Garcia-Vallv&#xe9;</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>DecoyFinder: An easy-to-use python GUI application for building target-specific decoy sets</article-title>. <source>Bioinformatics</source> <volume>28</volume> (<issue>12</issue>), <fpage>1661</fpage>&#x2013;<lpage>1662</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bts249</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ch&#xe1;vez-Hern&#xe1;ndez</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Natural products subsets: Generation and characterization</article-title>. <source>Artif. Intell. Life Sci.</source> <volume>3</volume>, <fpage>100066</fpage>. <pub-id pub-id-type="doi">10.1016/j.ailsci.2023.100066</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ch&#xe1;vez-Hern&#xe1;ndez</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>S&#xe1;nchez-Cruz</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2020a</year>). <article-title>A fragment library of natural products and its comparative chemoinformatic characterization</article-title>. <source>Mol. Inf.</source> <volume>39</volume> (<issue>11</issue>), <fpage>e2000050</fpage>. <pub-id pub-id-type="doi">10.1002/minf.202000050</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ch&#xe1;vez-Hern&#xe1;ndez</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>S&#xe1;nchez-Cruz</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>Fragment library of natural products and compound databases for drug discovery</article-title>. <source>Biomolecules</source> <volume>10</volume> (<issue>11</issue>), <fpage>1518</fpage>. <pub-id pub-id-type="doi">10.3390/biom10111518</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Chemriya</surname>
</name>
</person-group> (<year>2023</year>). <article-title>CHEMriya</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://chemriya.com/">https://chemriya.com/</ext-link> (accessed May 13, 2023)</comment>.</citation>
</ref>
<ref id="B20">
<citation citation-type="web">
<collab>Chemspace</collab> (<year>2023</year>). <article-title>Freedom space</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://chem-space.com/compounds/freedom-space">https://chem-space.com/compounds/freedom-space</ext-link> (accessed May 13, 2023)</comment>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>C. Y-C.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>TCM Database@Taiwan: The world&#x2019;s largest traditional Chinese medicine database for drug screening <italic>in silico</italic>
</article-title>. <source>PloS one</source> <volume>6</volume> (<issue>1</issue>), <fpage>e15939</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0015939</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gilson</surname>
<given-names>M. K.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>The binding database: Data management and interface design</article-title>. <source>Bioinformatics</source> <volume>18</volume> (<issue>1</issue>), <fpage>130</fpage>&#x2013;<lpage>139</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/18.1.130</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cherkasov</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>The &#x2018;Big Bang&#x2019; of the chemical universe</article-title>. <source>Nat. Chem. Biol.</source> <volume>19</volume>, <fpage>667</fpage>&#x2013;<lpage>668</lpage>. <pub-id pub-id-type="doi">10.1038/s41589-022-01233-x</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Corso</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>St&#xe4;rk</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jing</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <source>DiffDock: Diffusion steps, twists, and turns for molecular docking</source>. <comment>arXiv [q-bio.BM]. Available at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/2210.01776">http://arxiv.org/abs/2210.01776</ext-link>
</comment>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Costa</surname>
<given-names>R. P. O.</given-names>
</name>
<name>
<surname>Lucena</surname>
<given-names>L. F.</given-names>
</name>
<name>
<surname>Silva</surname>
<given-names>L. M. A.</given-names>
</name>
<name>
<surname>Zocolo</surname>
<given-names>G. J.</given-names>
</name>
<name>
<surname>Herrera-Acevedo</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Scotti</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>The SistematX web portal of natural products: An update</article-title>. <source>J. Chem. Inf. Model.</source> <volume>61</volume> (<issue>6</issue>), <fpage>2516</fpage>&#x2013;<lpage>2522</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.1c00083</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Davies</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nowotka</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Papadatos</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Dedman</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Gaulton</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Atkinson</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>ChEMBL web services: Streamlining access to drug discovery data and utilities</article-title>. <source>Nucleic acids Res.</source> <volume>43</volume> (<issue>W1</issue>), <fpage>W612</fpage>&#x2013;<lpage>W620</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkv352</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dos Santos Nascimento</surname>
<given-names>I. J.</given-names>
</name>
<name>
<surname>de Aquino</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>da Silva-J&#xfa;nior</surname>
<given-names>E. F.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Drug repurposing: A strategy for discovering inhibitors against emerging viral infections</article-title>. <source>Curr. Med. Chem.</source> <volume>28</volume> (<issue>15</issue>), <fpage>2887</fpage>&#x2013;<lpage>2942</lpage>. <pub-id pub-id-type="doi">10.2174/0929867327666200812215852</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="web">
<collab>DRUGBANK</collab> (<year>2023</year>). <article-title>Celecoxib</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://go.drugbank.com/drugs/DB00482">https://go.drugbank.com/drugs/DB00482</ext-link> (accessed May 13, 2023)</comment>.</citation>
</ref>
<ref id="B29">
<citation citation-type="web">
<collab>Enamine</collab> (<year>2023</year>). <article-title>Real database</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://enamine.net/compound-collections/real-compounds/real-database">https://enamine.net/compound-collections/real-compounds/real-database</ext-link> (accessed May 13, 2023)</comment>.</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Evans</surname>
<given-names>B. E.</given-names>
</name>
<name>
<surname>Rittle</surname>
<given-names>K. E.</given-names>
</name>
<name>
<surname>Bock</surname>
<given-names>M. G.</given-names>
</name>
<name>
<surname>DiPardo</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Freidinger</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Whitter</surname>
<given-names>W. L.</given-names>
</name>
<etal/>
</person-group> (<year>1988</year>). <article-title>Methods for drug discovery: Development of potent, selective, orally effective cholecystokinin antagonists</article-title>. <source>J. Med. Chem.</source> <volume>31</volume> (<issue>12</issue>), <fpage>2235</fpage>&#x2013;<lpage>2246</lpage>. <pub-id pub-id-type="doi">10.1021/jm00120a002</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fourches</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Muratov</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Tropsha</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Trust, but verify II: A practical guide to chemogenomics data curation</article-title>. <source>J. Chem. Inf. Model.</source> <volume>56</volume> (<issue>7</issue>), <fpage>1243</fpage>&#x2013;<lpage>1252</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.6b00129</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gallo</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kemmler</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Goede</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Becker</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Dunkel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Preissner</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>SuperNatural 3.0-a database of natural products and natural product-based derivatives</article-title>. <source>Nucleic acids Res.</source> <volume>51</volume> (<issue>D1</issue>), <fpage>D654</fpage>&#x2013;<lpage>D659</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkac1008</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>G&#xf3;mez-Bombarelli</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Duvenaud</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hern&#xe1;ndez-Lobato</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>S&#xe1;nchez-Lengeling</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sheberla</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Automatic chemical design using a data-driven continuous representation of molecules</article-title>. <source>ACS central Sci.</source> <volume>4</volume> (<issue>2</issue>), <fpage>268</fpage>&#x2013;<lpage>276</lpage>. <pub-id pub-id-type="doi">10.1021/acscentsci.7b00572</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>G&#xf3;mez-Garc&#xed;a</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Progress and impact of Latin American natural product databases</article-title>. <source>Biomolecules</source> <volume>12</volume> (<issue>9</issue>), <fpage>1202</fpage>. <pub-id pub-id-type="doi">10.3390/biom12091202</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Grebner</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Webinar: "exploration and mining of large virtual chemical spaces</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://youtu.be/fMrI11SXwpU">https://youtu.be/fMrI11SXwpU</ext-link> (accessed May 13, 2023)</comment>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Greener</surname>
<given-names>J. G.</given-names>
</name>
<name>
<surname>Kandathil</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Moffat</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>D. T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A guide to machine learning for biologists</article-title>. <source>Nat. Rev. Mol. Cell Biol.</source> <volume>23</volume> (<issue>1</issue>), <fpage>40</fpage>&#x2013;<lpage>55</lpage>. <pub-id pub-id-type="doi">10.1038/s41580-021-00407-0</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grigalunas</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Brakmann</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Waldmann</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Chemical evolution of natural product structure</article-title>. <source>J. Am. Chem. Soc.</source> <volume>144</volume> (<issue>8</issue>), <fpage>3314</fpage>&#x2013;<lpage>3329</lpage>. <pub-id pub-id-type="doi">10.1021/jacs.1c11270</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gui</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>H. Z.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Use of natural products as chemical library for drug discovery and network pharmacology</article-title>. <source>PloS one</source> <volume>8</volume> (<issue>4</issue>), <fpage>e62839</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0062839</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo J</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Janet</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Bauer</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Nittinger</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Giblin</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Papadopoulos</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>DockStream: A docking wrapper to enhance de novo molecular design</article-title>. <source>J. cheminformatics</source> <volume>13</volume> (<issue>1</issue>), <fpage>89</fpage>. <pub-id pub-id-type="doi">10.1186/s13321-021-00563-7</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Guo M</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Thost</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). &#x201c;<article-title>Data-efficient graph grammar learning for molecular generation</article-title>,&#x201d; in <source>International conference on learning representations</source>, <volume>9</volume>. <comment>February 2021. Available at: <ext-link ext-link-type="uri" xlink:href="https://research.ibm.com/publications/data-efficient-graph-grammar-learning-for-molecular-generation">https://research.ibm.com/publications/data-efficient-graph-grammar-learning-for-molecular-generation</ext-link> (accessed May 13, 2023)</comment>.</citation>
</ref>
<ref id="B41">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Thost</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Data-efficient graph grammar learning for molecular generation</article-title>. <comment>arXiv [cs.LG]. Available at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/2203.08031">http://arxiv.org/abs/2203.08031</ext-link>
</comment>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hayes</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hunter</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Why is publication of negative clinical trial data important?</article-title> <source>Br. J. Pharmacol.</source> <volume>167</volume> (<issue>7</issue>), <fpage>1395</fpage>&#x2013;<lpage>1397</lpage>. <pub-id pub-id-type="doi">10.1111/j.1476-5381.2012.02215.x</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Sutton</surname>
<given-names>S. C.</given-names>
</name>
<name>
<surname>Na</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kostrowicki</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Pfizer global virtual library (PGVL): A chemistry design tool powered by experimentally validated parallel synthesis information</article-title>. <source>ACS Comb. Sci.</source> <volume>14</volume> (<issue>11</issue>), <fpage>579</fpage>&#x2013;<lpage>589</lpage>. <pub-id pub-id-type="doi">10.1021/co300096q</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="web">
<collab>IBM</collab> (<year>2022</year>). <article-title>How to use AI to discover new drugs and materials with limited data</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://research.ibm.com/blog/ai-discovery-with-limited-data#fnref-1">https://research.ibm.com/blog/ai-discovery-with-limited-data&#x23;fnref-1</ext-link> (accessed April 16, 2023)</comment>.</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Irwin</surname>
<given-names>J. J.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Community benchmarks for virtual screening</article-title>. <source>J. computer-aided Mol. Des.</source> <volume>22</volume> (<issue>3-4</issue>), <fpage>193</fpage>&#x2013;<lpage>199</lpage>. <pub-id pub-id-type="doi">10.1007/s10822-008-9189-4</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jain</surname>
<given-names>A. N.</given-names>
</name>
<name>
<surname>Nicholls</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Recommendations for evaluation of computational methods</article-title>. <source>J. computer-aided Mol. Des.</source> <volume>22</volume> (<issue>3-4</issue>), <fpage>133</fpage>&#x2013;<lpage>139</lpage>. <pub-id pub-id-type="doi">10.1007/s10822-008-9196-5</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jumper</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Evans</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pritzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Figurnov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ronneberger</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Highly accurate protein structure prediction with AlphaFold</article-title>. <source>Nature</source> <volume>596</volume> (<issue>7873</issue>), <fpage>583</fpage>&#x2013;<lpage>589</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-021-03819-2</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Juskalian</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Regalado</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Orcutt</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <source>10 breakthrough technologies 2020</source>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://www.technologyreview.com/10-breakthrough-technologies/2020/">https://www.technologyreview.com/10-breakthrough-technologies/2020/</ext-link>
</comment> (<comment>accessed February 26, 2020)</comment>.</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kadurin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Aliper</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kazennov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mamoshina</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Vanhaelen</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Khrabrov</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>The cornucopia of meaningful leads: Applying deep adversarial autoencoders for new molecule development in oncology</article-title>. <source>Oncotarget</source> <volume>8</volume> (<issue>7</issue>), <fpage>10883</fpage>&#x2013;<lpage>10890</lpage>. <pub-id pub-id-type="doi">10.18632/oncotarget.14073</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Gindulyte</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>PubChem 2023 update</article-title>. <source>Nucleic acids Res.</source> <volume>51</volume> (<issue>D1</issue>), <fpage>D1373</fpage>&#x2013;<lpage>D1380</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkac956</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koes</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Camacho</surname>
<given-names>C. J.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>ZINCPharmer: Pharmacophore search of the ZINC database</article-title>. <source>Nucleic acids Res.</source> <volume>40</volume>, <fpage>W409</fpage>&#x2013;<lpage>W414</lpage>. <comment>Web Server issue)</comment>. <pub-id pub-id-type="doi">10.1093/nar/gks378</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Korkmaz</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deep learning-based imbalanced data classification for drug discovery</article-title>. <source>J. Chem. Inf. Model.</source> <volume>60</volume> (<issue>9</issue>), <fpage>4180</fpage>&#x2013;<lpage>4190</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.9b01162</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Korn</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ehrt</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ruggiu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Gastreich</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rarey</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Navigating large chemical spaces in early-phase drug discovery</article-title>. <source>Curr. Opin. Struct. Biol.</source> <volume>80</volume>, <fpage>102578</fpage>. <pub-id pub-id-type="doi">10.1016/j.sbi.2023.102578</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kramer</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Lewis</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>QSARs, data and error in the modern age of drug discovery</article-title>. <source>Curr. Top. Med. Chem.</source> <volume>12</volume> (<issue>17</issue>), <fpage>1896</fpage>&#x2013;<lpage>1902</lpage>. <pub-id pub-id-type="doi">10.2174/156802612804547380</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krenn</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>H&#xe4;se</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Nigam</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Friederich</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Aspuru-Guzik</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Self-referencing embedded strings (SELFIES): A 100% robust molecular string representation</article-title>. <source>Mach. Learn. Sci. Technol.</source> <volume>1</volume> (<issue>4</issue>), <fpage>045024</fpage>. <pub-id pub-id-type="doi">10.1088/2632-2153/aba947</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krishnan</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Bung</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Bulusu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Roy</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Accelerating de novo drug design against novel proteins using deep learning</article-title>. <source>J. Chem. Inf. Model.</source> <volume>61</volume> (<issue>2</issue>), <fpage>621</fpage>&#x2013;<lpage>630</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.0c01060</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Ananda Kumar</surname>
<given-names>T. D.</given-names>
</name>
<name>
<surname>Beeraka</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Pujar</surname>
<given-names>G. V.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Narayana Akshatha</surname>
<given-names>H. S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Machine learning and deep learning in data-driven decision making of drug discovery and challenges in high-quality data acquisition in the pharmaceutical industry</article-title>. <source>Future Med. Chem.</source> <volume>14</volume> (<issue>4</issue>), <fpage>245</fpage>&#x2013;<lpage>270</lpage>. <pub-id pub-id-type="doi">10.4155/fmc-2021-0243</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Leach</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Gillet</surname>
<given-names>V. J.</given-names>
</name>
</person-group> (<year>2007</year>). &#x201c;<article-title>Selecting diverse dets of compounds</article-title>,&#x201d; in <source>An introduction to chemoinformatics</source> (<publisher-loc>Dordrecht</publisher-loc>: <publisher-name>Springer Netherlands</publisher-name>), <fpage>119</fpage>&#x2013;<lpage>139</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-4020-6291-9_6</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>De Novo design of potential inhibitors against SARS-CoV-2 Mpro</article-title>. <source>Comput. Biol. Med.</source> <volume>147</volume>, <fpage>105728</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.105728</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Multi-objective de novo drug design with conditional graph generative model</article-title>. <source>J. cheminformatics</source> <volume>10</volume> (<issue>1</issue>), <fpage>33</fpage>. <pub-id pub-id-type="doi">10.1186/s13321-018-0287-6</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Rao</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>An insight into the medicinal chemistry perspective of macrocyclic derivatives with antitumor activity: A systematic review</article-title>. <source>Molecules</source> <volume>27</volume> (<issue>9</issue>), <fpage>2837</fpage>. <pub-id pub-id-type="doi">10.3390/molecules27092837</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lipinski</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Lombardo</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Dominy</surname>
<given-names>B. W.</given-names>
</name>
<name>
<surname>Feeney</surname>
<given-names>P. J.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Experimental and computational approaches to estimate solubility and permeability in drug discovery and development settings</article-title>. <source>Adv. drug Deliv. Rev.</source> <volume>46</volume> (<issue>1-3</issue>), <fpage>3</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1016/s0169-409x(00)00129-0</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>van Vlijmen</surname>
<given-names>H. W. T.</given-names>
</name>
<name>
<surname>Ijzerman</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>van Westen</surname>
<given-names>G. J. P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>An exploration strategy improves the diversity of de novo ligands using deep reinforcement learning: A case for the adenosine A2A receptor</article-title>. <source>J. cheminformatics</source> <volume>11</volume> (<issue>1</issue>), <fpage>35</fpage>. <pub-id pub-id-type="doi">10.1186/s13321-019-0355-6</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>L&#xf3;pez-L&#xf3;pez</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Bajorath</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2021a</year>). <article-title>Informatics for chemistry, biology, and biomedical sciences</article-title>. <source>J. Chem. Inf. Model.</source> <volume>61</volume> (<issue>1</issue>), <fpage>26</fpage>&#x2013;<lpage>35</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.0c01301</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>L&#xf3;pez-L&#xf3;pez</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Cerda-Garc&#xed;a-Rojas</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2021b</year>). <article-title>Tubulin inhibitors: A chemoinformatic analysis using cell-based data</article-title>. <source>Molecules</source> <volume>26</volume> (<issue>9</issue>), <fpage>2483</fpage>. <pub-id pub-id-type="doi">10.3390/molecules26092483</pub-id>
</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>L&#xf3;pez-L&#xf3;pez</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Fern&#xe1;ndez-de Gortari</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Yes SIR! On the structure-inactivity relationships in drug discovery</article-title>. <source>Drug Discov. today</source> <volume>27</volume> (<issue>8</issue>), <fpage>2353</fpage>&#x2013;<lpage>2362</lpage>. <pub-id pub-id-type="doi">10.1016/j.drudis.2022.05.005</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>L&#xf3;pez-L&#xf3;pez</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Towards decoding hepatotoxicity of approved drugs through navigation of multiverse and consensus chemical spaces</article-title>. <source>Biomolecules</source> <volume>13</volume> (<issue>1</issue>), <fpage>176</fpage>. <pub-id pub-id-type="doi">10.3390/biom13010176</pub-id>
</citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Terayama</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Matsumoto</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Isaka</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sasakura</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Iwata</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Structure-based de novo molecular generator combined with artificial intelligence and docking simulations</article-title>. <source>J. Chem. Inf. Model.</source> <volume>61</volume> (<issue>7</issue>), <fpage>3304</fpage>&#x2013;<lpage>3313</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.1c00679</pub-id>
</citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maziarka</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Pocha</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kaczmarczyk</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rataj</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Danel</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Warcho&#x142;</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Mol-CycleGAN: A generative model for molecular optimization</article-title>. <source>J. cheminformatics</source> <volume>12</volume> (<issue>1</issue>), <fpage>2</fpage>. <pub-id pub-id-type="doi">10.1186/s13321-019-0404-1</pub-id>
</citation>
</ref>
<ref id="B70">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Flores-Padilla</surname>
<given-names>E. A.</given-names>
</name>
<name>
<surname>Ch&#xe1;vez-Hern&#xe1;ndez</surname>
<given-names>A. L.</given-names>
</name>
</person-group> (<year>2022b</year>). &#x201c;<article-title>Chapter 23 - discovery and development of lead compounds from natural sources using computational approaches</article-title>,&#x201d; in <source>Evidence-based validation of herbal medicine</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Mukherjee</surname>
<given-names>P. K.</given-names>
</name>
</person-group> <edition>Second Edition</edition> (<publisher-name>Elsevier</publisher-name>), <fpage>539</fpage>&#x2013;<lpage>560</lpage>. <pub-id pub-id-type="doi">10.1016/B978-0-323-85542-6.00009-3</pub-id>
</citation>
</ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>L&#xf3;pez-L&#xf3;pez</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Andrade</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ruiz-Azuara</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Frei</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Guan</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2022a</year>). <article-title>Bridging informatics and medicinal inorganic chemistry: Toward a database of metallodrugs and metallodrug candidates</article-title>. <source>Drug Discov. today</source> <volume>27</volume> (<issue>5</issue>), <fpage>1420</fpage>&#x2013;<lpage>1430</lpage>. <pub-id pub-id-type="doi">10.1016/j.drudis.2022.02.021</pub-id>
</citation>
</ref>
<ref id="B72">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>L&#xf3;pez-L&#xf3;pez</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>The essence and transcendence of scientific publishing</article-title>. <source>Front. Res. metrics Anal.</source> <volume>7</volume>, <fpage>822453</fpage>. <pub-id pub-id-type="doi">10.3389/frma.2022.822453</pub-id>
</citation>
</ref>
<ref id="B73">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Martinez-Mayorga</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Meurice</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Balancing novelty with confined chemical space in modern drug discovery</article-title>. <source>Expert Opin. drug Discov.</source> <volume>9</volume> (<issue>2</issue>), <fpage>151</fpage>&#x2013;<lpage>165</lpage>. <pub-id pub-id-type="doi">10.1517/17460441.2014.872624</pub-id>
</citation>
</ref>
<ref id="B74">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Naveja</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>L&#xf3;pez-L&#xf3;pez</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Reaching for the bright StARs in chemical space</article-title>. <source>Drug Discov. today</source> <volume>24</volume> (<issue>11</issue>), <fpage>2162</fpage>&#x2013;<lpage>2169</lpage>. <pub-id pub-id-type="doi">10.1016/j.drudis.2019.09.013</pub-id>
</citation>
</ref>
<ref id="B75">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mendez</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Gaulton</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bento</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Chambers</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>De Veij</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>F&#xe9;lix</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>ChEMBL: Towards direct deposition of bioassay data</article-title>. <source>Nucleic acids Res.</source> <volume>47</volume> (<issue>D1</issue>), <fpage>D930</fpage>&#x2013;<lpage>D940</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky1075</pub-id>
</citation>
</ref>
<ref id="B76">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mohanraj</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Karthikeyan</surname>
<given-names>B. S.</given-names>
</name>
<name>
<surname>Vivek-Ananth</surname>
<given-names>R. P.</given-names>
</name>
<name>
<surname>Chand</surname>
<given-names>R. P. B.</given-names>
</name>
<name>
<surname>Aparna</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Mangalapandi</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Imppat: A curated database of indian medicinal plants, phytochemistry and therapeutics</article-title>. <source>Sci. Rep.</source> <volume>8</volume> (<issue>1</issue>), <fpage>4329</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-018-22631-z</pub-id>
</citation>
</ref>
<ref id="B77">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mouchlis</surname>
<given-names>V. D.</given-names>
</name>
<name>
<surname>Afantitis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Serra</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fratello</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Papadiamantis</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Aidinis</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Advances in de novo drug design: From conventional to machine learning methods</article-title>. <source>Int. J. Mol. Sci.</source> <volume>22</volume> (<issue>4</issue>), <fpage>1676</fpage>. <pub-id pub-id-type="doi">10.3390/ijms22041676</pub-id>
</citation>
</ref>
<ref id="B78">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mysinger</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Carchia</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Irwin</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Shoichet</surname>
<given-names>B. K.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Directory of useful decoys, enhanced (DUD-E): Better ligands and decoys for better benchmarking</article-title>. <source>J. Med. Chem.</source> <volume>55</volume> (<issue>14</issue>), <fpage>6582</fpage>&#x2013;<lpage>6594</lpage>. <pub-id pub-id-type="doi">10.1021/jm300687e</pub-id>
</citation>
</ref>
<ref id="B79">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Newman</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Cragg</surname>
<given-names>G. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Natural products as sources of new drugs over the nearly four decades from 01/1981 to 09/2019</article-title>. <source>J. Nat. Prod.</source> <volume>83</volume> (<issue>3</issue>), <fpage>770</fpage>&#x2013;<lpage>803</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jnatprod.9b01285</pub-id>
</citation>
</ref>
<ref id="B80">
<citation citation-type="web">
<collab>NIH</collab> (<year>2023</year>). <article-title>Natural product libraries</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://www.nccih.nih.gov/grants/natural-product-libraries">https://www.nccih.nih.gov/grants/natural-product-libraries</ext-link>
</comment>.</citation>
</ref>
<ref id="B81">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Niitsu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sugita</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Towards de novo design of transmembrane &#x3b1;-helical assemblies using structural modelling and molecular dynamics simulation</article-title>. <source>Phys. Chem. Chem. Phys. PCCP</source> <volume>25</volume> (<issue>5</issue>), <fpage>3595</fpage>&#x2013;<lpage>3606</lpage>. <pub-id pub-id-type="doi">10.1039/d2cp03972a</pub-id>
</citation>
</ref>
<ref id="B82">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Norinder</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Naveja</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>L&#xf3;pez-L&#xf3;pez</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Mucs</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Conformal prediction of HDAC inhibitors</article-title>. <source>SAR QSAR Environ. Res.</source> <volume>30</volume> (<issue>4</issue>), <fpage>265</fpage>&#x2013;<lpage>277</lpage>. <pub-id pub-id-type="doi">10.1080/1062936X.2019.1591503</pub-id>
</citation>
</ref>
<ref id="B83">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ntie-Kang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zofou</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Babiaka</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Meudom</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Scharfe</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lifongo</surname>
<given-names>L. L.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>AfroDb: A select highly potent and diverse natural product library from african medicinal plants</article-title>. <source>PloS one</source> <volume>8</volume> (<issue>10</issue>), <fpage>e78085</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0078085</pub-id>
</citation>
</ref>
<ref id="B84">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Olivecrona</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Blaschke</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Engkvist</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Molecular de-novo design through deep reinforcement learning</article-title>. <source>J. cheminformatics</source> <volume>9</volume> (<issue>1</issue>), <fpage>48</fpage>. <pub-id pub-id-type="doi">10.1186/s13321-017-0235-x</pub-id>
</citation>
</ref>
<ref id="B85">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Olmedo</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Chemoinformatic approach: The case of natural products of Panama</article-title>,&#x201d; in <source>Cheminformatics and its applications</source> (<publisher-name>IntechOpen</publisher-name>). <pub-id pub-id-type="doi">10.5772/intechopen.87779</pub-id>
</citation>
</ref>
<ref id="B86">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Olmedo</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Gonz&#xe1;lez-Medina</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>M. P.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Cheminformatic characterization of natural products from Panama</article-title>. <source>Mol. Divers.</source> <volume>21</volume> (<issue>4</issue>), <fpage>779</fpage>&#x2013;<lpage>789</lpage>. <pub-id pub-id-type="doi">10.1007/s11030-017-9781-4</pub-id>
</citation>
</ref>
<ref id="B87">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Palazzesi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Pozzan</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Deep learning applied to ligand-based de novo drug DesignDe novo drug design</article-title>,&#x201d; in <source>Artificial intelligence in drug design</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Heifetz</surname>
<given-names>A.</given-names>
</name>
</person-group> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Springer US</publisher-name>), <fpage>273</fpage>&#x2013;<lpage>299</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-0716-1787-8_12</pub-id>
</citation>
</ref>
<ref id="B88">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Papadopoulos</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Giblin</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Janet</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Patronov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Engkvist</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>De novo design with deep generative models based on 3D similarity scoring</article-title>. <source>Bioorg. Med. Chem.</source> <volume>44</volume>, <fpage>116308</fpage>. <pub-id pub-id-type="doi">10.1016/j.bmc.2021.116308</pub-id>
</citation>
</ref>
<ref id="B89">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Patel</surname>
<given-names>H. M.</given-names>
</name>
<name>
<surname>Noolvi</surname>
<given-names>M. N.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Jaiswal</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Bansal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lohan</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Quantitative structure&#x2013;activity relationship (QSAR) studies as strategic approach in drug discovery</article-title>. <source>Med. Chem. Res. Int. J. rapid Commun. Des. Mech. action Biol. Act. agents</source> <volume>23</volume> (<issue>12</issue>), <fpage>4991</fpage>&#x2013;<lpage>5007</lpage>. <pub-id pub-id-type="doi">10.1007/s00044-014-1072-3</pub-id>
</citation>
</ref>
<ref id="B90">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Perron</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>da Silva</surname>
<given-names>V. B. R.</given-names>
</name>
<name>
<surname>Atwood</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gaston-Math&#xe9;</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>Key points to succeed in Artificial Intelligence drug discovery projects</article-title>. <source>Chem. Int.</source> <volume>44</volume> (<issue>1</issue>), <fpage>19</fpage>&#x2013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.1515/ci-2022-0106</pub-id>
</citation>
</ref>
<ref id="B91">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Perron</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Mirguet</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Tajmouati</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Skiredj</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rojas</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gohier</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2022a</year>). <article-title>Deep generative models for ligand-based de novo design applied to multi-parametric optimization</article-title>. <source>J. Comput. Chem.</source> <volume>43</volume> (<issue>10</issue>), <fpage>692</fpage>&#x2013;<lpage>703</lpage>. <pub-id pub-id-type="doi">10.1002/jcc.26826</pub-id>
</citation>
</ref>
<ref id="B92">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pilon</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Valli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dametto</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Pinto</surname>
<given-names>M. E. F.</given-names>
</name>
<name>
<surname>Freire</surname>
<given-names>R. T.</given-names>
</name>
<name>
<surname>Castro-Gamboa</surname>
<given-names>I.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>NuBBEDB: An updated database to uncover chemical and biological information from Brazilian biodiversity</article-title>. <source>Sci. Rep.</source> <volume>7</volume> (<issue>1</issue>), <fpage>7215</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-017-07451-x</pub-id>
</citation>
</ref>
<ref id="B93">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pil&#xf3;n-Jim&#xe9;nez</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Sald&#xed;var-Gonz&#xe1;lez</surname>
<given-names>F. I.</given-names>
</name>
<name>
<surname>D&#xed;az-Eufracio</surname>
<given-names>B. I.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Biofacquim: A Mexican compound database of natural products</article-title>. <source>Biomolecules</source> <volume>9</volume> (<issue>1</issue>), <fpage>31</fpage>. <pub-id pub-id-type="doi">10.3390/biom9010031</pub-id>
</citation>
</ref>
<ref id="B94">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Polykovskiy</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhebrak</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sanchez-Lengeling</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Golovanov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tatanov</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Belyaev</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Molecular sets (MOSES): A benchmarking platform for molecular generation models</article-title>. <source>Front. Pharmacol.</source> <volume>11</volume>, <fpage>565644</fpage>. <pub-id pub-id-type="doi">10.3389/fphar.2020.565644</pub-id>
</citation>
</ref>
<ref id="B95">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>R&#xe9;au</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Langenfeld</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zagury</surname>
<given-names>J-F.</given-names>
</name>
<name>
<surname>Lagarde</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Montes</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Decoys selection in benchmarking datasets: Overview and perspectives</article-title>. <source>Front. Pharmacol.</source> <volume>9</volume>, <fpage>11</fpage>. <pub-id pub-id-type="doi">10.3389/fphar.2018.00011</pub-id>
</citation>
</ref>
<ref id="B96">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reymond</surname>
<given-names>J-L.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>The chemical space project</article-title>. <source>Accounts Chem. Res.</source> <volume>48</volume> (<issue>3</issue>), <fpage>722</fpage>&#x2013;<lpage>730</lpage>. <pub-id pub-id-type="doi">10.1021/ar500432k</pub-id>
</citation>
</ref>
<ref id="B97">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rohrer</surname>
<given-names>S. G.</given-names>
</name>
<name>
<surname>Baumann</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Maximum unbiased validation (MUV) data sets for virtual screening based on PubChem bioactivity data</article-title>. <source>J. Chem. Inf. Model.</source> <volume>49</volume> (<issue>2</issue>), <fpage>169</fpage>&#x2013;<lpage>184</lpage>. <pub-id pub-id-type="doi">10.1021/ci8002649</pub-id>
</citation>
</ref>
<ref id="B98">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sabe</surname>
<given-names>V. T.</given-names>
</name>
<name>
<surname>Ntombela</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Jhamba</surname>
<given-names>L. A.</given-names>
</name>
<name>
<surname>Maguire</surname>
<given-names>G. E. M.</given-names>
</name>
<name>
<surname>Govender</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Naicker</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Current trends in computer aided drug design and a highlight of drugs discovered via computational techniques: A review</article-title>. <source>Eur. J. Med. Chem.</source> <volume>224</volume>, <fpage>113705</fpage>. <pub-id pub-id-type="doi">10.1016/j.ejmech.2021.113705</pub-id>
</citation>
</ref>
<ref id="B99">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sald&#xed;var-Gonz&#xe1;lez</surname>
<given-names>F. I.</given-names>
</name>
<name>
<surname>Aldas-Bulos</surname>
<given-names>V. D.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Plisson</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Natural product drug discovery in the artificial intelligence era</article-title>. <source>Chem. Sci.</source> <volume>13</volume> (<issue>6</issue>), <fpage>1526</fpage>&#x2013;<lpage>1546</lpage>. <pub-id pub-id-type="doi">10.1039/d1sc04471k</pub-id>
</citation>
</ref>
<ref id="B100">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sald&#xed;var-Gonz&#xe1;lez</surname>
<given-names>F. I.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Approaches for enhancing the analysis of chemical space for drug discovery</article-title>. <source>Expert Opin. drug Discov.</source> <volume>17</volume> (<issue>7</issue>), <fpage>789</fpage>&#x2013;<lpage>798</lpage>. <pub-id pub-id-type="doi">10.1080/17460441.2022.2084608</pub-id>
</citation>
</ref>
<ref id="B101">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sald&#xed;var-Gonz&#xe1;lez</surname>
<given-names>F. I.</given-names>
</name>
<name>
<surname>Valli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Andricopulo</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>da Silva Bolzani</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Chemical space and diversity of the NuBBE database: A chemoinformatic characterization</article-title>. <source>J. Chem. Inf. Model.</source> <volume>59</volume> (<issue>1</issue>), <fpage>74</fpage>&#x2013;<lpage>85</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.8b00619</pub-id>
</citation>
</ref>
<ref id="B102">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>S&#xe1;nchez-Cruz</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Pil&#xf3;n-Jim&#xe9;nez</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Medina-Franco</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2019</year>) <article-title>Functional group and diversity analysis of BIOFACQUIM: A Mexican natural product database</article-title>. <source>F1000Research</source> <volume>8</volume>, <fpage>Chem Inf Sci-2071</fpage>. <pub-id pub-id-type="doi">10.12688/f1000research.21540.2</pub-id>
</citation>
</ref>
<ref id="B103">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Scannell</surname>
<given-names>J. W.</given-names>
</name>
<name>
<surname>Bosley</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hickman</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Dawson</surname>
<given-names>G. R.</given-names>
</name>
<name>
<surname>Truebel</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>G. S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Predictive validity in drug discovery: What it is, why it matters and how to improve it</article-title>. <source>Nat. Rev. Drug Discov.</source> <volume>21</volume> (<issue>12</issue>), <fpage>915</fpage>&#x2013;<lpage>931</lpage>. <pub-id pub-id-type="doi">10.1038/s41573-022-00552-x</pub-id>
</citation>
</ref>
<ref id="B104">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schneider</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Clark</surname>
<given-names>D. E.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Automated de novo drug design: Are we nearly there yet?</article-title> <source>Angew. Chem.</source> <volume>58</volume> (<issue>32</issue>), <fpage>10792</fpage>&#x2013;<lpage>10803</lpage>. <pub-id pub-id-type="doi">10.1002/anie.201814681</pub-id>
</citation>
</ref>
<ref id="B105">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schneider</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cl&#xe9;ment-Chomienne</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Hilfiger</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Schneider</surname>
</name>
<name>
<surname>Kirsch</surname>
</name>
<name>
<surname>B&#xf6;hm</surname>
</name>
<etal/>
</person-group> (<year>2000</year>). <article-title>Virtual screening for bioactive molecules by evolutionary de novo design</article-title>. <source>Angew. Chem.</source> <volume>39</volume> (<issue>22</issue>), <fpage>4130</fpage>&#x2013;<lpage>4133</lpage>. <pub-id pub-id-type="doi">10.1002/1521-3773(20001117)39:22&#x3c;4130:aid-anie4130&#x3e;3.0.co;2-e</pub-id>
</citation>
</ref>
<ref id="B106">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schneider</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Schneider</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Privileged structures revisited</article-title>. <source>Angew. Chem.</source> <volume>56</volume> (<issue>27</issue>), <fpage>7971</fpage>&#x2013;<lpage>7974</lpage>. <pub-id pub-id-type="doi">10.1002/anie.201702816</pub-id>
</citation>
</ref>
<ref id="B107">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schneider</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Walters</surname>
<given-names>W. P.</given-names>
</name>
<name>
<surname>Plowright</surname>
<given-names>A. T.</given-names>
</name>
<name>
<surname>Sieroka</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Listgarten</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Goodnow</surname>
<given-names>R. A.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Rethinking drug design in the artificial intelligence era</article-title>. <source>Nat. Rev. Drug Discov.</source> <volume>19</volume> (<issue>5</issue>), <fpage>353</fpage>&#x2013;<lpage>364</lpage>. <pub-id pub-id-type="doi">10.1038/s41573-019-0050-3</pub-id>
</citation>
</ref>
<ref id="B108">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Scotti</surname>
<given-names>M. T.</given-names>
</name>
<name>
<surname>Herrera-Acevedo</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Oliveira</surname>
<given-names>T. B.</given-names>
</name>
<name>
<surname>Costa</surname>
<given-names>R. P. O.</given-names>
</name>
<name>
<surname>Santos</surname>
<given-names>S. Y. K. d. O.</given-names>
</name>
<name>
<surname>Rodrigues</surname>
<given-names>R. P.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>SistematX, an online web-based cheminformatics tool for data management of secondary metabolites</article-title>. <source>Molecules</source> <volume>23</volume> (<issue>1</issue>), <fpage>103</fpage>. <pub-id pub-id-type="doi">10.3390/molecules23010103</pub-id>
</citation>
</ref>
<ref id="B109">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sheridan</surname>
<given-names>R. P.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Time-split cross-validation as a method for estimating the goodness of prospective prediction</article-title>. <source>J. Chem. Inf. Model.</source> <volume>53</volume> (<issue>4</issue>), <fpage>783</fpage>&#x2013;<lpage>790</lpage>. <pub-id pub-id-type="doi">10.1021/ci400084k</pub-id>
</citation>
</ref>
<ref id="B110">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shipman</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Hua</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Desaire</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>DecoyDeveloper: An on-demand, de novo decoy glycopeptide generator</article-title>. <source>J. proteome Res.</source> <volume>18</volume> (<issue>7</issue>), <fpage>2896</fpage>&#x2013;<lpage>2902</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jproteome.9b00203</pub-id>
</citation>
</ref>
<ref id="B111">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Simonovsky</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Komodakis</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>GraphVAE: Towards generation of small graphs using variational autoencoders</article-title>,&#x201d; in <source>Artificial neural networks and machine learning &#x2013; icann 2018</source> (<publisher-name>Springer International Publishing</publisher-name>), <volume>2018</volume>, <fpage>412</fpage>&#x2013;<lpage>422</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-01418-6_41</pub-id>
</citation>
</ref>
<ref id="B112">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Skalic</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jim&#xe9;nez</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sabbadin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>De Fabritiis</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019b</year>). <article-title>Shape-based generative modeling for de novo drug design</article-title>. <source>J. Chem. Inf. Model.</source> <volume>59</volume> (<issue>3</issue>), <fpage>1205</fpage>&#x2013;<lpage>1214</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.8b00706</pub-id>
</citation>
</ref>
<ref id="B113">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Skalic</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sabbadin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Sattarov</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sciabola</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>De Fabritiis</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019a</year>). <article-title>From target to drug: Generative modeling for the multimodal structure-based ligand design</article-title>. <source>Mol. Pharm.</source> <volume>16</volume> (<issue>10</issue>), <fpage>4282</fpage>&#x2013;<lpage>4291</lpage>. <pub-id pub-id-type="doi">10.1021/acs.molpharmaceut.9b00634</pub-id>
</citation>
</ref>
<ref id="B114">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Soares</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Nunes-Alves</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mazzolari</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ruggiu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>G. W.</given-names>
</name>
<name>
<surname>Merz</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>The (Re)-evolution of quantitative structure-activity relationship (qsar) studies propelled by the surge of machine learning methods</article-title>. <source>J. Chem. Inf. Model.</source> <volume>62</volume> (<issue>22</issue>), <fpage>5317</fpage>&#x2013;<lpage>5320</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.2c01422</pub-id>
</citation>
</ref>
<ref id="B115">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sorokina</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Merseburger</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Rajan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Yirik</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Steinbeck</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>COCONUT online: Collection of open natural products database</article-title>. <source>J. cheminformatics</source> <volume>13</volume> (<issue>1</issue>), <fpage>2</fpage>. <pub-id pub-id-type="doi">10.1186/s13321-020-00478-9</pub-id>
</citation>
</ref>
<ref id="B116">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tingle</surname>
<given-names>B. I.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>K. G.</given-names>
</name>
<name>
<surname>Castanon</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gutierrez</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Khurelbaatar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dandarchuluun</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>ZINC-22&#x2500;A free multi-billion-scale database of tangible compounds for ligand discovery</article-title>. <source>J. Chem. Inf. Model.</source> <volume>63</volume> (<issue>4</issue>), <fpage>1166</fpage>&#x2013;<lpage>1176</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.2c01253</pub-id>
</citation>
</ref>
<ref id="B117">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Generative models for de novo drug design</article-title>. <source>J. Med. Chem.</source> <volume>64</volume> (<issue>19</issue>), <fpage>14011</fpage>&#x2013;<lpage>14027</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jmedchem.1c00927</pub-id>
</citation>
</ref>
<ref id="B118">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ullanat</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Variational autoencoder as a generative tool to produce de-novo lead compounds for biological targets</article-title>,&#x201d; in <source>2020 14th international conference on innovations in information Technology (IIT)</source>, <fpage>102</fpage>&#x2013;<lpage>107</lpage>. <pub-id pub-id-type="doi">10.1109/IIT50501.2020.9299078</pub-id>
</citation>
</ref>
<ref id="B119">
<citation citation-type="web">
<collab>UNIIQUIM</collab> (<year>2015</year>). <article-title>Uniiquim</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://uniiquim.iquimica.unam.mx/">https://uniiquim.iquimica.unam.mx/</ext-link> (accessed May 13, 2023)</comment>.</citation>
</ref>
<ref id="B120">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Valli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>dos Santos</surname>
<given-names>R. N.</given-names>
</name>
<name>
<surname>Figueira</surname>
<given-names>L. D.</given-names>
</name>
<name>
<surname>Nakajima</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Castro-Gamboa</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Andricopulo</surname>
<given-names>A. D.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Development of a natural products database from the biodiversity of Brazil</article-title>. <source>J. Nat. Prod.</source> <volume>76</volume> (<issue>3</issue>), <fpage>439</fpage>&#x2013;<lpage>444</lpage>. <pub-id pub-id-type="doi">10.1021/np3006875</pub-id>
</citation>
</ref>
<ref id="B121">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vamathevan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Clark</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Czodrowski</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Dunham</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Ferran</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Applications of machine learning in drug discovery and development</article-title>. <source>Nat. Rev. Drug Discov.</source> <volume>18</volume> (<issue>6</issue>), <fpage>463</fpage>&#x2013;<lpage>477</lpage>. <pub-id pub-id-type="doi">10.1038/s41573-019-0024-5</pub-id>
</citation>
</ref>
<ref id="B122">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Veber</surname>
<given-names>D. F.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>H-Y.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>B. R.</given-names>
</name>
<name>
<surname>Ward</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Kopple</surname>
<given-names>K. D.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Molecular properties that influence the oral bioavailability of drug candidates</article-title>. <source>J. Med. Chem.</source> <volume>45</volume> (<issue>12</issue>), <fpage>2615</fpage>&#x2013;<lpage>2623</lpage>. <pub-id pub-id-type="doi">10.1021/jm020017n</pub-id>
</citation>
</ref>
<ref id="B123">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Rader: A RApid DEcoy retriever to facilitate decoy based assessment of virtual screening</article-title>. <source>Bioinformatics</source> <volume>33</volume> (<issue>8</issue>), <fpage>1235</fpage>&#x2013;<lpage>1237</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btw783</pub-id>
</citation>
</ref>
<ref id="B124">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hsieh</surname>
<given-names>C-Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Weng</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Relation: A deep generative model for structure-based de novo drug design</article-title>. <source>J. Med. Chem.</source> <volume>65</volume> (<issue>13</issue>), <fpage>9478</fpage>&#x2013;<lpage>9492</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jmedchem.2c00732</pub-id>
</citation>
</ref>
<ref id="B125">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Warr</surname>
<given-names>W. A.</given-names>
</name>
<name>
<surname>Nicklaus</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Nicolaou</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Rarey</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Exploration of ultralarge compound collections for drug discovery</article-title>. <source>J. Chem. Inf. Model.</source> <volume>62</volume> (<issue>9</issue>), <fpage>2021</fpage>&#x2013;<lpage>2034</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.2c00224</pub-id>
</citation>
</ref>
<ref id="B126">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Warr</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Report on an NIH workshop on ultralarge chemistry databases</article-title>. <comment>Chemrxiv: 43. Available at: <ext-link ext-link-type="uri" xlink:href="https://chemrxiv.org/engage/chemrxiv/article-details/60c75883bdbb89984ea3ada5">https://chemrxiv.org/engage/chemrxiv/article-details/60c75883bdbb89984ea3ada5</ext-link>
</comment>.</citation>
</ref>
<ref id="B127">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weininger</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>1988</year>). <article-title>SMILES, a chemical language and information system. 1. Introduction to methodology and encoding rules</article-title>. <source>J. Chem. Inf. Comput. Sci.</source> <volume>28</volume> (<issue>1</issue>), <fpage>31</fpage>&#x2013;<lpage>36</lpage>. <pub-id pub-id-type="doi">10.1021/ci00057a005</pub-id>
</citation>
</ref>
<ref id="B128">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wishart</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Feunang</surname>
<given-names>Y. D.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Lo</surname>
<given-names>E. J.</given-names>
</name>
<name>
<surname>Marcu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Grant</surname>
<given-names>J. R.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>DrugBank 5.0: A major update to the DrugBank database for 2018</article-title>. <source>Nucleic acids Res.</source> <volume>46</volume> (<issue>D1</issue>), <fpage>D1074</fpage>&#x2013;<lpage>D1082</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkx1037</pub-id>
</citation>
</ref>
<ref id="B129">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wishart</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Knox</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Shrivastava</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tzur</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>DrugBank: A knowledgebase for drugs, drug actions and drug targets</article-title>. <source>Nucleic acids Res.</source> <volume>36</volume>, <fpage>D901</fpage>&#x2013;<lpage>D906</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkm958</pub-id>
</citation>
</ref>
<ref id="B130">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wishart</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Knox</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Shrivastava</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hassanali</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Stothard</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2006</year>). <article-title>DrugBank: A comprehensive resource for <italic>in silico</italic> drug discovery and exploration</article-title>. <source>Nucleic acids Res.</source> <volume>34</volume>, <fpage>D668</fpage>&#x2013;<lpage>D672</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkj067</pub-id>
</citation>
</ref>
<ref id="B131">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhuang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2023a</year>). <article-title>Elucidating structures of complex organic compounds using a machine learning model based on the 13C NMR chemical shifts</article-title>. <source>Precis. Chem.</source> <volume>1</volume> (<issue>1</issue>), <fpage>57</fpage>&#x2013;<lpage>68</lpage>. <pub-id pub-id-type="doi">10.1021/prechem.3c00005</pub-id>
</citation>
</ref>
<ref id="B132">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2023b</year>). <article-title>DeepCancerMap: A versatile deep learning platform for target- and cell-based anticancer drug discovery</article-title>. <source>Eur. J. Med. Chem.</source> <volume>255</volume>, <fpage>115401</fpage>. <pub-id pub-id-type="doi">10.1016/j.ejmech.2023.115401</pub-id>
</citation>
</ref>
<ref id="B133">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ramsundar</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Feinberg</surname>
<given-names>E. N.</given-names>
</name>
<name>
<surname>Gomes</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Geniesse</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Pappu</surname>
<given-names>A. S.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>MoleculeNet: A benchmark for molecular machine learning</article-title>. <source>Chem. Sci.</source> <volume>9</volume> (<issue>2</issue>), <fpage>513</fpage>&#x2013;<lpage>530</lpage>. <pub-id pub-id-type="doi">10.1039/c7sc02664a</pub-id>
</citation>
</ref>
<ref id="B134">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lai</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Pei</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Advances and challenges in de novo drug design using three-dimensional deep generative models</article-title>. <source>J. Chem. Inf. Model.</source> <volume>62</volume> (<issue>10</issue>), <fpage>2269</fpage>&#x2013;<lpage>2279</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.2c00042</pub-id>
</citation>
</ref>
<ref id="B135">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Freely accessible chemical database resources of compounds for <italic>in silico</italic> drug discovery</article-title>. <source>Curr. Med. Chem.</source> <volume>26</volume> (<issue>42</issue>), <fpage>7581</fpage>&#x2013;<lpage>7597</lpage>. <pub-id pub-id-type="doi">10.2174/0929867325666180508100436</pub-id>
</citation>
</ref>
<ref id="B136">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Chu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <source>The balanced matrix factorization for computational drug repositioning</source>. <comment>arXiv [cs.CE]. Available at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/2301.06448">http://arxiv.org/abs/2301.06448</ext-link>
</comment>.</citation>
</ref>
<ref id="B137">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Responsible use of negative research outcomes-accelerating the discovery and development of new antibiotics</article-title>. <source>J. antibiotics</source> <volume>74</volume> (<issue>9</issue>), <fpage>543</fpage>&#x2013;<lpage>546</lpage>. <pub-id pub-id-type="doi">10.1038/s41429-021-00439-w</pub-id>
</citation>
</ref>
<ref id="B138">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>T. Y.</given-names>
</name>
<name>
<surname>Bai</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Application of computational biology and artificial intelligence in drug design</article-title>. <source>Int. J. Mol. Sci.</source> <volume>23</volume> (<issue>21</issue>), <fpage>13568</fpage>. <pub-id pub-id-type="doi">10.3390/ijms232113568</pub-id>
</citation>
</ref>
<ref id="B139">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhavoronkov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ivanenkov</surname>
<given-names>Y. A.</given-names>
</name>
<name>
<surname>Aliper</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Veselov</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Aladinskiy</surname>
<given-names>V. A.</given-names>
</name>
<name>
<surname>Aladinskaya</surname>
<given-names>A. V.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Deep learning enables rapid identification of potent DDR1 kinase inhibitors</article-title>. <source>Nat. Biotechnol.</source> <volume>37</volume> (<issue>9</issue>), <fpage>1038</fpage>&#x2013;<lpage>1040</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-019-0224-x</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>