<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2024.1374148</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>PED: a novel predictor-encoder-decoder model for Alzheimer drug molecular generation</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Liu</surname> <given-names>Dayan</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2635889/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Song</surname> <given-names>Tao</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1316895/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Na</surname> <given-names>Kang</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Wang</surname> <given-names>Shudong</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1277774/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>College of Computer Science and Technology, China University of Petroleum (East China)</institution>, <addr-line>Qingdao</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>The Ninth Department of Health Care Administration, The Second Medical Center, Chinese PLA General Hospital</institution>, <addr-line>Beijing</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Pan Zheng, University of Canterbury, New Zealand</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Tongmao Ma, Polytechnic University of Madrid, Spain</p>
<p>Shailesh Tripathi, University of Applied Sciences Upper Austria, Austria</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Shudong Wang <email>wangsd&#x00040;upc.edu.cn</email></corresp>
<fn fn-type="equal" id="fn001"><p>&#x02020;These authors share first authorship</p></fn></author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>04</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>7</volume>
<elocation-id>1374148</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>01</day>
<month>04</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Liu, Song, Na and Wang.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Liu, Song, Na and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Alzheimer&#x00027;s disease (AD) is a gradually advancing neurodegenerative disorder characterized by a concealed onset. Acetylcholinesterase (AChE) is an efficient hydrolase that catalyzes the hydrolysis of acetylcholine (ACh), which regulates the concentration of ACh at synapses and then terminates ACh-mediated neurotransmission. There are inhibitors to inhibit the activity of AChE currently, but its side effects are inevitable. In various application fields where Al have gained prominence, neural network-based models for molecular design have recently emerged and demonstrate encouraging outcomes. However, in the conditional molecular generation task, most of the current generation models need additional optimization algorithms to generate molecules with intended properties which make molecular generation inefficient. Consequently, we introduce a cognitive-conditional molecular design model, termed PED, which leverages the variational auto-encoder. Its primary function is to adeptly produce a molecular library tailored for specific properties. From this library, we can then identify molecules that inhibit AChE activity without adverse effects. These molecules serve as lead compounds, hastening AD treatment and concurrently enhancing the AI&#x00027;s cognitive abilities. In this study, we aim to fine-tune a VAE model pre-trained on the ZINC database using active compounds of AChE collected from Binding DB. Different from other molecular generation models, the PED can simultaneously perform both property prediction and molecule generation, consequently, it can generate molecules with intended properties without additional optimization process. Experiments of evaluation show that proposed model performs better than other methods benchmarked on the same data sets. The results indicated that the model learns a good representation of potential chemical space, it can well generate molecules with intended properties. Extensive experiments on benchmark datasets confirmed PED&#x00027;s efficiency and efficacy. Furthermore, we also verified the binding ability of molecules to AChE through molecular docking. The results showed that our molecular generation system for AD shows excellent cognitive capacities, the molecules within the molecular library could bind well to AChE and inhibit its activity, thus preventing the hydrolysis of ACh.</p></abstract>
<kwd-group>
<kwd>molecular generation</kwd>
<kwd>Alzheimer</kwd>
<kwd>deep learning</kwd>
<kwd>neural networks</kwd>
<kwd>drug design</kwd>
</kwd-group>
<contract-num rid="cn001">2021YFA1000102</contract-num>
<contract-num rid="cn001">2021YFA1000103</contract-num>
<contract-num rid="cn002">ZR2021QF023</contract-num>
<contract-num rid="cn003">24CX04029A</contract-num>
<contract-sponsor id="cn001">National Key Research and Development Program of China<named-content content-type="fundref-id">10.13039/501100012166</named-content></contract-sponsor>
<contract-sponsor id="cn002">Natural Science Foundation of Shandong Province<named-content content-type="fundref-id">10.13039/501100007129</named-content></contract-sponsor>
<contract-sponsor id="cn003">Fundamental Research Funds for the Central Universities<named-content content-type="fundref-id">10.13039/501100012226</named-content></contract-sponsor>
<counts>
<fig-count count="12"/>
<table-count count="4"/>
<equation-count count="11"/>
<ref-count count="45"/>
<page-count count="14"/>
<word-count count="7057"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Machine Learning and Artificial Intelligence</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Alzheimer&#x00027;s disease (AD) is a neurodegenerative condition that progresses subtly from its onset (Cummings and Cole, <xref ref-type="bibr" rid="B9">2002</xref>). Symptoms in the clinical setting encompass memory deterioration, speech difficulties, apraxia, agnosia, deficits in visual-spatial abilities, executive function disturbances, and shifts in personality and behavior, etc. The etiology is unknown so far. AD has the characteristics of long course of disease, many causes and complicated pathology. There are other irregularities of neurotransmitters in the center in addition to the drop-in acetylcholine levels in the brain. Additionally, the aggregation of A, the disturbance of metal ion metabolism, the imbalance of calcium balance, the rise in free radicals, and the onset of inflammation are the primary causes of AD. In view of the above causes, the therapeutic targets of AD mainly include acetylcholinesterase (AChE), metal ions, Beta Amyloid Peptide (&#x003B2;-AP), monoamine oxidase (MAO), free radicals, tau protein, N-methyl-D-aspartate (NMDA) receptor and other related targets (Casal et al., <xref ref-type="bibr" rid="B5">2002</xref>; Sambamurti et al., <xref ref-type="bibr" rid="B37">2011</xref>). Acetylcholine (ACh) is the first neurotransmitter discovered by human beings, and its mediated neurotransmission is the basis of nervous system function. Sudden interruption of ACH-mediated neurotransmission is fatal, and its gradual loss is associated with progressive deterioration of cognitive, autonomic and neuromuscular functions (Klinkenberg et al., <xref ref-type="bibr" rid="B24">2011</xref>). However, AChE is an efficient hydrolase that catalyzes the hydrolysis of ACh, which regulates the concentration of ACh at synapses and then terminates ACh-mediated neurotransmission. There are inhibitors to inhibit the activity of AChE, but its side effects are inevitable (Alonso et al., <xref ref-type="bibr" rid="B2">2005</xref>) Therefore, there is an increasing demand for developing active compounds with stronger inhibitory function and minimal side effects. Artificial intelligence (AI) leverages acquired knowledge and insights to formulate decisions and strategize subsequent actions. Modern methods incorporate a variety of strategies relevant to areas such as decision-making or cognitive-enhanced network security. Given that modern machines often lack intuition, emotional intelligence, common sense, and other human-centric attributes essential for effective planning and decision-making, there&#x00027;s potential to enhance planning-focused cognitive technology through broader artificial intelligence research (Fintz et al., <xref ref-type="bibr" rid="B16">2022</xref>; Liu et al., <xref ref-type="bibr" rid="B28">2022</xref>; Qiu et al., <xref ref-type="bibr" rid="B35">2022</xref>).</p>
<p>To efficiently generate molecule library with intended properties, we propose a cognitive conditional molecular design model based on VAE which can predict properties and generate molecules concurrently, named PED, to screen molecules that can inhibit AChE activity from the generated molecular libraries as lead compounds and accelerate the treatment of AD. In this study, we aim to finetune a VAE pre-trained on the ZINC database using active compounds of AChE collected from BindingDB. On the same data sets, PED performs better than other methods. Meanwhile, we show that the model can well generate specified molecular properties. Furthermore, we also verified the binding ability of molecules to AChE through molecular docking. The results showed that the molecules in the molecular library could bind to AChE well.</p>
<p>The main contributions of this manuscript are summarized as below.</p>
<p>1. We put forth a novel deep learning model based on variational auto-encoder, namely PED, to efficiently generate molecular library with desired properties for AChE, simultaneously, show the cognitive capacities of AI. PED is engineered to manage both property forecasting and molecular generation in tandem, striving for superior outcomes relative to advanced methods.</p>
<p>2. In PED, given a specific set of properties, it samples new molecules directly from the conditionally generated distribution without adding additional optimization processes like other models.</p>
<p>3. Extensive testing was carried out on the ZINC database to assess PED&#x00027;s efficacy. The outcomes from these tests highlighted PED&#x00027;s predominant performance over other deep generative frameworks.</p>
<p>4. Using active AChE compounds sourced from Binding DB, we refine the PED initially pre-trained on the ZINC database to produce a molecular library. Furthermore, we also verified the binding ability of molecules to AChE through molecular docking. The results showed that the molecules in the molecular library could bind to AChE well.</p>
<p>The subsequent sections of this research are structured in the following manner. Some previous studies in de novo molecular design are reviewed in Section 2. Our model is introduced in Section 3. Experimental results and conclusions are presented in Section 4. Performance Analysis and Section 5. Conclusion and the future work respectively.</p>
</sec>
<sec id="s2">
<title>2 Related works</title>
<p>In this section, we first review the development of deep learning in molecular generation in recent years, and then introduce several commonly used molecular generation strategies and models.</p>
<sec>
<title>2.1 Deep learning in molecular generation</title>
<p>Within the realm of molecular design, virtual screening (VS) has conventionally been employed to pinpoint molecules potentially yielding optimal experimental outcomes (Shoichet, <xref ref-type="bibr" rid="B39">2004</xref>). Contrasting de novo molecular design, the source of molecules is distinctive: in virtual screening, the structure is known in advance, while in molecular de novo design, it is an attempt to generate the structure to be evaluated. Although virtual screening libraries have become very large according to the standards of drug discovery, the chemical space corresponding to these libraries only occupies a small part. When considering such a compound library, the evaluation method may inevitably sacrifice the accuracy of prediction. By using de novo molecular design to generate molecules in a directional way, computational workers hope to cross the chemical space more effectively and obtain the best chemical solution while analyzing fewer molecules than large chemical libraries. In addition, for a given target, there may be many acceptable regions in chemical space. Hence, the objective of the molecular design approach is to strike a balance between exploring global solutions and harnessing local minima (Schneider, <xref ref-type="bibr" rid="B38">2010</xref>; M&#x000FC;ller et al., <xref ref-type="bibr" rid="B31">2022</xref>).</p>
<p>Recently, with the advancement of artificial intelligence (AI), new practical experience has been gained in the field of drug discovery (Ding et al., <xref ref-type="bibr" rid="B14">2017</xref>, <xref ref-type="bibr" rid="B13">2022</xref>; Chu et al., <xref ref-type="bibr" rid="B7">2022</xref>). In typical data domains like computer vision (Voulodimos et al., <xref ref-type="bibr" rid="B42">2018</xref>; Borhani et al., <xref ref-type="bibr" rid="B4">2022</xref>) and natural language processing (NLP) (Chowdhary, <xref ref-type="bibr" rid="B6">2020</xref>; Ferruz et al., <xref ref-type="bibr" rid="B15">2022</xref>), deep generative models have significantly advanced in representing data distributions (Meyers et al., <xref ref-type="bibr" rid="B30">2021</xref>). Such techniques are also employed to mimic molecular distributions, understand the probabilistic distributions of vast molecule sets, and produce novel molecules by drawing samples from these distributions (Dauparas et al., <xref ref-type="bibr" rid="B10">2022</xref>). In the realm of molecular structure generation, a variety of deep learning models have been suggested by scholars. These encompass techniques such as generative adversarial networks (GANs), variational autoencoders (VAEs), and recurrent neural systems (RNNs) (Creswell et al., <xref ref-type="bibr" rid="B8">2018</xref>; Korshunova et al., <xref ref-type="bibr" rid="B25">2022</xref>). In these methods, a molecule is represented as a simplified molecular-input line-entry system (SMILES) (Weininger, <xref ref-type="bibr" rid="B44">1988</xref>). Most of the current molecular generation models are based on conditional molecular design and finally generate new molecules with properties close to the predetermined target conditions (Xu et al., <xref ref-type="bibr" rid="B45">2019</xref>; Walters and Barzilay, <xref ref-type="bibr" rid="B43">2020</xref>).</p>
</sec>
<sec>
<title>2.2 Molecular generation model</title>
<p>The descriptors of SMILES are generally implemented by using long-term and short-term memory networks (LSTM). Serving as a unique temporal cycle neural network, LSTM was crafted explicitly to tackle the pervasive issue of long-term dependencies inherent in traditional RNNs. Due to the characteristics of this cyclic algorithm, the cyclic structure and chiral center of molecules expressed by SMILES are presented more perfectly. LSTM can be used to generate molecular sets with or without filters. Grisoni et al. suggested the use of bidirectional generative RNNs for designing molecules based on SMILES. In pursuit of this, they employed two proven bidirectional approaches and pioneered a novel technique for augmenting data and generating SMILES strings, termed as bidirectional molecule design by alternate learning (BIMODAL) (Grisoni et al., <xref ref-type="bibr" rid="B19">2020</xref>). In addition, Li et al. studied the ability of RNN-based de novo molecular design method to produce new molecular inhibitors in the research field of chemical space (Li et al., <xref ref-type="bibr" rid="B27">2020</xref>). In their quest to formulate novel inhibitors for proto-oncogene serine/threonine protein kinase 1 (PIM1) and CDK4 kinase, they evaluated four compounds. Their efforts culminated in the identification of a potent PIM1 inhibitor and two primary compounds that hinder CDK4 activity.</p>
<p>There is also a class of deep learning algorithms for automatic encoders, such as VAE and adversarial auto-encoder (AAE), which use the description method of molecules in latent space to generate molecules. On the one hand, the molecular features of the training set are stored in latent space by encoder, and on the other hand, these molecular features are reconstituted into new molecules by decoder. Owing to this approach&#x00027;s utilization of continuous latent space accumulation, the newly generated molecular set retains the physicochemical property distribution inherent in the training set. Many models have been proposed that employ reasonable substructures as building blocks for generating high-quality molecules. Previous studies introduced a model termed chemical-vae (G&#x000F3;mez-Bombarelli et al., <xref ref-type="bibr" rid="B18">2018</xref>), designed to produce novel molecules, enabling effective exploration and refinement within expansive chemical compound spaces. In order to generate effective molecular graphs, MHG-VAE proposed molecular hypergraph grammar (MHG) to encode chemical constraints (Kajino, <xref ref-type="bibr" rid="B22">2019</xref>). The authors have proposed a reaction model to forecast the interaction among reactants, resulting in the creation of novel molecules. In lieu of VAE, the objective function incorporates minimization to acquire model parameters.</p>
<p>Another popular deep learning algorithm is GAN. The algorithm uses two functions, the generator and the discriminator, against each other to generate the desired molecules. Because of the discontinuity of the atoms that make up the molecule, the discriminator can&#x00027;t directly feedback the information to the generator. Referring to the method adopted in NLP, the information feedback is realized by a reward function or policy gradient. The reward equation serves as a filtering criterion. It not only preserves the property distribution of the generated set akin to the training set but also nudges the property distribution of the created set to shift toward a different direction. This algorithm can use various molecular description methods, such as SMILES, Latent space, or graph, and can meet various requirements by combining various screening conditions, so it is a potential algorithm. Prykhodko et al. introduced LatentGAN, a novel deep learning framework that integrates an autoencoder with a generative adversarial neural network, tailored for de novo molecular design (Prykhodko et al., <xref ref-type="bibr" rid="B34">2019</xref>).</p>
<p>To refine a sequence-based generative model specifically for molecular de novo design, Marcus and team formulated a technique capable of learning to construct structures with predetermined desirable characteristics, employing enhanced episodic likelihood (Olivecrona et al., <xref ref-type="bibr" rid="B32">2017</xref>). M Popova et al. proposed a unique computational approach for the de novo design of molecules with targeted characteristics, named ReLeaSE, which utilizes deep learning and reinforcement learning methodologies (Popova et al., <xref ref-type="bibr" rid="B33">2018</xref>). These methods used additional optimization processes instead of directly generating molecules of intended properties, which becomes inefficient.</p>
<sec>
<title>2.2.1 Atom-based molecular generation</title>
<p>Numerous atom-centric generative models employ SMILES for depicting molecules. Given that SMILES serves as a text-centric representation, chemistry generation methodologies can leverage sequence-appropriate deep learning structures like RNNs. By extensively pre-training on vast molecular structure datasets, the emergent model gains inherent knowledge, encapsulating the effective nuances of SMILES grammar and syntax. Initial endeavors leveraged transfer learning to skew generation toward desired chemical spaces. The prevalent approach now integrates generative tasks with RL algorithms, striving to attain higher rewards by discovering optimal molecules within the search landscape. Beyond the realm of SMILES-centric models, there&#x00027;s a growing fascination with models directly interpreting the topographical configurations of molecular graphs, where atoms and connections represent nodes and edges respectively. These graph-informed models aim to sidestep the synthetic facets of SMILES notation, offering a more innate depiction of molecular frameworks.GraphVAE and MolGAN are based on the method of generating graphs (De Cao and Kipf, <xref ref-type="bibr" rid="B11">2018</xref>; Simonovsky and Komodakis, <xref ref-type="bibr" rid="B40">2018</xref>), which can learn to generate the adjacency matrix of the whole graph at one time. Others describe the method of learning to generate molecules step by step by iteratively modifying molecular graphs. Recently, the RL method has shown promising results in the settings of the diagram.</p>
</sec>
<sec>
<title>2.2.2 Fragment-based molecular generation</title>
<p>While atom-based generative models with prior training exhibit a strong inherent capability toward substructures present in their training sets, they retain the ability to adjust each molecular atom individually. Such adaptability enhances the model&#x00027;s expressiveness, thereby broadening its reach across the chemical space. Conversely, the fragment-based methodology employs a more generalized molecular depiction to constrain the exploration domain. Jin and his colleagues elucidated the workings of JTVAE, a dual-phase generation procedure (Jin et al., <xref ref-type="bibr" rid="B21">2018</xref>). Initially, a nodal tree is developed to mirror the assembly of molecular subcomponents (resembling a simplification graph). Subsequently, a network transmitting graph information deciphers the ultimate molecular form. DeepFMPO, by weighing fragment resemblances in its optimization, attains superior efficacy (Al Jumaily et al., <xref ref-type="bibr" rid="B1">2022</xref>).</p>
<p>Based on the above research, we propose a conditional molecular design model based on VAE, named PED, to efficiently generate a molecule library with intended properties and screen molecules that can inhibit AChE activity without negative consequences from the library as lead compounds, aiming to accelerate the treatment of AD. Unlike previous studies, the PED can perform property prediction and molecule generation simultaneously, which means that it can generate molecules with intended properties without additional optimization processes. This model can improve the efficiency of molecules generation while ensuring the quality of generated molecules.</p>
</sec>
</sec>
</sec>
<sec sec-type="materials and methods" id="s3">
<title>3 Materials and methods</title>
<p>In this section, we first introduce the proposed variational auto-Encoder model for de novo molecular design, named PED. PED consists of three modules, namely predictor, encoder and decoder, by doing so, the model can predict the properties while generating molecules without additional optimization, aiming to ensure the efficiency of molecular generation. Finally, we introduce training procedure and evaluation metrics.</p>
<sec>
<title>3.1 Model overview</title>
<p>The model consists of three 250-dimensional gated recurrent unit (GRU) networks: the predictor network, the encoder network and the decoder network. The predictor and encoder are made up of bidirectional GRUs network, while the decoder is a unidirectional GRU.</p>
<p>In order to forecast the subsequent character within the SMILES strings that depict molecular structures, the final layer incorporates a dense output layer coupled with a neuron unit that utilizes a softmax activation function. In this context, the synthesis of donepezil is showcased as a case study. Principally prescribed for Alzheimer&#x00027;s management, donepezil can be represented by the SMILES notation: &#x0201C;O=C(C(C=C(OC)C(OC)=C1)=C1C2)C2CC(CC3)CCN3CC4=CC=CC=C4&#x0201D;. The initial data for the system comprises a &#x0201C;one-hot&#x0201D; delineation of a SMILES string, whereby each string undergoes segmentation into various tokens. Here, the inaugural token is &#x0201C;O&#x0201D;, transformed into a &#x0201C;one-hot&#x0201D; vector and fed into the linguistic model. Subsequently, the model revises its concealed state and forecasts the probability spread over forthcoming viable tokens, decoded as &#x0201C;=&#x0201D; in this instance. Supplying the one-hot representation of &#x0201C;=&#x0201D; prompts the model to modify its concealed state during the forthcoming cycle, leading to the revelation of the succeeding token. This recurrent process, tackling one token at a time, persists until the &#x0201C;\n&#x0201D; character surfaces, signifying the culmination of the SMILES sequence, thus generating the final SMILES notation for donepezil (<xref ref-type="fig" rid="F1">Figure 1</xref>).</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>The workflow of PED model. Gray and blue areas separate PED into two components: (1) Property prediction, the labeled data are used for building a property predictor which introduces the gaussian distribution in order to address the intractability of y and (2) molecular generation, the autoencoder are trained by the labeled data. The RNN encoder maps x and y to the latent space z, similarly introducing the gaussian distribution, and then RNN decoder maps y and z to original x.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-g0001.tif"/>
</fig>
<p>In the case of the previously delineated predictor and encoder networks, we introduce distinct fixed-form distributions. Specifically, we define q<sub>&#x003D5;</sub>(y &#x02223; x) and q<sub>&#x003D5;</sub>(<italic>z</italic> &#x02223; y, x), each parametrized by &#x003D5;. These distributions aim to approximate the true posterior distribution, employing a widely employed method in efficient variational inference, as described in <xref ref-type="disp-formula" rid="E1">Equations (1</xref>, <xref ref-type="disp-formula" rid="E2">2)</xref>:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mtext>q</mml:mtext></mml:mrow><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>y</mml:mtext><mml:mo>&#x02223;</mml:mo><mml:mtext>x</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo>&#x02223;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mo class="qopname">diag</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mtext>q</mml:mtext></mml:mrow><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>z</mml:mi><mml:mo>&#x02223;</mml:mo><mml:mtext>y</mml:mtext><mml:mo>,</mml:mo><mml:mtext>x</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>z</mml:mi><mml:mo>&#x02223;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mo class="qopname">diag</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where <italic>x</italic> represents a molecule and y represents its continuous valued properties. Given a variable <italic>x</italic>, the properties <italic>y</italic> are predicted as <xref ref-type="disp-formula" rid="E3">Equation (3)</xref>:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext>y</mml:mtext><mml:mo>&#x0007E;</mml:mo><mml:mi>N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mo class="qopname">diag</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In the molecule generation process, we use the decoder network p<sub>&#x003B8;</sub>(x &#x02223; y, z) to generate molecules by the following equation:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>g</mml:mi><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo class="qopname">max</mml:mo></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec>
<title>3.2 Generative model objective</title>
<p>The definition of loss function refers to previous research (Kingma et al., <xref ref-type="bibr" rid="B23">2014</xref>). In this study, the variational lower bound &#x02212;<italic>L</italic>(<italic>x, y</italic>) of the log-probability of a labeled instance (<italic>x, y</italic>) is showed in <xref ref-type="disp-formula" rid="E5">Equation (5)</xref>:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M5"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mi>log</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02265;</mml:mo><mml:msub><mml:mi mathvariant='double-struck'>E</mml:mi><mml:mrow><mml:msub><mml:mi>q</mml:mi><mml:mi>&#x003D5;</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>z</mml:mi></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>log</mml:mi><mml:msub><mml:mi>p</mml:mi><mml:mi>&#x003B8;</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>z</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo>+</mml:mo><mml:mi>log</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>+</mml:mo><mml:mi>log</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>z</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>log</mml:mi><mml:msub><mml:mi>q</mml:mi><mml:mi>&#x003D5;</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>z</mml:mi></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant='double-struck'>E</mml:mi><mml:mrow><mml:msub><mml:mi>q</mml:mi><mml:mi>&#x003D5;</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>z</mml:mi></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>log</mml:mi><mml:msub><mml:mi>p</mml:mi><mml:mi>&#x003B8;</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>z</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mi>log</mml:mi><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mi mathvariant='script'>D</mml:mi><mml:mrow><mml:mtext>KL</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>q</mml:mi><mml:mi>&#x003D5;</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>z</mml:mi></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02016;</mml:mo><mml:mi>p</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>z</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>&#x02112;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>x</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>y</mml:mi></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Given the data distributions of labeled <inline-formula><mml:math id="M7"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, the loss function is defined as <xref ref-type="disp-formula" rid="E6">Equation (6)</xref>:</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi mathvariant="script">J</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0007E;</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munder></mml:mstyle><mml:mi mathvariant="bold-script">L</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>&#x003B2;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0007E;</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo>&#x0007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munder></mml:mstyle><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x1D53C;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>q</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>|</mml:mo><mml:msup><mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the last term is mean squared error for generative learning.</p>
<p>We use the decoder network <italic>p</italic><sub>&#x003B8;</sub>(<bold>x</bold> &#x02223; <bold>y</bold>, <bold>z</bold>) to generate a molecule. A molecule representation <inline-formula><mml:math id="M9"><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula> is obtained from <bold>y</bold> and <bold>z</bold> by <xref ref-type="disp-formula" rid="E4">Equation (4)</xref>. At each time step j of the decoder, the output <bold>x</bold><sup>(<italic>j</italic>)</sup> is predicted by conditioning on all the previous outputs (<bold>x</bold><sup>(1)</sup>, ..., <bold>x</bold><sup>(<italic>j</italic>&#x02212;1)</sup>), <bold>y</bold>, and <bold>z</bold>, because we decompose <italic>p</italic><sub>&#x003B8;</sub>(<bold>x</bold> &#x02223; <bold>y</bold>, <bold>z</bold>) as <xref ref-type="disp-formula" rid="E7">Equation (7)</xref></p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x0220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>|</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
<sec>
<title>3.3 Training procedure and evaluation metrics</title>
<p>The model undergoes training for 300 cycles utilizing the Adam optimizer. To mitigate the risk of overfitting, we employ early stopping during the training process. This means that if the model&#x00027;s performance on the validation set deteriorates compared to the previous cycle, we halt the training and adopt the parameters from the prior iteration as the final outcome. The loss stabilizes after the 25th cycle, as depicted in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Loss comparison for the training data set and test data set over 30 epochs, where the blue lines indicate the training loss, the red lines indicate the test loss.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-g0002.tif"/>
</fig>
<p>The model was implemented using TensorFlow (v2.4.0) in Python (v3.8). We trained it on an NVIDIA 3090 GPU with the learning rate set to 0.001. And the metrics we used are as follows:</p>
<p>&#x02022; <bold>Validity</bold>: The model&#x00027;s learning capability is evidenced by the rating of authentic molecules among the synthesized compounds as follows (<xref ref-type="disp-formula" rid="E8">Equation 8</xref>).</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>V</mml:mi><mml:mtext>al</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:mi>V</mml:mi><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>&#x02022; <bold>Uniqueness</bold>: the percentage of molecules that were really unique when they were generated. Low uniqueness points to recurrent molecule production and a model with little distribution learning as follows (<xref ref-type="disp-formula" rid="E9">Equation 9</xref>).</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>U</mml:mi><mml:mi>n</mml:mi><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:mo class="qopname">set</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>V</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>&#x02022; <bold>Novelty</bold>: the percentage of authentically unique molecules that were generated but were not in the training set as follows (<xref ref-type="disp-formula" rid="E10">Equation 10</xref>).</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M13"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>N</mml:mi><mml:mi>o</mml:mi><mml:mi>v</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:mo class="qopname">set</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02229;</mml:mo><mml:mi>X</mml:mi><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mi>V</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where X is the list of molecules from the provided training set, n is the number of generated samples, and V is the list of created chemically valid molecules.</p>
</sec>
</sec>
<sec sec-type="results" id="s4">
<title>4 Results</title>
<p>In this section, we introduce two publicly available compounds datasets, ZINC and BindingDB, describe the parameters and performance metrics of the evaluation experiments, and evaluate the performance of the proposed method.</p>
<sec>
<title>4.1 Datasets</title>
<p>For the pretraining set, we collected 310,000 SMILES strings of drug-like molecules from the ZINC database (Irwin and Shoichet, <xref ref-type="bibr" rid="B20">2005</xref>) with molecular weight (MolWt) ranging from 200 to 500 and logP ranging from 0 to 5. <xref ref-type="fig" rid="F3">Figures 3</xref>, <xref ref-type="fig" rid="F4">4</xref> show the larger property distribution of the ZINC database, the larger the property distribution, the stronger the fitting ability of our model. Furthermore, in order to maintain the standardization and unity of data, we used RDkit toolkit (Landrum et al., <xref ref-type="bibr" rid="B26">2013</xref>) to canonicalize the SMILES strings.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>The property distribution of ZINC dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-g0003.tif"/>
</fig>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>The property distribution of ZINC dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-g0004.tif"/>
</fig>
<p>We collected molecules with pIC50 or pEC50 greater than 6 as the active molecules of AChE, delete duplicate molecules, and got the fine-tuning sets from Binding DB (Liu et al., <xref ref-type="bibr" rid="B29">2007</xref>). Molecules from SciFinder (Gabrielson, <xref ref-type="bibr" rid="B17">2018</xref>) that fit the aforementioned requirements were added to the fine-tuning set to enlarge it. Finally, 4,996 molecules were obtained and canonicalized using the RDKit toolkit.</p>
<p>In order to show the diversity of molecules in the data set, we screened molecules with QED (quantitative estimate of drug-likeness) values greater than 8 and logP values between 0 and 3, then randomly selected 10 molecules to calculate molecular similarity. The results are shown in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>The similarity of randomly selected 10 molecules.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-g0005.tif"/>
</fig>
<p>Some properties of the molecules used in the training process of the model are the following:</p>
<list list-type="bullet">
<list-item><p><bold>logP</bold>: The logarithmic value of a substance&#x00027;s partition coefficient in water and n-octane (gasoline). The chemical is more lipophilic the higher the logP value. On the other hand, the more hydrophilic something is, the better its water solubility, and the smaller the logP value.</p></list-item>
<list-item><p><bold>QED</bold>: QED is not based on the properties of chemical structure, but a combination of several molecular properties, which is used to evaluate the drug similarity of molecules. QED quantifies the drug similarity to [0, 1], and the higher the QED score, the higher the drug similarity of molecules.</p></list-item>
<list-item><p><bold>MolWt</bold>: The relative mass of molecules, which refers to the sum of the relative atomic masses of all atoms constituting a molecule. By observing the MW distribution of two groups of molecules, we can check whether the properties of the molecules generated by the model are unbiased or shift toward a certain distribution.</p></list-item>
<list-item><p><bold>SAScore</bold>: Synthesizability score of drug-like molecules based on fragment contribution and complexity penalty.</p></list-item>
</list>
</sec>
<sec>
<title>4.2 Performance evaluation</title>
<sec>
<title>4.2.1 Nonconditioned molecular generation</title>
<p>In this section, we evaluate the PED&#x00027;s unconditional molecular generation capability and compare it with other molecular generation models. <xref ref-type="fig" rid="F6">Figure 6</xref> shows the property distribution of unconditionally generated molecules and ZINC dataset, demonstrating how well our model has absorbed the properties of the training set. Moreover, we contrast PED&#x00027;s performance with that of RNN, VAE, and GAN on the ZINC data set. All models use SMILES as input. <xref ref-type="table" rid="T1">Table 1</xref> reports the model&#x00027;s performance on the ZINC data set, we show the SMILES and its 2D structure randomly selected from which were generated by the different models simultaneously.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>The property distribution of unconditionally generated molecules and ZINC dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-g0006.tif"/>
</fig>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Comparison of the Different Metrics Corresponding to Nonconditioned Generation of Molecules Using Different Approaches Trained on ZINC Data Set.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Models</bold></th>
<th valign="top" align="center"><bold>Val</bold>.</th>
<th valign="top" align="center"><bold>Uni</bold>.</th>
<th valign="top" align="center"><bold>Nov</bold>.</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">RNN (Grisoni et al., <xref ref-type="bibr" rid="B19">2020</xref>)</td>
<td valign="top" align="center">0.970</td>
<td valign="top" align="center">0.999</td>
<td valign="top" align="center">0.786</td>
</tr> <tr>
<td valign="top" align="left">VAE (G&#x000F3;mez-Bombarelli et al., <xref ref-type="bibr" rid="B18">2018</xref>)</td>
<td valign="top" align="center">0.963</td>
<td valign="top" align="center">0.999</td>
<td valign="top" align="center">0.532</td>
</tr> <tr>
<td valign="top" align="left">GAN (Prykhodko et al., <xref ref-type="bibr" rid="B34">2019</xref>)</td>
<td valign="top" align="center">0.926</td>
<td valign="top" align="center">0.999</td>
<td valign="top" align="center">0.921</td>
</tr> <tr>
<td valign="top" align="left">PED</td>
<td valign="top" align="center"><bold>0.991</bold></td>
<td valign="top" align="center"><bold>1</bold></td>
<td valign="top" align="center"><bold>0.906</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values indicates the metric scores of our proposed model.</p>
</table-wrap-foot>
</table-wrap>
<p>From the <xref ref-type="table" rid="T1">Table 1</xref>, PED generates the most reliable and distinctive compounds. However, in the case of novelty, GAN is more likely to generate new molecules. In a nutshell, our model shows the best results in terms of uniqueness and validity, and its novelty is only 0.015 less than the GAN. Therefore, in comparison to other models, PED is the preferable method. <xref ref-type="table" rid="T2">Table 2</xref> displays the randomly selected SMILES generated by the PED and other models.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Randomly selected SMILES generated by the different models.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Models</bold></th>
<th valign="top" align="center"><bold>Sampled SMILES</bold></th>
<th valign="top" align="center"><bold>Structure</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="2">RNN (Grisoni et al., <xref ref-type="bibr" rid="B19">2020</xref>)</td>
<td valign="top" align="center">CN(Cc1ccccc1)C(=O)C1CCCN(S(=O)(=O)c2ccccc2)C1</td>
<td valign="top" align="center"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-i0001.tif"/></td>
</tr>
 <tr>
<td valign="top" align="center">CCc1ccc(C(=O)NC)nn1</td>
<td valign="top" align="center"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-i0002.tif"/></td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">VAE (G&#x000F3;mez-Bombarelli et al., <xref ref-type="bibr" rid="B18">2018</xref>)</td>
<td valign="top" align="center">O=C(c1ccccc1)N1CCN(c2ccc([N&#x0002B;](=O)[O-])cn2)CC1</td>
<td valign="top" align="center"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-i0003.tif"/></td>
</tr>
 <tr>
<td valign="top" align="center">CCN(CC)S(=O)(=O)c1cccc(C(=O)Nc2ccc(OC(C)C)cc2)c1</td>
<td valign="top" align="center"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-i0004.tif"/></td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">GAN (Prykhodko et al., <xref ref-type="bibr" rid="B34">2019</xref>)</td>
<td valign="top" align="center">CC1CCCCC12NC(=O)N(CC(=O)Nc1ccccc1C(=O)O)C2=O</td>
<td valign="top" align="center"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-i0005.tif"/></td>
</tr>
 <tr>
<td valign="top" align="center">CCCCCCCCCCCCCCCCCCCCCC1CCC(O)C1(CCC)CCCCCCCCCCCCCCC</td>
<td valign="top" align="center"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-i0006.tif"/></td>
</tr> <tr>
<td valign="top" align="left" rowspan="2">PED</td>
<td valign="top" align="center">COc1ccccc1NC(=O)CSc1nc(=O)n2ccccc2n1</td>
<td valign="top" align="center"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-i0007.tif"/></td>
</tr>
 <tr>
<td valign="top" align="center">Cc1ccc(NC(=O)C2CCCN(c3ncnc4onc(C)c34)C2)cc1C</td>
<td valign="top" align="center"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-i0008.tif"/></td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<title>4.2.2 Generation-based on single properties</title>
<p>In this part, we assessed PED&#x00027;s ability to generate molecules with the desired property. We defined the values of LogP, MolWt, and QED accordingly, and created 5,000 molecules under each scenario to evaluate the ability of the model to produce molecules possessing the targeted characteristics. <xref ref-type="table" rid="T3">Table 3</xref> illustrates the validity, distinctiveness, and originality scores for each specific condition. PED can still efficiently generate high-quality molecules during the conditional molecular generation process. The property distribution of the generated molecules is shown in <xref ref-type="fig" rid="F7">Figure 7</xref>.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Comparison of different metrics while generating molecules conditioned on single property based on training on ZINC data set.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Condition</bold></th>
<th valign="top" align="center"><bold>Val</bold>.</th>
<th valign="top" align="center"><bold>Uni</bold>.</th>
<th valign="top" align="center"><bold>Nov</bold>.</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">MolWt</td>
<td valign="top" align="center">0.985</td>
<td valign="top" align="center">0.913</td>
<td valign="top" align="center">1</td>
</tr> <tr>
<td valign="top" align="left">LogP</td>
<td valign="top" align="center">0.966</td>
<td valign="top" align="center">0.951</td>
<td valign="top" align="center">1</td>
</tr> <tr>
<td valign="top" align="left">QAE</td>
<td valign="top" align="center">0.975</td>
<td valign="top" align="center">0.920</td>
<td valign="top" align="center">1</td>
</tr></tbody>
</table>
</table-wrap>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Distributions of properties of generated molecules while controlling a single property.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-g0007.tif"/>
</fig>
<p><xref ref-type="table" rid="T3">Table 3</xref> reveals that when the density of the target value in the training set&#x00027;s distribution was diminished, there was a slight increase in the percentage of inaccurate molecules. Moreover, the model yielded a greater number of replicated molecules when the property forecast for a given condition was notably precise. In particular, the molecules generated by the model did not appear in the training set. In <xref ref-type="fig" rid="F7">Figure 7</xref>, distributions of generated molecules&#x00027; attributes are shown while optimizing single property, and the distribution is centered around the desired value. The results show that the model can still generate molecules with desired properties without additional optimization steps.</p>
</sec>
</sec>
<sec>
<title>4.3 Generating the molecular library for AChE</title>
<p>Acetylcholine (ACh) is the first neurotransmitter discovered by human beings, and its mediated neurotransmission is the basis of nervous system function. Nevertheless, AChE is an efficient hydrolase that catalyzes the hydrolysis of ACh, which regulates the concentration of ACh at synapses and then terminates ACh-mediated neurotransmission. Thus, in this segment, our goal was to create a compound library targeting the AChE receptor. This would aid in the synthesis of newer inhibitors that are not only more potent but also exhibit reduced side effects.</p>
<p>We gathered molecules from Binding DB that displayed activity toward AChE receptors to formulate the training dataset. In the end, 4,996 molecules were utilized to constitute the fine-tuning dataset. As depicted in <xref ref-type="fig" rid="F8">Figure 8</xref>, we chose 500 molecules synthesized by the fine-tuned model, which exhibit physicochemical attributes and occupy the chemical space analogous to the fine-tuning set. Additionally, the distributions of QED in the generated molecules are similar to those in the fine-tuning set of compounds. Furthermore, the distribution of SAScore is concentrated between 1 and 5, as shown in <xref ref-type="fig" rid="F9">Figure 9</xref>, indicating that most of the molecules generated are easy to synthesize.</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Comparison of property distribution of fine-tuning data set and molecules generated by fine-tuning model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-g0008.tif"/>
</fig>
<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p>The distribution of SAScore.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-g0009.tif"/>
</fig>
<p><xref ref-type="fig" rid="F10">Figure 10</xref> shows examples of the generated molecules. Simultaneously, we compared the molecules with donepezil (Birks and Harvey, <xref ref-type="bibr" rid="B3">2018</xref>) (ID: E20) by ECFP4 (Rogers and Hahn, <xref ref-type="bibr" rid="B36">2010</xref>) similarity method (<xref ref-type="fig" rid="F11">Figure 11</xref>). As shown in the <xref ref-type="fig" rid="F11">Figure 11</xref>, the model can effectively generate molecules similar to the training set. This indicates that our model can generate molecules that are effective for ACHE to a great extent. In order to further verify this conclusion, we conducted molecular docking experiments with Autodock Vina (Trott and Olson, <xref ref-type="bibr" rid="B41">2010</xref>). The ligand in the crystal structure of human acetylcholinesterase in complex with donepezil (Dileep et al., <xref ref-type="bibr" rid="B12">2022</xref>) (PDB ID:7E3H) was removed, then recorded the pocket position simultaneously. Molecules in the molecular library are docked with receptors at the pocket position using Autodock Vina. The docking scores are shown in <xref ref-type="table" rid="T4">Table 4</xref>, and the results showed that the molecules in the molecular library had high affinity with the target receptor (<xref ref-type="fig" rid="F12">Figure 12</xref>). Based on the molecular docking results, as depicted in <xref ref-type="fig" rid="F12">Figure 12A</xref>, the interaction between the small molecule and the protein primarily involves hydrogen bonding and hydrophobic interactions. Specifically, the N atom of the small molecule forms hydrogen bonds with the hydroxyl O atom of the Tyr124 amino acid residue (Tyr124=O... H- N, 2.3&#x000C5;), as well as with the hydroxyl O atom of the Tyr337 residue (Tyr337=O... H- N, 2.5&#x000C5;). Additionally, <xref ref-type="fig" rid="F12">Figure 12B</xref> illustrates that the heteroatoms in the small molecule can engage in hydrogen bonding interactions with the active pocket of the protein, with the distribution of hydrogen bond donors and acceptors shown in <xref ref-type="fig" rid="F12">Figure 12B</xref>. Furthermore, the 2D interaction analysis (<xref ref-type="fig" rid="F12">Figure 12C</xref>) revealed that the hydrophobic carbon chain of the small molecule interacts with the hydrophobic amino acids Thr83, sn87, Trp286, Phe338, and Ile451 of the protein. Moreover, the small molecule forms &#x003C0;&#x02212;&#x003C0; stacking interactions with the amino acid residues Trp86, Trp286, and Trp341, enhancing its binding affinity to the protein. Given the close proximity between the small molecule and other amino acid residues, it is hypothesized that van der Waals interactions may occur between them.</p>
<fig id="F10" position="float">
<label>Figure 10</label>
<caption><p>Examples of generated molecules using PED fine-tuning model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-g0010.tif"/>
</fig>
<fig id="F11" position="float">
<label>Figure 11</label>
<caption><p>The most similar molecules to Donepezil among the generated molecules.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-g0011.tif"/>
</fig>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Results of the molecular docking.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Protein</bold></th>
<th valign="top" align="center"><bold>Generated SMILES</bold></th>
<th valign="top" align="center"><bold>Affinity(kcal/mol)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="5">Acetylcholinesterase (Uniprot ID:P22303)</td>
<td valign="top" align="center">O=C(NCCCCCCNc1c2c(nc3ccccc13)CCCC2)c1ccc2nc(-c3ccc(Cl)cc3)c3c(c2c1)NCCC3</td>
<td valign="top" align="center">-12.9</td>
</tr>
 <tr>
<td valign="top" align="center">O=C(Cc1cc(=O)oc2cc(O)ccc12)NCCCNc1c2c(nc3cc(Cl)ccc13)CCCC2</td>
<td valign="top" align="center">-12.7</td>
</tr>
 <tr>
<td valign="top" align="center">COc1cc2c(cc1OC)C(=O)C(=Cc1ccc(N3CC[N&#x0002B;](C)(Cc4ccccc4)CC3)cc1)C2</td>
<td valign="top" align="center">-12.7</td>
</tr>
 <tr>
<td valign="top" align="center">COc1ccc(Cn2cc(C(=O)NCCCNc3c4c(nc5ccccc35)CCCC4)c(=O)c3ccccc32)cc1</td>
<td valign="top" align="center">-12.6</td>
</tr>
 <tr>
<td valign="top" align="center">COc1ccc(Cn2cc(C(=O)NCCCNc3c4c(nc5cc(Cl)ccc35)CCCC4)c(=O)c3ccccc32)cc1</td>
<td valign="top" align="center">-12.5</td>
</tr></tbody>
</table>
</table-wrap>
<fig id="F12" position="float">
<label>Figure 12</label>
<caption><p>The interaction diagram between the generated molecular and the Acetylcholinesterase (Uniprot ID: P22303). <bold>(A)</bold> Molecular docking and 3D display of the interaction diagram between the generated molecular and the Acetylcholinesterase (Uniprot ID: P22303). <bold>(B)</bold> Hydrogen bond coloring display of the interaction diagram between the generated molecular and the Acetylcholinesterase (Uniprot ID: P22303). <bold>(C)</bold> Local 2D display of the interaction diagram between the generated molecular and the Acetylcholinesterase (Uniprot ID: P22303).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-07-1374148-g0012.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="s5">
<title>5 Discussion</title>
<p>In this work, in order to improve the cognitive technology of AI, we propose a cognitive conditional molecular design model based on VAE to efficiently generate a molecule library with intended properties and screen molecules that can inhibit AChE activity from the library as lead compounds to accelerate the treatment of AD. The model can simultaneously perform both property prediction and molecule generation. We see through our benchmarking experiments that our model shows very high validity, novelty and uniqueness scores for the data sets. Furthermore, the statistics indicate our model&#x00027;s strong control over intended properties for molecular generation under conditional molecular generation. In addition, the statistical data show that our model has a strong ability to control the expected properties of molecular generation under conditional generation. We used AChE active molecule data set to fine-tune the model, generated a molecular library for the target receptor, and ultimately verified the binding ability of molecules to AChE through molecular docking. PED has shown to be promising from a practical and theoretical point of view. It can generate new drug-like molecules for AChE and provide guidance for AD to develop new drugs, which means our model has strong cognitive capacities.</p>
<p>However, there are still some limitations of this work. In this study, we used the traditional strategy to verify the molecular activity, And it is aimed at a single target for molecular generation. Because the occurrence and development of AD involves various complex regulatory networks and changes of regulatory factors, multi-target compounds are the trend of AD drug research and development at present. In the future, we will study a new generation model for multi-target molecular generation and introduce a new deep learning model to predict the activity of generated molecular against target, aiming to automatically generating active molecule library.</p>
</sec>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: ZINC database (<ext-link ext-link-type="uri" xlink:href="https://zinc.docking.org/">zinc.docking.org/</ext-link>) and Binding DB (<ext-link ext-link-type="uri" xlink:href="https://www.bindingdb.org/">www.bindingdb.org/</ext-link>).</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>DL: Conceptualization, Data curation, Methodology, Writing&#x02014;original draft, Writing&#x02014;review &#x00026; editing. TS: Data curation, Funding acquisition, Methodology, Project administration, Resources, Supervision, Visualization, Writing&#x02014;review &#x00026; editing. SW: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Supervision, Validation, Visualization, Writing&#x02014;review &#x00026; editing. KN: Formal Analysis, Investigation, Resources, Supervision, Validation, Visualization, Writing&#x02014;review &#x00026; editing.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work was supported by National Key Research and Development Project of China (2021YFA1000102, 2021YFA1000103), Natural Science Foundation of China (Grant Nos. 61873280, 61972416), Taishan Scholarship (tsqn201812029), Foundation of Science and Technology Development of Jinan (201907116), Shandong Provincial Natural Science Foundation (ZR2021QF023), Fundamental Research Funds for the Central Universities (24CX04029A), and Spanish project PID2019-106960GB-I00, Juan de la Cierva IJC2018-038539-I.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Al Jumaily</surname> <given-names>A.</given-names></name> <name><surname>Mukaidaisi</surname> <given-names>M.</given-names></name> <name><surname>Vu</surname> <given-names>A.</given-names></name> <name><surname>Tchagang</surname> <given-names>A.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name></person-group> (<year>2022</year>). <article-title>Exploring multi-objective deep reinforcement learning methods for drug design</article-title>, in <source>2022 IEEE Conference on Computational Intelligence in Bioinformatics and Computational Biology (CIBCB)</source> (<publisher-loc>Ottawa, ON</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>8</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alonso</surname> <given-names>D.</given-names></name> <name><surname>Dorronsoro</surname> <given-names>I.</given-names></name> <name><surname>Rubio</surname> <given-names>L.</given-names></name> <name><surname>Munoz</surname> <given-names>P.</given-names></name> <name><surname>Garc&#x00027;&#x00131;a-Palomero</surname> <given-names>E.</given-names></name> <name><surname>Del Monte</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2005</year>). <article-title>Donepezil-tacrine hybrid related derivatives as new dual binding site inhibitors of ache</article-title>. <source>Bioorganic Med. Chem</source>. <volume>13</volume>, <fpage>6588</fpage>&#x02013;<lpage>6597</lpage>. <pub-id pub-id-type="doi">10.1016/j.bmc.2005.09.029</pub-id></citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Birks</surname> <given-names>J. S.</given-names></name> <name><surname>Harvey</surname> <given-names>R. J.</given-names></name></person-group> (<year>2018</year>). <article-title>Donepezil for dementia due to alzheimer&#x00027;s disease</article-title>. <source>Cochrane Database Syst. Rev</source>. <volume>6</volume>, <fpage>CD001190</fpage>. <pub-id pub-id-type="doi">10.1002/14651858.CD001190.pub3</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Borhani</surname> <given-names>Y.</given-names></name> <name><surname>Khoramdel</surname> <given-names>J.</given-names></name> <name><surname>Najafi</surname> <given-names>E.</given-names></name></person-group> (<year>2022</year>). <article-title>A deep learning based approach for automated plant disease classification using vision transformer</article-title>. <source>Sci. Rep</source>. <volume>12</volume>, <fpage>1</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-022-15163-0</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Casal</surname> <given-names>C.</given-names></name> <name><surname>Serratosa</surname> <given-names>J.</given-names></name> <name><surname>Tusell</surname> <given-names>J. M.</given-names></name></person-group> (<year>2002</year>). <article-title>Relationship between &#x003B2;-ap peptide aggregation and microglial activation</article-title>. <source>Brain Res</source>. <volume>928</volume>, <fpage>76</fpage>&#x02013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.1016/S0006-8993(01)03362-5</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chowdhary</surname> <given-names>K.</given-names></name></person-group> (<year>2020</year>). <article-title>Natural language processing</article-title>. <source>Fund. Artif. Intellig</source>. <volume>19</volume>, <fpage>603</fpage>&#x02013;<lpage>649</lpage>. <pub-id pub-id-type="doi">10.1007/978-81-322-3972-7_19</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chu</surname> <given-names>Z.</given-names></name> <name><surname>Huang</surname> <given-names>F.</given-names></name> <name><surname>Fu</surname> <given-names>H.</given-names></name> <name><surname>Quan</surname> <given-names>Y.</given-names></name> <name><surname>Zhou</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Hierarchical graph representation learning for the prediction of drug-target binding affinity</article-title>. <source>Inf. Sci</source>. <volume>613</volume>, <fpage>507</fpage>&#x02013;<lpage>523</lpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2022.09.043</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Creswell</surname> <given-names>A.</given-names></name> <name><surname>White</surname> <given-names>T.</given-names></name> <name><surname>Dumoulin</surname> <given-names>V.</given-names></name> <name><surname>Arulkumaran</surname> <given-names>K.</given-names></name> <name><surname>Sengupta</surname> <given-names>B.</given-names></name> <name><surname>Bharath</surname> <given-names>A. A.</given-names></name></person-group> (<year>2018</year>). <article-title>Generative adversarial networks: an overview</article-title>. <source>IEEE Signal Process. Mag</source>. <volume>35</volume>, <fpage>53</fpage>&#x02013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.1109/MSP.2017.2765202</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cummings</surname> <given-names>J. L.</given-names></name> <name><surname>Cole</surname> <given-names>G.</given-names></name></person-group> (<year>2002</year>). <article-title>Alzheimer disease</article-title>. <source>JAMA</source> <volume>287</volume>, <fpage>2335</fpage>&#x02013;<lpage>2338</lpage>. <pub-id pub-id-type="doi">10.1001/jama.287.18.2335</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dauparas</surname> <given-names>J.</given-names></name> <name><surname>Anishchenko</surname> <given-names>I.</given-names></name> <name><surname>Bennett</surname> <given-names>N.</given-names></name> <name><surname>Bai</surname> <given-names>H.</given-names></name> <name><surname>Ragotte</surname> <given-names>R. J.</given-names></name> <name><surname>Milles</surname> <given-names>L. F.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Robust deep learning-based protein sequence design using proteinmpnn</article-title>. <source>Science</source>. <volume>378</volume>, <fpage>49</fpage>&#x02013;<lpage>56</lpage>. <pub-id pub-id-type="doi">10.1126/science.add2187</pub-id></citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>De Cao</surname> <given-names>N.</given-names></name> <name><surname>Kipf</surname> <given-names>T.</given-names></name></person-group> (<year>2018</year>). <article-title>Molgan: An implicit generative model for small molecular graphs</article-title>, in <source>arXiv</source> preprint arXiv:1805.11973.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dileep</surname> <given-names>K.</given-names></name> <name><surname>Ihara</surname> <given-names>K.</given-names></name> <name><surname>Mishima-Tsumagari</surname> <given-names>C.</given-names></name> <name><surname>Kukimoto-Niino</surname> <given-names>M.</given-names></name> <name><surname>Yonemochi</surname> <given-names>M.</given-names></name> <name><surname>Hanada</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Crystal structure of human acetylcholinesterase in complex with tacrine: Implications for drug discovery</article-title>. <source>Int. J. Biol. Macromol</source>. <volume>210</volume>, <fpage>172</fpage>&#x02013;<lpage>181</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijbiomac.2022.05.009</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>W.</given-names></name> <name><surname>Abdel-Basset</surname> <given-names>M.</given-names></name> <name><surname>Hawash</surname> <given-names>H.</given-names></name> <name><surname>Ali</surname> <given-names>A. M.</given-names></name></person-group> (<year>2022</year>). <article-title>Explainability of artificial intelligence methods, applications and challenges: a comprehensive survey</article-title>. <source>Inform. Sci</source>. <volume>615</volume>, <fpage>238</fpage>&#x02013;<lpage>292</lpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2022.10.013</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>Y.</given-names></name> <name><surname>Tang</surname> <given-names>J.</given-names></name> <name><surname>Guo</surname> <given-names>F.</given-names></name></person-group> (<year>2017</year>). <article-title>Identification of drug-target interactions via multiple information integration</article-title>. <source>Inf. Sci</source>. <volume>418</volume>, <fpage>546</fpage>&#x02013;<lpage>560</lpage>. <pub-id pub-id-type="doi">10.1016/j.ins.2017.08.045</pub-id></citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ferruz</surname> <given-names>N.</given-names></name> <name><surname>Schmidt</surname> <given-names>S.</given-names></name> <name><surname>H&#x000F6;cker</surname> <given-names>B.</given-names></name></person-group> (<year>2022</year>). <article-title>Protgpt2 is a deep unsupervised language model for protein design</article-title>. <source>Nat. Commun</source>. <volume>13</volume>, <fpage>1</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-022-32007-7</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fintz</surname> <given-names>M.</given-names></name> <name><surname>Osadchy</surname> <given-names>M.</given-names></name> <name><surname>Hertz</surname> <given-names>U.</given-names></name></person-group> (<year>2022</year>). <article-title>Using deep learning to predict human decisions and using cognitive models to explain deep learning models</article-title>. <source>Sci. Rep</source>. <volume>12</volume>, <fpage>1</fpage>&#x02013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-022-08863-0</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gabrielson</surname> <given-names>S. W.</given-names></name></person-group> (<year>2018</year>). <article-title>Scifinder</article-title>. <source>J. Med. Library Assoc.: JMLA</source> <volume>106</volume>, <fpage>588</fpage>. <pub-id pub-id-type="doi">10.5195/jmla.2018.515</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>G&#x000F3;mez-Bombarelli</surname> <given-names>R.</given-names></name> <name><surname>Wei</surname> <given-names>J. N.</given-names></name> <name><surname>Duvenaud</surname> <given-names>D.</given-names></name> <name><surname>Hern&#x000E1;ndez-Lobato</surname> <given-names>J. M.</given-names></name> <name><surname>S&#x000E1;nchez-Lengeling</surname> <given-names>B.</given-names></name> <name><surname>Sheberla</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Automatic chemical design using a data-driven continuous representation of molecules</article-title>. <source>ACS Central Sci</source>. <volume>4</volume>, <fpage>268</fpage>&#x02013;<lpage>276</lpage>. <pub-id pub-id-type="doi">10.1021/acscentsci.7b00572</pub-id></citation>
</ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grisoni</surname> <given-names>F.</given-names></name> <name><surname>Moret</surname> <given-names>M.</given-names></name> <name><surname>Lingwood</surname> <given-names>R.</given-names></name> <name><surname>Schneider</surname> <given-names>G.</given-names></name></person-group> (<year>2020</year>). <article-title>Bidirectional molecule generation with recurrent neural networks</article-title>. <source>J. Chem. Inf. Model</source>. <volume>60</volume>, <fpage>1175</fpage>&#x02013;<lpage>1183</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.9b00943</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Irwin</surname> <given-names>J. J.</given-names></name> <name><surname>Shoichet</surname> <given-names>B. K.</given-names></name></person-group> (<year>2005</year>). <article-title>Zinc- a free database of commercially available compounds for virtual screening</article-title>. <source>J. Chem. Inf. Model</source>. <volume>45</volume>, <fpage>177</fpage>&#x02013;<lpage>182</lpage>. <pub-id pub-id-type="doi">10.1021/ci049714</pub-id>&#x0002B;</citation>
</ref>
<ref id="B21">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Jin</surname> <given-names>W.</given-names></name> <name><surname>Barzilay</surname> <given-names>R.</given-names></name> <name><surname>Jaakkola</surname> <given-names>T.</given-names></name></person-group> (<year>2018</year>). <article-title>Junction tree variational autoencoder for molecular graph generation</article-title>, in <source>International Conference on Machine Learning</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>2323</fpage>&#x02013;<lpage>2332</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kajino</surname> <given-names>H.</given-names></name></person-group> (<year>2019</year>). <article-title>Molecular hypergraph grammar with its application to molecular optimization</article-title>, in <source>International Conference on Machine Learning</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>3183</fpage>&#x02013;<lpage>3191</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kingma</surname> <given-names>D. P.</given-names></name> <name><surname>Mohamed</surname> <given-names>S.</given-names></name> <name><surname>Jimenez Rezende</surname> <given-names>D.</given-names></name> <name><surname>Welling</surname> <given-names>M.</given-names></name></person-group> (<year>2014</year>). <article-title>Semi-supervised learning with deep generative models</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. 2014, <fpage>27</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1406.5298</pub-id></citation>
</ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Klinkenberg</surname> <given-names>I.</given-names></name> <name><surname>Sambeth</surname> <given-names>A.</given-names></name> <name><surname>Blokland</surname> <given-names>A.</given-names></name></person-group> (<year>2011</year>). <article-title>Acetylcholine and attention</article-title>. <source>Behav. Brain Res</source>. <volume>221</volume>, <fpage>430</fpage>&#x02013;<lpage>442</lpage>. <pub-id pub-id-type="doi">10.1016/j.bbr.2010.11.033</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Korshunova</surname> <given-names>M.</given-names></name> <name><surname>Huang</surname> <given-names>N.</given-names></name> <name><surname>Capuzzi</surname> <given-names>S.</given-names></name> <name><surname>Radchenko</surname> <given-names>D. S.</given-names></name> <name><surname>Savych</surname> <given-names>O.</given-names></name> <name><surname>Moroz</surname> <given-names>Y. S.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Generative and reinforcement learning approaches for the automated de novo design of bioactive compounds</article-title>. <source>Commun. Chem</source>.<volume>5</volume>, <fpage>1</fpage>&#x02013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1038/s42004-022-00733-0</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Landrum</surname> <given-names>G.</given-names></name></person-group> (<year>2013</year>). <source>Rdkit: A Software Suite for Cheminformatics, Computational Chemistry, and Predictive Modeling</source>. <publisher-loc>Kentucky</publisher-loc>: <publisher-name>Greg Landrum</publisher-name>.</citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Xu</surname> <given-names>Y.</given-names></name> <name><surname>Yao</surname> <given-names>H.</given-names></name> <name><surname>Lin</surname> <given-names>K.</given-names></name></person-group> (<year>2020</year>). <article-title>Chemical space exploration based on recurrent neural networks: applications in discovering kinase inhibitors</article-title>. <source>J. Cheminform</source>. <volume>12</volume>, <fpage>1</fpage>&#x02013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1186/s13321-020-00446-3</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Masurkar</surname> <given-names>A. V.</given-names></name> <name><surname>Rusinek</surname> <given-names>H.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>B.</given-names></name> <name><surname>Zhu</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Generalizable deep learning model for early alzheimer&#x00027;s disease detection from structural mris</article-title>. <source>Sci. Rep</source>. <volume>12</volume>, <fpage>1</fpage>&#x02013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-022-20674-x</pub-id></citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>T.</given-names></name> <name><surname>Lin</surname> <given-names>Y.</given-names></name> <name><surname>Wen</surname> <given-names>X.</given-names></name> <name><surname>Jorissen</surname> <given-names>R. N.</given-names></name> <name><surname>Gilson</surname> <given-names>M. K.</given-names></name></person-group> (<year>2007</year>). <article-title>Bindingdb: a web-accessible database of experimentally determined protein-ligand binding affinities</article-title>. <source>Nucleic Acids Res</source>. <volume>35</volume>, <fpage>D198</fpage>&#x02013;<lpage>D201</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkl999</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Meyers</surname> <given-names>J.</given-names></name> <name><surname>Fabian</surname> <given-names>B.</given-names></name> <name><surname>Brown</surname> <given-names>N.</given-names></name></person-group> (<year>2021</year>). <article-title>De novo molecular design and generative models</article-title>. <source>Drug Discov. Today</source> <volume>26</volume>, <fpage>2707</fpage>&#x02013;<lpage>2715</lpage>. <pub-id pub-id-type="doi">10.1016/j.drudis.2021.05.019</pub-id></citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>M&#x000FC;ller</surname> <given-names>T. D.</given-names></name> <name><surname>Bl&#x000FC;her</surname> <given-names>M.</given-names></name> <name><surname>Tsch&#x000F6;p</surname> <given-names>M. H.</given-names></name> <name><surname>DiMarchi</surname> <given-names>R. D.</given-names></name></person-group> (<year>2022</year>). <article-title>Anti-obesity drug discovery: advances and challenges</article-title>. <source>Nat. Rev. Drug Disco</source>. <volume>21</volume>, <fpage>201</fpage>&#x02013;<lpage>223</lpage>. <pub-id pub-id-type="doi">10.1038/s41573-021-00337-8</pub-id></citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Olivecrona</surname> <given-names>M.</given-names></name> <name><surname>Blaschke</surname> <given-names>T.</given-names></name> <name><surname>Engkvist</surname> <given-names>O.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name></person-group> (<year>2017</year>). <article-title>Molecular de-novo design through deep reinforcement learning</article-title>. <source>J. Cheminform</source>. <volume>9</volume>, <fpage>1</fpage>&#x02013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1186/s13321-017-0235-x</pub-id></citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Popova</surname> <given-names>M.</given-names></name> <name><surname>Isayev</surname> <given-names>O.</given-names></name> <name><surname>Tropsha</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>Deep reinforcement learning for de novo drug design</article-title>. <source>Sci. Adv</source>. <volume>4</volume>, <fpage>eaap7885</fpage>. <pub-id pub-id-type="doi">10.1126/sciadv.aap7885</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Prykhodko</surname> <given-names>O.</given-names></name> <name><surname>Johansson</surname> <given-names>S. V.</given-names></name> <name><surname>Kotsias</surname> <given-names>P.-C.</given-names></name> <name><surname>Ar&#x000FA;s-Pous</surname> <given-names>J.</given-names></name> <name><surname>Bjerrum</surname> <given-names>E. J.</given-names></name> <name><surname>Engkvist</surname> <given-names>O.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>A de novo molecular generation method using latent vector based generative adversarial network</article-title>. <source>J. Cheminform</source>. <volume>11</volume>, <fpage>1</fpage>&#x02013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1186/s13321-019-0397-9</pub-id></citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qiu</surname> <given-names>S.</given-names></name> <name><surname>Miller</surname> <given-names>M. I.</given-names></name> <name><surname>Joshi</surname> <given-names>P. S.</given-names></name> <name><surname>Lee</surname> <given-names>J. C.</given-names></name> <name><surname>Xue</surname> <given-names>C.</given-names></name> <name><surname>Ni</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Multimodal deep learning for alzheimer&#x00027;s disease dementia assessment</article-title>. <source>Nat. Commun</source>. <volume>13</volume>, <fpage>1</fpage>&#x02013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1038/s41467-022-31037-5</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rogers</surname> <given-names>D.</given-names></name> <name><surname>Hahn</surname> <given-names>M.</given-names></name></person-group> (<year>2010</year>). <article-title>Extended-connectivity fingerprints</article-title>. <source>J. Chem. Inf. Model</source>. <volume>50</volume>, <fpage>742</fpage>&#x02013;<lpage>754</lpage>. <pub-id pub-id-type="doi">10.1021/ci100050t</pub-id></citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sambamurti</surname> <given-names>K.</given-names></name> <name><surname>Greig</surname> <given-names>N. H.</given-names></name> <name><surname>Utsuki</surname> <given-names>T.</given-names></name> <name><surname>Barnwell</surname> <given-names>E. L.</given-names></name> <name><surname>Sharma</surname> <given-names>E.</given-names></name> <name><surname>Mazell</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>Targets for ad treatment: conflicting messages from &#x003B3;-secretase inhibitors</article-title>. <source>J. Neurochem</source>. <volume>117</volume>, <fpage>359</fpage>&#x02013;<lpage>374</lpage>. <pub-id pub-id-type="doi">10.1111/j.1471-4159.2011.07213.x</pub-id></citation>
</ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schneider</surname> <given-names>G.</given-names></name></person-group> (<year>2010</year>). <article-title>Virtual screening: an endless staircase?</article-title> <source>Nat. Rev. Drug Disco</source>. <volume>9</volume>, <fpage>273</fpage>&#x02013;<lpage>276</lpage>. <pub-id pub-id-type="doi">10.1038/nrd3139</pub-id></citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shoichet</surname> <given-names>B. K.</given-names></name></person-group> (<year>2004</year>). <article-title>Virtual screening of chemical libraries</article-title>. <source>Nature</source> <volume>432</volume>, <fpage>862</fpage>&#x02013;<lpage>865</lpage>. <pub-id pub-id-type="doi">10.1038/nature03197</pub-id></citation>
</ref>
<ref id="B40">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Simonovsky</surname> <given-names>M.</given-names></name> <name><surname>Komodakis</surname> <given-names>N.</given-names></name></person-group> (<year>2018</year>). <article-title>Graphvae: Towards generation of small graphs using variational autoencoders</article-title>, in <source>International Conference on Artificial Neural Networks</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>412</fpage>&#x02013;<lpage>422</lpage>.</citation>
</ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Trott</surname> <given-names>O.</given-names></name> <name><surname>Olson</surname> <given-names>A. J.</given-names></name></person-group> (<year>2010</year>). <article-title>Autodock vina: improving the speed and accuracy of docking with a new scoring function, efficient optimization, and multithreading</article-title>. <source>J. Comput. Chem</source>. <volume>31</volume>, <fpage>455</fpage>&#x02013;<lpage>461</lpage>. <pub-id pub-id-type="doi">10.1002/jcc.21334</pub-id></citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Voulodimos</surname> <given-names>A.</given-names></name> <name><surname>Doulamis</surname> <given-names>N.</given-names></name> <name><surname>Doulamis</surname> <given-names>A.</given-names></name> <name><surname>Protopapadakis</surname> <given-names>E.</given-names></name></person-group> (<year>2018</year>). <article-title>Deep learning for computer vision: a brief review</article-title>. <source>Comput. Intell. Neurosci</source>. 2018, <fpage>7068349</fpage>. <pub-id pub-id-type="doi">10.1155/2018/7068349</pub-id></citation>
</ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Walters</surname> <given-names>W. P.</given-names></name> <name><surname>Barzilay</surname> <given-names>R.</given-names></name></person-group> (<year>2020</year>). <article-title>Applications of deep learning in molecule generation and molecular property prediction</article-title>. <source>Acc. Chem. Res</source>. <volume>54</volume>, <fpage>263</fpage>&#x02013;<lpage>270</lpage>. <pub-id pub-id-type="doi">10.1021/acs.accounts.0c00699</pub-id></citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Weininger</surname> <given-names>D.</given-names></name></person-group> (<year>1988</year>). <article-title>Smiles, a chemical language and information system. 1. introduction to methodology and encoding rules</article-title>. <source>J. Chem. Informat. Comp.Sci</source>. <volume>28</volume>, <fpage>31</fpage>&#x02013;<lpage>36</lpage>. <pub-id pub-id-type="doi">10.1021/ci00057a005</pub-id></citation>
</ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>K.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Cai</surname> <given-names>C.</given-names></name> <name><surname>Song</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Deep learning for molecular generation</article-title>. <source>Future Med. Chem</source>. <volume>11</volume>, <fpage>567</fpage>&#x02013;<lpage>597</lpage>. <pub-id pub-id-type="doi">10.4155/fmc-2018-0358</pub-id></citation>
</ref>
</ref-list>
</back>
</article>