<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="review-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Toxicol.</journal-id>
<journal-title>Frontiers in Toxicology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Toxicol.</abbrev-journal-title>
<issn pub-type="epub">2673-3080</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1340860</article-id>
<article-id pub-id-type="doi">10.3389/ftox.2023.1340860</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Toxicology</subject>
<subj-group>
<subject>Mini Review</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Computational models for predicting liver toxicity in the deep learning era</article-title>
<alt-title alt-title-type="left-running-head">Mostafa and Chen</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/ftox.2023.1340860">10.3389/ftox.2023.1340860</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Mostafa</surname>
<given-names>Fahad</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1586786/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Chen</surname>
<given-names>Minjun</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/304873/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Mathematics and Statistics</institution>, <institution>Texas Tech University</institution>, <addr-line>Lubbock</addr-line>, <addr-line>TX</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Division of Bioinformatics and Biostatistics</institution>, <institution>National Center for Toxicological Research</institution>, <institution>U.S. Food and Drug Administration</institution>, <addr-line>Jefferson</addr-line>, <addr-line>AR</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2147057/overview">Chao Ji</ext-link>, Indiana University, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/464446/overview">Sijie Lin</ext-link>, Tongji University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Minjun Chen, <email>minjun.chen@fda.hhs.gov</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>19</day>
<month>01</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>5</volume>
<elocation-id>1340860</elocation-id>
<history>
<date date-type="received">
<day>19</day>
<month>11</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>22</day>
<month>12</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Mostafa and Chen.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Mostafa and Chen</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Drug-induced liver injury (DILI) is a severe adverse reaction caused by drugs and may result in acute liver failure and even death. Many efforts have centered on mitigating risks associated with potential DILI in humans. Among these, quantitative structure-activity relationship (QSAR) was proven to be a valuable tool for early-stage hepatotoxicity screening. Its advantages include no requirement for physical substances and rapid delivery of results. Deep learning (DL) made rapid advancements recently and has been used for developing QSAR models. This review discusses the use of DL in predicting DILI, focusing on the development of QSAR models employing extensive chemical structure datasets alongside their corresponding DILI outcomes. We undertake a comprehensive evaluation of various DL methods, comparing with those of traditional machine learning (ML) approaches, and explore the strengths and limitations of DL techniques regarding their interpretability, scalability, and generalization. Overall, our review underscores the potential of DL methodologies to enhance DILI prediction and provides insights into future avenues for developing predictive models to mitigate DILI risk in humans.</p>
</abstract>
<kwd-group>
<kwd>drug-induced liver injury (DILI)</kwd>
<kwd>machine learning</kwd>
<kwd>deep learning</kwd>
<kwd>drug safety</kwd>
<kwd>predictive model</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Toxicology and Informatics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Drug-induced liver injury (DILI) is a substantial safety concern, with a reported potential for more than 1,000 drugs or supplements to induce liver damage (<xref ref-type="bibr" rid="B2">Alempijevic et al., 2017</xref>; <xref ref-type="bibr" rid="B50">Zhu et al., 2018</xref>). DILI presents a significant challenge for healthcare professionals, pharmaceutical developers, and regulatory authorities (<xref ref-type="bibr" rid="B11">George et al., 2018</xref>), and frequently results in the discontinuation of drug candidates during their development (<xref ref-type="bibr" rid="B39">Weber and Gerbes, 2022</xref>). It also is a primary reason for the withdrawal of over 50 medications from the market (<xref ref-type="bibr" rid="B8">Devarbhavi, 2012</xref>; <xref ref-type="bibr" rid="B43">Wu et al., 2022</xref>) and ranks as a leading cause of acute liver failure in both the United States and Europe (<xref ref-type="bibr" rid="B3">Andrade et al., 2019</xref>). Despite notable advancements in drug safety, there is a continuing need for innovative approaches and methodologies to identify drugs candidates in development with potential hepatotoxicity in humans, and for reliable biomarkers to facilitate the early detection of DILI (<xref ref-type="bibr" rid="B4">Chen et al., 2014</xref>).</p>
<p>The need to enhance safety assessments in drug development has driven new approaches for predicting toxicity. Conventional methods often lack the precision and efficiency required to mitigate the risks associated with liver toxicity. However, machine learning, (ML), which includes Quantitative Structure-Activity Relationship (QSAR) modeling (<xref ref-type="bibr" rid="B17">Idakwo et al., 2019</xref>; <xref ref-type="bibr" rid="B35">Shin et al., 2023</xref>) as a pivotal component, harnesses extensive datasets, chemical structures, and biological assays to establish quantitative associations between molecular properties and toxicity outcomes (<xref ref-type="bibr" rid="B41">Wu et al., 2017</xref>). This approach has the potential to facilitate the detection of liver toxicity during the early stage drug development process, enabling screening of drug candidates and their analogs prior to chemical synthesis.</p>
<p>An advanced ML technique, deep learning (DL), signifies a transformative approach in the field of liver toxicity prediction, offering the potential for exceptionally accurate, data-driven insights. DL harnesses neural networks and extensive datasets, which encompass chemical data, biological assays, and omics information, to construct predictive models of outstanding performance (<xref ref-type="bibr" rid="B45">Xu et al., 2015</xref>; <xref ref-type="bibr" rid="B12">Goh et al., 2018</xref>). Its integration into liver toxicity prediction empowers researchers and pharmaceutical companies to identify potential risks associated with drug candidates at an early stage in the development process. Moreover, its capacity to analyze diverse and intricate data sources facilitates a better understanding of toxicity mechanisms. Consequently, DL not only advances patient safety by aiding in identifying harmful compounds, but also is cost-effective and contributes to the accelerated development of safer and more effective medications.</p>
<p>In this review, we focus on cutting-edge research using ML/DL applications to predict liver toxicity. We first examine the application of ML in liver toxicity prediction, with a particular emphasis on the development of QSAR models. Next, we provide a systematic evaluation of DL methods and their application for predicting liver toxicity, drawing comparisons with traditional ML approaches. Finally, we discuss the strengths and limitations of DL methods in interpretability, scalability, and generalization.</p>
</sec>
<sec id="s2">
<title>2 Machine learning for predicting liver toxicity</title>
<p>Machine learning (ML) algorithms have extensive applications in classification tasks, including the prediction of liver toxicity (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref>). In binary classification, compounds are typically categorized into two classes: a toxic class (commonly labeled class 1) and a non-toxic class (class 0). ML algorithms can learn from historical data and categorize new instances into one of these two classes by considering their observed characteristics, such as chemical structures. Among the various ML methods available, Naive Bayes Classifier (NBC), Support Vector Machines (SVM), and Random Forests are widely employed in this context.</p>
<sec id="s2-1">
<title>2.1 Naive Bayes classifier</title>
<p>The Naive Bayes classifier (NBC) is a probabilistic ML algorithm widely used for both binary and multiclass classification tasks (<xref ref-type="bibr" rid="B31">Rish, 2001</xref>; <xref ref-type="bibr" rid="B13">Hastie et al., 2009</xref>). It is rooted in Bayes&#x2019; theorem, which quantifies the probability of an event based on prior knowledge of related events. The &#x201c;Naive&#x201d; component of its name comes from the assumption that input features are conditionally independent, simplifying calculations and enhancing computational efficiency. Thus, the NBC computes the conditional probability of a given instance belonging to a specific class by making the &#x201c;naive&#x201d; assumption of feature independence. Mathematically, it leverages Bayes&#x2019; theorem:<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">C</mml:mi>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msub>
<mml:mo>&#x7c;</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">C</mml:mi>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msub>
<mml:mo>&#x7c;</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x220f;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi mathvariant="bold-italic">n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">x</mml:mi>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:msub>
<mml:mo>&#x7c;</mml:mo>
<mml:mi mathvariant="bold-italic">C</mml:mi>
</mml:mrow>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">C</mml:mi>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the class, <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the feature vectors.</p>
<p>The NBC makes the &#x201c;naive&#x201d; assumption that features are independent given the class. This strong assumption might not hold in all real-world scenarios. However, despite this simplification, it often performs surprisingly well and is computationally efficient. Critical steps to train and use NBC are listed below:<list list-type="simple">
<list-item>
<p>1. Calculate Class Priors: Estimate the prior probabilities <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:mi mathvariant="bold-italic">P</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">C</mml:mi>
<mml:mi mathvariant="bold-italic">k</mml:mi>
</mml:msub>
<mml:mo>&#x7c;</mml:mo>
<mml:mi mathvariant="bold-italic">x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> for each class based on the training data.</p>
</list-item>
<list-item>
<p>2. Calculate Feature Probabilities: Estimate the conditional probabilities <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x7c;</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> for each feature and class pair based on the training data. This involves counting occurrences of features in each class.</p>
</list-item>
<list-item>
<p>3. Classification: Given a new instance with features <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, calculate the posterior probability for each class using Bayes&#x2019; theorem. The class with the highest probability is the predicted class.</p>
</list-item>
</list>
</p>
<p>Variations of NBCs are based on types of data and assumptions. Some common variations are:<list list-type="simple">
<list-item>
<p>&#x2022; Gaussian Naive Bayes: Assumes features follow a Gaussian (normal) distribution.</p>
</list-item>
<list-item>
<p>&#x2022; Multinomial Naive Bayes: Suited for discrete features like text data, and often used for document classification.</p>
</list-item>
<list-item>
<p>&#x2022; Bernoulli Naive Bayes: Designed for binary feature data (presence/absence), and often used for text classification tasks.</p>
</list-item>
</list>
</p>
<p>NBC is a straightforward yet effective classifier, particularly suitable for binary classification tasks, provided that the assumption of feature independence is reasonably met. While it may not be the optimal choice for all data types, it serves as a standardized baseline classifier and is extensively employed in the prediction DILI through QSAR modeling (<xref ref-type="bibr" rid="B1">Ai et al., 2019</xref>; <xref ref-type="bibr" rid="B40">Williams et al., 2019</xref>; <xref ref-type="bibr" rid="B42">Wu Y. et al., 2021</xref>). For instance, <xref ref-type="bibr" rid="B48">Zhang et al. (2016)</xref> employed NBC to construct a computational model for assessing DILI risk. Their model exhibited a 94.0% accuracy in 5-fold cross-validation during the training phase, with a concordance rate of 72.6% on an external test set. They identified key molecular characteristics associated with DILI risk.</p>
<p>
<xref ref-type="bibr" rid="B37">Tang et al. (2020)</xref> developed QSAR models for mitochondrial toxicity using five machine learning methods, including NBC along with various chemical signatures. They adopted a threshold moving strategy to rectify data imbalance and implemented consensus models to enhance prediction performance, achieving up to 88.3% accuracy in external validation. Notably, the study highlighted the significance of substructures such as phenol, carboxylic acid, nitro compounds, and aryl chloride in classification. In another work, <xref ref-type="bibr" rid="B30">Rao et al. (2023)</xref>, proposed an integrated artificial intelligence (AI)/ML model that employed physicochemical properties and <italic>in silico</italic> off-target interactions to predict the severity of DILI for small molecules. They utilized data from 603 compounds categorized by the U.S. Food and Drug Administration (FDA) as Most DILI, Less DILI, and No DILI, and combined the NBC with other ML approaches to enhance DILI prediction, surpassing the performance of QSAR models based solely on chemical properties.</p>
</sec>
<sec id="s2-2">
<title>2.2 Support vector machine classifier</title>
<p>The Support Vector Machine (SVM) is a potent and versatile machine learning algorithm used for a range of tasks, including regression, binary, and multiclass classification, as is the case in predicting DILI through QSAR modeling with chemical structures (<xref ref-type="bibr" rid="B23">Li et al., 2020a</xref>; <xref ref-type="bibr" rid="B37">Tang et al., 2020</xref>; <xref ref-type="bibr" rid="B44">Wu Z et al., 2021</xref>; <xref ref-type="bibr" rid="B30">Rao et al., 2023</xref>). Its primary goal is to identify a hyperplane that maximizes the margin between the nearest data points from the two classes. The fundamental concept is to optimize this margin between classes, resulting in improved generalization to new, unseen data. These closest data points are referred to as &#x201c;support vectors.&#x201d; The margin is defined as the distance between the hyperplane and these support vectors. Mathematically, the SVM tries to solve the following optimization problem:<disp-formula id="e2">
<mml:math id="m7">
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mi mathvariant="italic">min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Subject to<disp-formula id="equ1">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>&#x2219;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mi>f</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where:<list list-type="simple">
<list-item>
<p>&#x2022; w is the weight vector perpendicular to the hyperplane.</p>
</list-item>
<list-item>
<p>&#x2022; b is the bias term.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf6">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the feature vectors.</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf7">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the class labels (1 or 0) for each data point.</p>
</list-item>
<list-item>
<p>&#x2022; n is the number of data points.</p>
</list-item>
</list>
</p>
<p>The above optimization problem ensures that data points are correctly classified with a margin. Support vectors are the data points that lie on the margins or violate the margin constraint. In many cases, the data may not be linearly separable in the original feature space. To handle these cases, SVMs often use a kernel trick. A kernel function transforms the original feature space into a higher-dimensional space, where the data might become separable. Common kernel functions include:<list list-type="simple">
<list-item>
<p>&#x2022; Linear Kernel: <inline-formula id="inf8">
<mml:math id="m11">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2219;</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2022; Polynomial Kernel: <inline-formula id="inf9">
<mml:math id="m12">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2219;</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>d</mml:mi>
</mml:msup>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2022; Radial Basis Function (RBF) Kernel: <inline-formula id="inf10">
<mml:math id="m13">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msup>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
</list>
</p>
<p>In some cases, the data might not be perfectly separable, or there could be outliers. In such situations, SVM employed a soft margin to allow for some misclassification by introducing a slack variable. The optimization problem becomes:<disp-formula id="e3">
<mml:math id="m14">
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mi mathvariant="italic">min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Subject to:<disp-formula id="equ2">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>&#x2219;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf11">
<mml:math id="m16">
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a hyperparameter that controls the trade-off between maximizing the margin and minimizing misclassification. Major steps to train and apply the SVM Classifier are listed below.<list list-type="simple">
<list-item>
<p>1. Data Preparation: Gather and preprocess the data, ensuring it is properly labeled, and features are appropriately represented.</p>
</list-item>
<list-item>
<p>2. Choose a Kernel: Decide on a kernel function based on the data characteristics, which can be critical to improving accuracy in DILI predictions (<xref ref-type="bibr" rid="B42">Wu Y. et al., 2021</xref>).</p>
</list-item>
<list-item>
<p>3. Train the SVM: Use an optimization algorithm to find the optimal hyperplane parameters (weights w and bias b) that minimize the objective function.</p>
</list-item>
<list-item>
<p>4. Classification: Given a new instance with features x, calculate the decision function <inline-formula id="inf12">
<mml:math id="m17">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo>&#x2219;</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. If <inline-formula id="inf13">
<mml:math id="m18">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, classify as class 1; if <inline-formula id="inf14">
<mml:math id="m19">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, classify as class 0. SVMs can be computationally intensive for large datasets, and tuning the hyperparameters, such as the choice of kernel and the regularization parameter C, is essential for optimal performance.</p>
</list-item>
</list>
</p>
<p>SVM has been applied extensively in predicting DILI, particularly in scenarios with limited data, owing to its robust prediction accuracy and computational efficiency. Notable studies (<xref ref-type="bibr" rid="B38">Wang et al., 2019</xref>; <xref ref-type="bibr" rid="B24">Li et al., 2020b</xref>; <xref ref-type="bibr" rid="B14">Hemmerich et al., 2020</xref>; <xref ref-type="bibr" rid="B28">Mora et al., 2020</xref>; <xref ref-type="bibr" rid="B44">Wu Z. et al., 2021</xref>) have employed SVM for DILI prediction. In a comprehensive analysis conducted by <xref ref-type="bibr" rid="B42">Wu Y. et al. (2021)</xref>, involving 14 sets of QSAR data and 16 ML algorithms, the radial basis function SVM (rbf-SVM) emerged as the top-performing method among all ML techniques, underscoring its efficacy in this domain.</p>
<p>However, certain limitations were associated with the SVM algorithm. SVM tends to be computationally expensive and may not be well-suited for very large datasets. When the dataset exhibits extra noise, such as overlapping target classes, SVM&#x2019;s performance can be compromised. Furthermore, SVM may perform suboptimally when the number of features for each data point exceeds the number of training data samples. These considerations are critical when deciding on the suitability of SVM for specific DILI prediction tasks.</p>
</sec>
<sec id="s2-3">
<title>2.3 Random forest classifier</title>
<p>The Random Forest classifier is a robust ensemble ML algorithm frequently employed for both binary and multiclass classification tasks. It is an extension of the decision tree algorithm, having the primary objective of enhancing generalization and mitigating overfitting by forming an ensemble of multiple decision trees. In binary classification, the Random Forest classifier is designed to classify new instances into one of two classes based on their features. The process involves the construction of multiple decision trees during the training phase, and their collective predictions are amalgamated to reach the final classification decision. Key steps for training and applying the Random Forest classifier are outlined below.<list list-type="simple">
<list-item>
<p>1. Bootstrapped Sampling: For each tree in the forest, a random subset of the training data is selected with replacement. This process is known as bootstrapped sampling. It creates diversity among the trees, as each tree is trained on a slightly different data subset.</p>
</list-item>
<list-item>
<p>2. Random Feature Selection: At each split point in a decision tree, only a subset of the available features is considered for splitting. This introduces further randomness and prevents individual trees from relying on any one feature.</p>
</list-item>
<list-item>
<p>3. Tree Building: Each decision tree is constructed using the bootstrapped training data and random feature selection. The tree is grown until a stopping criterion is met, usually involving the maximum depth of the tree or the minimum number of samples required to split a node.</p>
</list-item>
<list-item>
<p>4. Voting for Classification: During prediction, each tree in the forest independently classifies the input data. The final classification decision is made by taking a majority vote among the individual tree predictions. In the case of binary classification, the class with the most votes wins.</p>
</list-item>
</list>
</p>
<p>Compared with other machine learning methods, random forest has several unique characteristics:<list list-type="simple">
<list-item>
<p>&#x2022; Reduced Overfitting: The ensemble of trees helps to mitigate overfitting by averaging out the noise and biases present in individual trees.</p>
</list-item>
<list-item>
<p>&#x2022; Improved Generalization: Random Forests are robust to outliers and noisy data due to the aggregation of multiple trees.</p>
</list-item>
<list-item>
<p>&#x2022; Feature Importance: Random Forests can provide insights into feature importance by analyzing how much each feature contributes to the model&#x2019;s performance.</p>
</list-item>
<list-item>
<p>&#x2022; Non-linearity Handling: Random Forests can capture complex relationships in the data without requiring explicit feature extraction/selection.</p>
</list-item>
</list>
</p>
<p>Random Forest emerges as an invaluable machine learning technique for the classification of liver toxicity. <xref ref-type="bibr" rid="B10">Gadaleta et al. (2018)</xref> employed Random Forest classifiers and DRAGON molecular descriptors to create QSAR models designed to predict molecular initiating events leading to hepatic steatosis. They effectively used a Balanced Random Forest classifier, alongside the strategy of under-sampling, to construct robust QSAR models from unbalanced DILI datasets. Both techniques yielded comparable predictive results, achieving approximately 75% accuracy in toxicity prediction.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Deep learning for predicting liver toxicity</title>
<p>Deep learning (DL) represents a new class of machine learning methods characterized by the use of highly complex neural networks. Networks are structured in deeply nested architectures, often incorporating advanced operations like convolutions and multiple activation functions. These distinctive features empower DL with the unique capability to process raw input data and autonomously uncover hidden patterns for learning tasks. In the context of predicting liver toxicity, several DL methods are commonly employed for classification tasks. These methods deploy neural networks with diverse architectures and techniques to achieve precise and efficient classification (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref>). We provide a brief overview of various DL methods, including multi-layer perceptron (MLP), deep neural network (DNN), convolutional neural network (CNN), graph neural network (GNN), recurrent neural network (RNN), generative adversarial network (GAN), and transformer.</p>
<sec id="s3-1">
<title>3.1 Multi-layer perceptron</title>
<p>The Multilayer Perceptron (MLP), also known as an Artificial Neural Network (ANN), is a fundamental neural network architecture used for a wide range of machine learning applications, including classification and regression. An MLP consists of multiple layers of artificial neurons, typically structured into an input layer, one or more hidden layers, and an output layer. Each neuron within a layer is connected to each neuron in the layers above and below it, creating a densely interconnected network. Connections between neurons, represented as weights (often denoted as W), are learned during the training process. The output of each neuron is determined by applying an activation function, such as the sigmoid, ReLU, or tanh function, to a weighted sum of its inputs. Mathematically, the output (O) of a neuron in a hidden or output layer is computed as follows:<disp-formula id="equ3">
<mml:math id="m20">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>.</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf15">
<mml:math id="m21">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the output of the neuron. <inline-formula id="inf16">
<mml:math id="m22">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the activation function. <inline-formula id="inf17">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the weight associated with the i-th input connection. <inline-formula id="inf18">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the i-th input to the neuron, and <inline-formula id="inf19">
<mml:math id="m25">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the bias term. The training process involves adjusting these weights and biases using techniques like backpropagation and gradient descent to minimize a loss function, allowing the MLP to learn complex relationships within the data.</p>
<p>MLPs are versatile and can approximate a wide range of functions, making them a popular choice for various ML applications. <xref ref-type="bibr" rid="B7">Cruz-Monteagudo et al. (2008)</xref> investigated computational approaches for predicting idiosyncratic hepatotoxicity using 3D chemical structures such as linear discriminant analysis (LDA) and ANNs. The RBF architecture was used in a neural network classification method that used the same descriptors as those in the LDA model. In the training series, this RBF neural network outperforms the LDA model, achieving an accuracy of 91.07%, sensitivity of 92.00%, and specificity of 90.32%. Examination of the Receiver Operating Characteristic (ROC) curve proved its continuously superior performance.</p>
</sec>
<sec id="s3-2">
<title>3.2 Deep neural networks</title>
<p>A Deep Neural Network (DNN) can be mathematically represented as a composition of functions (<xref ref-type="bibr" rid="B34">Schmidhuber, 2015</xref>). Given an input vector <inline-formula id="inf20">
<mml:math id="m26">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the output <inline-formula id="inf21">
<mml:math id="m27">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of a DNN with <inline-formula id="inf22">
<mml:math id="m28">
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> layers can be expressed as:<disp-formula id="equ4">
<mml:math id="m29">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mo>&#x2218;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2218;</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>&#x2218;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2218;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Each layer <inline-formula id="inf23">
<mml:math id="m30">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> applies a linear transformation <inline-formula id="inf24">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> followed by an activation function <inline-formula id="inf25">
<mml:math id="m32">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> where <inline-formula id="inf26">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the weight matrix and <inline-formula id="inf27">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the bias vector for that layer. The final output is obtained by applying an appropriate activation function at the last layer. Like MLP, DNN can be trained by adjusting the weights and biases to minimize a chosen loss function through techniques like backpropagation and optimization algorithms. DNNs are excellent for automatically extracting key features from large inputs, making them the perfect choice for transcriptomic data containing a wide variety of features.</p>
<p>DNNs (<xref ref-type="bibr" rid="B15">Hinton et al., 2006</xref>) have been effectively used to address the challenge of predicting various types of chemically induced liver injuries, including biliary hyperplasia, fibrosis, and necrosis, using DNA microarray data (<xref ref-type="bibr" rid="B9">Feng et al., 2019</xref>). <xref ref-type="bibr" rid="B38">Wang et al. (2019)</xref> used multi-task DNNs to evaluate gene and pathway-level feature selection strategies for these liver injuries. The DNN models exhibited high predictive accuracy and endpoint specificity, surpassing the performance of Random Forest and SVM models. In another study, <xref ref-type="bibr" rid="B24">Li et al. (2020b)</xref>, developed a DNN model with eight layers using transcriptome profiles of human cell lines to predict DILI. The model leveraged a substantial binary DILI annotation dataset, achieving AUCs of 0.802 and 0.798 for the training and independent validation sets, respectively. These results outperformed traditional machine learning algorithms, including K-nearest neighbors, SVM, and Random Forest.</p>
<p>In a study conducted by <xref ref-type="bibr" rid="B19">Kang and Kang (2021)</xref>, a DNN-based model was designed to predict DILI risk. This model used extended connectivity fingerprinting of diameter 4 (ECFP4) to represent molecular substructures. The data for this predictive model was meticulously collected from various sources, including publications like DILIrank and LiverTox. A model was developed through stratified 10-fold cross-validation, and the best DNN model showed an accuracy of 0.731, a sensitivity of 0.714, and a specificity of 0.750 when validated in the complete applicability domain. <xref ref-type="bibr" rid="B18">Jain et al. (2021)</xref> used a large-scale acute toxicity dataset encompassing over 80,000 compounds measured against 59 toxicity endpoints. They compared multiple single and multitask models using RF, DNN, CNN, and GNN approaches and found that multitask DL methods performed best.</p>
</sec>
<sec id="s3-3">
<title>3.3 Convolutional neural networks</title>
<p>Convolutional Neural Networks (CNNs) are mainly applied in image and speech recognition. These networks are well-suited for capturing spatial hierarchies and local patterns within images. CNNs typically incorporate convolutional layers for feature extraction, pooling layers for dimensionality reduction, and fully connected layers for classification. Architectures like AlexNet, VGG, ResNet, and InceptionNet have consistently demonstrated exceptional performance on various image classification tasks (<xref ref-type="bibr" rid="B21">Krizhevsky et al., 2012</xref>; <xref ref-type="bibr" rid="B36">Szegedy et al., 2015</xref>; <xref ref-type="bibr" rid="B46">Yamashita et al., 2018</xref>; <xref ref-type="bibr" rid="B18">Jain et al., 2021</xref>). Mathematically, CNN can be written as follows: Let <inline-formula id="inf28">
<mml:math id="m35">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> be the dataset having <inline-formula id="inf29">
<mml:math id="m36">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> number of images. The input feature size is denoted as <inline-formula id="inf30">
<mml:math id="m37">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Then the convolution layer is written as:<disp-formula id="e4">
<mml:math id="m38">
<mml:mrow>
<mml:msubsup>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>f</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>f</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>q</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mi>b</mml:mi>
<mml:mi>l</mml:mi>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ5">
<mml:math id="m39">
<mml:mrow>
<mml:msup>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Here, <inline-formula id="inf31">
<mml:math id="m40">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are the number of convolution layers. Next, the CNN has the pooling layer:<disp-formula id="equ6">
<mml:math id="m41">
<mml:mrow>
<mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mi mathvariant="italic">max</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mi>j</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mi>j</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="e5">
<mml:math id="m42">
<mml:mrow>
<mml:msup>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Flatten the pooled feature maps to obtain a vector of size <inline-formula id="inf32">
<mml:math id="m43">
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Now the fully connected layer is defined as:<disp-formula id="equ7">
<mml:math id="m44">
<mml:mrow>
<mml:msup>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mo>&#x2219;</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="e6">
<mml:math id="m45">
<mml:mrow>
<mml:msup>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>The output layer has a single neuron for binary classification or multiple neurons for multi-class classification:<disp-formula id="e7">
<mml:math id="m46">
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>A suitable loss function, such as binary cross-entropy, is used for classification. The network is trained using gradient descent-based optimization to minimize the chosen loss function.</p>
<p>CNN was also used for DILI prediction (<xref ref-type="bibr" rid="B29">Nguyen-Vo et al., 2020</xref>; <xref ref-type="bibr" rid="B5">Chen X. et al., 2022</xref>). <xref ref-type="bibr" rid="B29">Nguyen-Vo et al. (2020)</xref> introduced a novel computational model for the prediction of DILI utilizing CNNs and molecular fingerprints based on 1,597 compounds. The model came up with an average accuracy of 0.89, a Matthews correlation coefficient of 0.80, and an AUC of 0.96.</p>
</sec>
<sec id="s3-4">
<title>3.4 Graph Neural Networks</title>
<p>Graph Neural Networks (GNNs) are a class of neural networks explicitly tailored for operating on graph data structures. They are particularly well-suited for tasks involving graphs, such as social network analysis, chemical structure analysis, and computational vision (<xref ref-type="bibr" rid="B49">Zhou et al., 2020</xref>). Node-level tasks are used in DILI prediction and chemical structure analysis, and involve predicting the properties or characteristics of individual chemical components, such as molecules, within a graph or network structure.</p>
<p>To learn node representations, GNNs combine information from nearby nodes, effectively capturing intricate relationships in graphs. For example, let <inline-formula id="inf33">
<mml:math id="m47">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> be the molecular graph, where <inline-formula id="inf34">
<mml:math id="m48">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the set of nodes (atoms) and <inline-formula id="inf35">
<mml:math id="m49">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the set of edges (bonds). The graph convolutional layer updates node representations based on their neighbors&#x2019; features. Let <inline-formula id="inf36">
<mml:math id="m50">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> be the initial node features (molecular fingerprint-embedded features) for all nodes in the graph. The output of the <inline-formula id="inf37">
<mml:math id="m51">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> th graph convolutional layer can be represented as <inline-formula id="inf38">
<mml:math id="m52">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> using the following equation:<disp-formula id="e8">
<mml:math id="m53">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mover accent="true">
<mml:mi>D</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msup>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:msup>
<mml:mover accent="true">
<mml:mi>D</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Here, <inline-formula id="inf39">
<mml:math id="m54">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the adjacency matrix of the graph with added self-loops, <inline-formula id="inf40">
<mml:math id="m55">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>D</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> is the diagonal of matrix <inline-formula id="inf41">
<mml:math id="m56">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf42">
<mml:math id="m57">
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the learnable weights for <inline-formula id="inf43">
<mml:math id="m58">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> th layers, and <inline-formula id="inf44">
<mml:math id="m59">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the activation function. Pooling or aggregation layers were incorporated to combine node features across different neighborhoods. Similar with the CNN architecture, one or more fully-connected layers were used to learn higher-level representations from the aggregated features.</p>
<p>The final layer produces the network&#x2019;s output. Depending on the task (e.g., regression, classification), the number of neurons and the activation function in the output layer can be adjusted. The forward pass through the GNN can be represented mathematically as written below. Let <inline-formula id="inf45">
<mml:math id="m60">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> be the molecular fingerprint-embedded features. The graph convolution is:<disp-formula id="e9">
<mml:math id="m61">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mover accent="true">
<mml:mi>D</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msup>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:msup>
<mml:mover accent="true">
<mml:mi>D</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtext>Aggregated&#x2009;Features</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Pooling</mml:mtext>
<mml:mo>/</mml:mo>
<mml:mtext>Aggregation&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Dense&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>Aggregated&#x2009;Features</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>neurons</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>activation</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Dense&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>neurons</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>activation</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtext>Output</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Dense&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>output&#x2009;neurons</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>output&#x2009;activation</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>Here, k represents the number of dense layers in the network.</p>
<p>GNNs have demonstrated their efficacy in addressing node-level tasks related to DILI predictions (<xref ref-type="bibr" rid="B16">Hwang et al., 2020</xref>). <xref ref-type="bibr" rid="B27">Ma et al. (2020)</xref> used a MV-GNN based model as a backbone to propose a property augmentation approach to involving more data for four datasets with liver toxicity-relevant properties. The GNN-based approach significantly outperformed existing baselines on DILI datasets, achieving an impressive 81.4% accuracy using cross-validation with random splitting. <xref ref-type="bibr" rid="B25">Lim et al. (2023)</xref> introduced a novel technique known as supervised subgraph mining (SSM). SSM effectively identifies explicit subgraph features through iterative optimization of graph transitions. This approach surpasses conventional machine learning methods such as SVM, Random Forest, k-Nearest Neighbors, and deep learning neural networks in DILI classification using two datasets, DILIst and TDC-benchmark. By employing structure-based pattern matching, the proposed approach can also identify subgraph characteristics associated with specific medication groups.</p>
</sec>
<sec id="s3-5">
<title>3.5 Recurrent neural networks</title>
<p>Mathematically, a recurrent neural network (RNN) can be represented as follows: at each time step <inline-formula id="inf46">
<mml:math id="m62">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the RNN takes an input vector <inline-formula id="inf47">
<mml:math id="m63">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and computes the hidden state <inline-formula id="inf48">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the output <inline-formula id="inf49">
<mml:math id="m65">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> using the following equations:<disp-formula id="e10">
<mml:math id="m66">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2a;</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2a;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="e11">
<mml:math id="m67">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2a;</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>Here, <inline-formula id="inf50">
<mml:math id="m68">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the hidden state at time <inline-formula id="inf51">
<mml:math id="m69">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf52">
<mml:math id="m70">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the input at time <inline-formula id="inf53">
<mml:math id="m71">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf54">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the output at time <inline-formula id="inf55">
<mml:math id="m73">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf56">
<mml:math id="m74">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf57">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf58">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are weight matrices, <inline-formula id="inf59">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>h</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf60">
<mml:math id="m78">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are bias vectors, <inline-formula id="inf61">
<mml:math id="m79">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf62">
<mml:math id="m80">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are activation functions (typically sigmoid or hyperbolic tangent for <inline-formula id="inf63">
<mml:math id="m81">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and softmax for <inline-formula id="inf64">
<mml:math id="m82">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>). The hidden state <inline-formula id="inf65">
<mml:math id="m83">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> captures information from previous time steps, allowing RNNs to model temporal dependencies in sequential data.</p>
<p>
<xref ref-type="bibr" rid="B45">Xu et al. (2015)</xref> employed undirected graph recursive neural networks (UGRNN) to develop DL models for predicting DILI for drugs and small moleculars. Their DL-combined model outperformed ANN and DNN models, achieving an accuracy of 86.9% and an AUC of 0.955 when predicting the DILI of 198 drugs in the external validation set. The model also successfully identified important molecular substructures relevant to DILI, demonstrating the power of DL in this context. In another study <xref ref-type="bibr" rid="B32">Ruiz Puentes et al. (2021)</xref>, investigators proposed using PharmaNet, a machine learning method that employs RNNs, to search for novel pharmaceutical candidates. PharmaNet was applied to discover ligands for 102 cell receptors and achieved impressive performance with a 97.7% Receiver Operating Characteristic curve-Area Under the Curve (ROC-AUC).</p>
</sec>
<sec id="s3-6">
<title>3.6 Generative adversarial network</title>
<p>Generative Adversarial Network (GAN) is an advanced generative model composed of two neural networks: a generator and a discriminator. These networks are trained in opposition to each other. The generator&#x2019;s objective is to create synthetic data that is virtually indistinguishable from genuine data, while the discriminator&#x2019;s role is to differentiate between real and generated data. GANs operate through a minimax game where the generator and discriminator compete. As training progresses, the generator becomes increasingly skilled at generating realistic data, while the discriminator becomes better at distinguishing between real and fake data. This dynamic process drives the generator to produce high-quality synthetic data, establishing GANs as a foundational technology in a wide range of applications, such as picture production, style transfer, and data augmentation.</p>
<p>
<xref ref-type="bibr" rid="B6">Chen Z. et al. (2022)</xref> developed Tox-GAN, which employed deep GANs to generate fresh animal study results without the need for extra tests. They demonstrated its effectiveness by creating transcriptome profiles with remarkable similarity (0.997 &#xb1; 0.002 in intensity and 0.740 &#xb1; 0.082 in fold change) to real-world data obtained from rat liver toxicogenomic studies. In a related study, <xref ref-type="bibr" rid="B22">Li et al. (2023)</xref> introduced the TransOrGAN framework, which aims to map gene expression patterns across multiple rodent organs, sexes, and ages. TransOrGAN generated synthetic transcriptomic profiles with an average cosine similarity of 0.984 compared to their corresponding real profiles. This proof-of-concept study involved 288 samples from nine different organs, showcasing the potential of TransOrGAN to generate realistic transcriptomic data for various research applications.</p>
</sec>
<sec id="s3-7">
<title>3.7 Transformers</title>
<p>The field of Natural Language Processing (NLP) has undergone a transformative shift with the introduction of transformer-based models (<xref ref-type="bibr" rid="B20">Kang et al., 2020</xref>). These models have enabled the automatic analysis and comprehension of text data in scientific literature. In the domain of DILI studies, NLP models have proven to be valuable tools for extracting insights from textual sources.</p>
<p>
<xref ref-type="bibr" rid="B47">Zhan et al. (2022)</xref> developed NLP techniques specifically for biomedical texts, allowing the automated processing of 28,000 titles and abstracts retrieved from the PubMed database. By comparing five different text embedding techniques, they found that the model using term frequency-inverse document frequency and logistic regression performed best, with an accuracy of 0.957 on the validation set. <xref ref-type="bibr" rid="B44">Wu Z. et al. (2021)</xref> employed a NLP approach based on Bidirectional Encoder Representations from Transformers (BERT) to classify DILI and decipher the meanings of complex text in drug labeling documents. This AI-based model utilized BERT&#x2019;s power to enhance understanding of text data, particularly in the context of drug safety assessments.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Comparison of machine learning and deep learning for DILI prediction</title>
<p>Machine learning and deep learning techniques have emerged as powerful tools for developing models to predict DILI (<xref ref-type="table" rid="T1">Table 1</xref>). Machine learning uses algorithms to discover patterns and make predictions based on labeled data, whereas deep learning, a subset of machine learning, uses artificial neural networks to replicate the sophisticated functioning of the human brain. Machine learning algorithms analyze a set of predefined features to identify patterns associated with liver injury in the context of DILI prediction, whereas deep learning models can automatically extract intricate features from raw data, providing a more nuanced understanding of complex relationships. The major distinction between the two is in the level of abstraction and data representation (<xref ref-type="fig" rid="F1">Figure 1</xref>). Machine learning is based on feature engineering, in which the algorithm needs to select important features from high-dimensional dataset, whereas deep learning can develop hierarchical representations from raw data, possibly catching subtle nuances that typical machine learning algorithms may overlook. Both approaches provide important contributions to improving our ability to detect and alleviate DILI, giving essential insights for drug development and patient safety.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Comparative analysis of machine learning and deep learning for DILI prediction.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left">Machine learning</th>
<th align="left">Deep learning</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Definition</td>
<td align="left">Machine learning, as an application and subset of artificial intelligence, enables systems to autonomously learn from experiences and improve without manual intervention. Machine learning primarily generates outputs in the form of numerical values, such as score classifications</td>
<td align="left">In contrast, deep learning is essentially a subset of machine learning that intricately connects recurrent neural networks and artificial neural networks. Deep learning produces outputs ranging from free-form elements, such as unrestricted sound and text, to numerical values</td>
</tr>
<tr>
<td align="left">Data uses and presentation</td>
<td align="left">Machine learning utilizes unstructured data and information, resulting in distinct data representation scenarios. It involves handling thousands of diverse data points, contributing to its learning process</td>
<td align="left">Deep learning, leveraging artificial neural networks, introduces a different data representation paradigm, emphasizing neural networks. It is characterized by a vast amount of data, incorporates millions of data points, facilitating a more nuanced understanding of patterns and relationships. Deep learning models, especially deep neural networks, often require large amounts of labeled data for training</td>
</tr>
<tr>
<td align="left">Algorithm</td>
<td align="left">Machine learning employs a variety of automated algorithms, transforming them into numerous model functions capable of predicting future actions based on data patterns. Feature extraction is important for ML algorithms. Traditional machine learning models often have lower computational requirements compared to deep learning models</td>
<td align="left">In contrast, deep learning relies on neural networks to transport input through multiple processing levels, elucidating the characteristics and relationships within the current dataset. However, it is not necessary to extract or select important features for deep learning algorithms because it can be adjusted by weights in the hidden layers of the network</td>
</tr>
<tr>
<td align="left">Application of DILI prediction</td>
<td align="left">Machine learning stays competitive on identifying hidden patterns from a small amount of input dataset. It assists in various aspects of DILI prediction and management</td>
<td align="left">Deep learning excels in resolving complex machine learning challenges within a system, and its efficacy for DILI prediction will become more prominent with the progress of data accumulation in the field</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The general flowcharts for machine learning and deep learning techniques for developing models to predict DILI.</p>
</caption>
<graphic xlink:href="ftox-05-1340860-g001.tif"/>
</fig>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>Deep learning approaches have indeed shown significant promise in predicting DILI, leveraging the advantages of large datasets and the ability to capture intricate patterns. In the context of QSAR modeling, DL methods have often been reported to outperform conventional machine learning methods. However, it is essential to recognize that DL&#x2019;s superiority is not always guaranteed and can depend on the specific characteristics of the dataset and the problem. For instance, <xref ref-type="bibr" rid="B26">Liu et al. (2018)</xref> pointed out that global performance metrics, which typically show DNNs as superior to conventional machine learning, may not be appropriate for datasets with highly imbalanced sample distributions. They argued that for highly toxic chemicals, DNNs trained on all samples often perform worse than indicated by global performance metrics.</p>
<p>Imbalanced datasets can lead to misrepresentations of the actual performance, especially in cases where the minority class (highly toxic chemicals, in this example) is of particular interest. Similarly, <xref ref-type="bibr" rid="B33">Russo et al. (2018)</xref> compared DNN with conventional machine learning algorithms, including Naive Bayes, AdaBoost Decision Tree, Random Forest, and SVM, in the development of QSAR models for predicting endocrine disrupting endpoints using up to 7,500 compounds. Their results revealed that while DNNs may achieve higher accuracy on the training set, they did not consistently outperform classic machine learning methods in 5-fold cross-validation and predictions on external test sets. The performance of machine learning models can be influenced by various factors, including the nature of the data, the choice of molecular descriptors, and the specific problem being addressed.</p>
<p>Deep learning has specific characteristics for toxicity prediction. Scalability is a primary one, since DL models can handle vast amounts of data and understand nuanced correlations, enabling the discovery of small DILI risk variables that older approaches may overlook. Furthermore, by collecting latent characteristics across varied datasets, these models can accomplish impressive generalization, boosting the capacity to predict DILI across different chemicals and patient groups. However, interpretability is a key weakness of DL in this scenario. Because the models are intrinsically complex, deciphering the precise biological or chemical elements leading to DILI forecasts is difficult, limiting one&#x2019;s capacity to grasp the underlying processes. Additionally, DL also requires a large amount of high-quality data for training, and like machine learning, is also prone to overfitting when the training data is noisy or when the model is too complex.</p>
<p>Researchers and practitioners in this field must carefully consider these advantages and challenges when choosing and implementing DL approaches for toxicity prediction. Balancing the needs for accuracy and interpretability is crucial in improving our understanding and prediction of DILI and other toxicities.</p>
</sec>
</body>
<back>
<sec id="s6">
<title>Author contributions</title>
<p>FM: Data curation, Formal Analysis, Methodology, Writing&#x2013;original draft. MC: Conceptualization, Funding acquisition, Methodology, Supervision, Validation, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The authors declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<ack>
<p>The authors thank Joanne Berger, FDA Library, for manuscript editing assistance.</p>
</ack>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The authors declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Author disclaimer</title>
<p>This article reflects the views of the authors and does not necessarily reflect those of the U.S. Food and Drug Administration. Any mention of commercial products is for clarification only and is not intended as approval, endorsement, or recommendation.</p>
</sec>
<sec id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/ftox.2023.1340860/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/ftox.2023.1340860/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ai</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>QSAR modelling study of the bioconcentration factor and toxicity of organic compounds to aquatic organisms using machine learning and ensemble methods</article-title>. <source>Ecotoxicol. Environ. Saf.</source> <volume>179</volume>, <fpage>71</fpage>&#x2013;<lpage>78</lpage>. <pub-id pub-id-type="doi">10.1016/j.ecoenv.2019.04.035</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alempijevic</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zec</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Milosavljevic</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Drug-induced liver injury: do we know everything?</article-title> <source>World J. hepatology</source> <volume>9</volume> (<issue>10</issue>), <fpage>491</fpage>&#x2013;<lpage>502</lpage>. <pub-id pub-id-type="doi">10.4254/wjh.v9.i10.491</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andrade</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Chalasani</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Bj&#xf6;rnsson</surname>
<given-names>E. S.</given-names>
</name>
<name>
<surname>Suzuki</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kullak-Ublick</surname>
<given-names>G. A.</given-names>
</name>
<name>
<surname>Watkins</surname>
<given-names>P. B.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Drug-induced liver injury</article-title>. <source>Nat. Rev. Dis. Prim.</source> <volume>5</volume> (<issue>1</issue>), <fpage>58</fpage>. <pub-id pub-id-type="doi">10.1038/s41572-019-0105-0</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Borlak</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tong</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Predicting idiosyncratic drug-induced liver injury&#x2013;some recent advances</article-title>. <source>Expert Rev. Gastroenterology Hepatology</source> <volume>8</volume> (<issue>7</issue>), <fpage>721</fpage>&#x2013;<lpage>723</lpage>. <pub-id pub-id-type="doi">10.1586/17474124.2014.922871</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Roberts</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Tong</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Tox-GAN: an artificial intelligence approach alternative to animal studies&#x2014;a case study with toxicogenomics</article-title>. <source>Toxicol. Sci.</source> <volume>186</volume> (<issue>2</issue>), <fpage>242</fpage>&#x2013;<lpage>259</lpage>. <pub-id pub-id-type="doi">10.1093/toxsci/kfab157</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>ResNet18DNN: prediction approach of drug-induced liver injury by deep neural network with ResNet18</article-title>. <source>Briefings Bioinforma.</source> <volume>23</volume> (<issue>1</issue>), <fpage>bbab503</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbab503</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cruz&#x2010;Monteagudo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cordeiro</surname>
<given-names>M. N. D.</given-names>
</name>
<name>
<surname>Borges</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Computational chemistry approach for the early detection of drug&#x2010;induced idiosyncratic liver toxicity</article-title>. <source>J. Comput. Chem.</source> <volume>29</volume> (<issue>4</issue>), <fpage>533</fpage>&#x2013;<lpage>549</lpage>. <pub-id pub-id-type="doi">10.1002/jcc.20812</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Devarbhavi</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>An update on drug-induced liver injury</article-title>. <source>J. Clin. Exp. hepatology</source> <volume>2</volume> (<issue>3</issue>), <fpage>247</fpage>&#x2013;<lpage>259</lpage>. <pub-id pub-id-type="doi">10.1016/j.jceh.2012.05.002</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Gene expression data based deep learning model for accurate prediction of drug-induced liver injury in advance</article-title>. <source>J. Chem. Inf. Model.</source> <volume>59</volume> (<issue>7</issue>), <fpage>3240</fpage>&#x2013;<lpage>3250</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.9b00143</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gadaleta</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Manganelli</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Roncaglioni</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Toma</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Benfenati</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Mombelli</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>QSAR modeling of ToxCast assays relevant to the molecular initiating events of AOPs leading to hepatic steatosis</article-title>. <source>J. Chem. Inf. Model.</source> <volume>58</volume> (<issue>8</issue>), <fpage>1501</fpage>&#x2013;<lpage>1517</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.8b00297</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>George</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yuen</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hunt</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Suzuki</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Interplay of gender, age and drug properties on reporting frequency of drug-induced liver injury</article-title>. <source>Regul. Toxicol. Pharmacol.</source> <volume>94</volume>, <fpage>101</fpage>&#x2013;<lpage>107</lpage>. <pub-id pub-id-type="doi">10.1016/j.yrtph.2018.01.018</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Goh</surname>
<given-names>G. B.</given-names>
</name>
<name>
<surname>Siegel</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Vishnu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hodas</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Baker</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>How much chemistry does a deep neural network need to know to make accurate predictions?</article-title>,&#x201d; in <conf-name>2018 IEEE Winter Conference on Applications of Computer Vision (WACV)</conf-name> (<publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B13">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hastie</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tibshirani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Friedman</surname>
<given-names>J. H.</given-names>
</name>
<name>
<surname>Friedman</surname>
<given-names>J. H.</given-names>
</name>
</person-group> (<year>2009</year>). <source>The elements of statistical learning: data mining, inference, and prediction</source>. <publisher-name>Springer</publisher-name>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hemmerich</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Asilar</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ecker</surname>
<given-names>G. F.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>COVER: conformational oversampling as data augmentation for molecules</article-title>. <source>J. cheminformatics</source> <volume>12</volume> (<issue>1</issue>), <fpage>18</fpage>. <pub-id pub-id-type="doi">10.1186/s13321-020-00420-z</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hinton</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Osindero</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Teh</surname>
<given-names>Y.-W.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>A fast learning algorithm for deep belief nets</article-title>. <source>Neural Comput.</source> <volume>18</volume> (<issue>7</issue>), <fpage>1527</fpage>&#x2013;<lpage>1554</lpage>. <pub-id pub-id-type="doi">10.1162/neco.2006.18.7.1527</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Hwang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Jeon</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>A drug-induced liver injury prediction model using transcriptional response data with graph neural network</article-title>,&#x201d; in <conf-name>2020 IEEE International Conference on Big Data and Smart Computing (BigComp)</conf-name> (<publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Idakwo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Luttrell IV</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). <source>A review of feature reduction methods for QSAR-based toxicity prediction</source>. <publisher-name>Springer</publisher-name>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jain</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Siramshetty</surname>
<given-names>V. B.</given-names>
</name>
<name>
<surname>Alves</surname>
<given-names>V. M.</given-names>
</name>
<name>
<surname>Muratov</surname>
<given-names>E. N.</given-names>
</name>
<name>
<surname>Kleinstreuer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Tropsha</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Large-scale modeling of multispecies acute toxicity end points using consensus of multitask deep learning methods</article-title>. <source>J. Chem. Inf. Model.</source> <volume>61</volume> (<issue>2</issue>), <fpage>653</fpage>&#x2013;<lpage>663</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.0c01164</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kang</surname>
<given-names>M.-G.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>N. S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Predictive model for drug-induced liver injury using deep neural networks based on substructure space</article-title>. <source>Molecules</source> <volume>26</volume> (<issue>24</issue>), <fpage>7548</fpage>. <pub-id pub-id-type="doi">10.3390/molecules26247548</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>C.-W.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Natural language processing (NLP) in management research: a literature review</article-title>. <source>J. Manag. Anal.</source> <volume>7</volume> (<issue>2</issue>), <fpage>139</fpage>&#x2013;<lpage>172</lpage>. <pub-id pub-id-type="doi">10.1080/23270012.2020.1756939</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krizhevsky</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sutskever</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Imagenet classification with deep convolutional neural networks</article-title>, <source>Adv. neural Inf. Process. Syst.</source>, <volume>25</volume>. <pub-id pub-id-type="doi">10.1145/3065386</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Roberts</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Tong</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>TransOrGAN: an artificial intelligence mapping of rat transcriptomic profiles between organs, ages, and sexes</article-title>. <source>Chem. Res. Toxicol.</source> <volume>36</volume>, <fpage>916</fpage>&#x2013;<lpage>925</lpage>. <pub-id pub-id-type="doi">10.1021/acs.chemrestox.3c00037</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tong</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Roberts</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Thakkar</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020a</year>). <article-title>Deep learning on high-throughput transcriptomics to predict drug-induced liver injury</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>8</volume>, <fpage>562677</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2020.562677</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tong</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Roberts</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Thakkar</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>DeepDILI: deep learning-powered drug-induced liver injury prediction using model-level representation</article-title>. <source>Chem. Res. Toxicol.</source> <volume>34</volume> (<issue>2</issue>), <fpage>550</fpage>&#x2013;<lpage>565</lpage>. <pub-id pub-id-type="doi">10.1021/acs.chemrestox.0c00374</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lim</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shin</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Supervised chemical graph mining improves drug-induced liver injury prediction</article-title>. <source>iScience</source> <volume>26</volume> (<issue>1</issue>), <fpage>105677</fpage>. <pub-id pub-id-type="doi">10.1016/j.isci.2022.105677</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Madore</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Glover</surname>
<given-names>K. P.</given-names>
</name>
<name>
<surname>Feasel</surname>
<given-names>M. G.</given-names>
</name>
<name>
<surname>Wallqvist</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Assessing deep and shallow learning methods for quantitative prediction of acute chemical toxicity</article-title>. <source>Toxicol. Sci.</source> <volume>164</volume> (<issue>2</issue>), <fpage>512</fpage>&#x2013;<lpage>526</lpage>. <pub-id pub-id-type="doi">10.1093/toxsci/kfy111</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>An</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deep graph learning with property augmentation for predicting drug-induced liver injury</article-title>. <source>Chem. Res. Toxicol.</source> <volume>34</volume> (<issue>2</issue>), <fpage>495</fpage>&#x2013;<lpage>506</lpage>. <pub-id pub-id-type="doi">10.1021/acs.chemrestox.0c00322</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mora</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Marrero-Ponce</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Garc&#xed;a-Jacas</surname>
<given-names>C. R.</given-names>
</name>
<name>
<surname>Suarez Causado</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Ensemble models based on QuBiLS-MAS features and shallow learning for the prediction of drug-induced liver toxicity: improving deep learning and traditional approaches</article-title>. <source>Chem. Res. Toxicol.</source> <volume>33</volume> (<issue>7</issue>), <fpage>1855</fpage>&#x2013;<lpage>1873</lpage>. <pub-id pub-id-type="doi">10.1021/acs.chemrestox.0c00030</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nguyen-Vo</surname>
<given-names>T.-H.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Do</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>P. H.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>T.-N.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>B. P.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Predicting drug-induced liver injury using convolutional neural network and molecular fingerprint-embedded features</article-title>. <source>ACS omega</source> <volume>5</volume> (<issue>39</issue>), <fpage>25432</fpage>&#x2013;<lpage>25439</lpage>. <pub-id pub-id-type="doi">10.1021/acsomega.0c03866</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nassiri</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Alhambra</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Snoeys</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Van Goethem</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Irrechukwu</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>AI/ML models to predict the severity of drug-induced liver injury for small molecules</article-title>. <source>Chem. Res. Toxicol.</source> <volume>36</volume>, <fpage>1129</fpage>&#x2013;<lpage>1139</lpage>. <pub-id pub-id-type="doi">10.1021/acs.chemrestox.3c00098</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Rish</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2001</year>). &#x201c;<article-title>An empirical study of the naive Bayes classifier</article-title>,&#x201d; in <conf-name>IJCAI 2001 workshop on empirical methods in artificial intelligence</conf-name>.</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ruiz Puentes</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Valderrama</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Gonz&#xe1;lez</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Daza</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Mu&#xf1;oz-Camargo</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Cruz</surname>
<given-names>J. C.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>PharmaNet: pharmaceutical discovery with deep recurrent neural networks</article-title>. <source>Plos one</source> <volume>16</volume> (<issue>4</issue>), <fpage>e0241728</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0241728</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Russo</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Zorn</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Clark</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ekins</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Comparing multiple machine learning algorithms and metrics for estrogen receptor binding prediction</article-title>. <source>Mol. Pharm.</source> <volume>15</volume> (<issue>10</issue>), <fpage>4361</fpage>&#x2013;<lpage>4370</lpage>. <pub-id pub-id-type="doi">10.1021/acs.molpharmaceut.8b00546</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schmidhuber</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Deep learning in neural networks: an overview</article-title>. <source>Neural Netw.</source> <volume>61</volume>, <fpage>85</fpage>&#x2013;<lpage>117</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2014.09.003</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shin</surname>
<given-names>H. K.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>
<italic>In silico</italic> modeling-based new alternative methods to predict drug and herb-induced liver injury: a review</article-title>. <source>Food Chem. Toxicol.</source> <volume>179</volume>, <fpage>113948</fpage>. <pub-id pub-id-type="doi">10.1016/j.fct.2023.113948</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Szegedy</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sermanet</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Reed</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Anguelov</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). &#x201c;<article-title>Going deeper with convolutions</article-title>,&#x201d; in <conf-name>proceedings of the IEEE computer society conference on computer vision and pattern recognition</conf-name>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Discriminant models on mitochondrial toxicity improved by consensus modeling and resolving imbalance in training</article-title>. <source>Chemosphere</source> <volume>253</volume>, <fpage>126768</fpage>. <pub-id pub-id-type="doi">10.1016/j.chemosphere.2020.126768</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Schyman</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wallqvist</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Deep neural network models for predicting chemically induced liver toxicity endpoints from transcriptomic responses</article-title>. <source>Front. Pharmacol.</source> <volume>10</volume>, <fpage>42</fpage>. <pub-id pub-id-type="doi">10.3389/fphar.2019.00042</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weber</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gerbes</surname>
<given-names>A. L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Challenges and future of drug-induced liver injury research&#x2014;laboratory tests</article-title>. <source>Int. J. Mol. Sci.</source> <volume>23</volume> (<issue>11</issue>), <fpage>6049</fpage>. <pub-id pub-id-type="doi">10.3390/ijms23116049</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Williams</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Lazic</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Foster</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Semenova</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Morgan</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Predicting drug-induced liver injury with Bayesian machine learning</article-title>. <source>Chem. Res. Toxicol.</source> <volume>33</volume> (<issue>1</issue>), <fpage>239</fpage>&#x2013;<lpage>248</lpage>. <pub-id pub-id-type="doi">10.1021/acs.chemrestox.9b00264</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Auerbach</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>McEuen</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Integrating drug&#x2019;s mode of action into quantitative structure&#x2013;activity relationships for improved prediction of drug-induced liver injury</article-title>. <source>J. Chem. Inf. Model.</source> <volume>57</volume> (<issue>4</issue>), <fpage>1000</fpage>&#x2013;<lpage>1006</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.6b00719</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tong</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>BERT-based Natural Language Processing of drug labeling documents: a case study for classifying drug-induced liver injury risk</article-title>. <source>Front. Artif. Intell.</source> <volume>4</volume>, <fpage>729834</fpage>. <pub-id pub-id-type="doi">10.3389/frai.2021.729834</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Tong</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Borlak</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A systematic comparison of hepatobiliary adverse drug reactions in FDA and EMA drug labeling reveals discrepancies</article-title>. <source>Drug Discov. Today</source> <volume>27</volume> (<issue>1</issue>), <fpage>337</fpage>&#x2013;<lpage>346</lpage>. <pub-id pub-id-type="doi">10.1016/j.drudis.2021.09.009</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Leung</surname>
<given-names>E. L.-H.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Do we need different machine learning algorithms for QSAR modeling? A comprehensive assessment of 16 machine learning algorithms on 14 QSAR data sets</article-title>. <source>Briefings Bioinforma.</source> <volume>22</volume> (<issue>4</issue>), <fpage>bbaa321</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaa321</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pei</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lai</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Deep learning for drug-induced liver injury</article-title>. <source>J. Chem. Inf. Model.</source> <volume>55</volume> (<issue>10</issue>), <fpage>2085</fpage>&#x2013;<lpage>2093</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.5b00238</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yamashita</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Nishio</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Do</surname>
<given-names>R. K. G.</given-names>
</name>
<name>
<surname>Togashi</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Convolutional neural networks: an overview and application in radiology</article-title>. <source>Insights into imaging</source> <volume>9</volume>, <fpage>611</fpage>&#x2013;<lpage>629</lpage>. <pub-id pub-id-type="doi">10.1007/s13244-018-0639-9</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Gevaert</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Reliably filter drug-induced liver injury literature with Natural Language Processing and conformal prediction</article-title>. <source>IEEE J. Biomed. Health Inf.</source> <volume>26</volume> (<issue>10</issue>), <fpage>5033</fpage>&#x2013;<lpage>5041</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2022.3193365</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>S.-Q.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>H.-G.</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>W.-B.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Predicting drug-induced liver injury in human with Na&#xef;ve Bayes classifier approach</article-title>. <source>J. computer-aided Mol. Des.</source> <volume>30</volume>, <fpage>889</fpage>&#x2013;<lpage>898</lpage>. <pub-id pub-id-type="doi">10.1007/s10822-016-9972-6</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Graph neural networks: a review of methods and applications</article-title>. <source>AI open</source> <volume>1</volume>, <fpage>57</fpage>&#x2013;<lpage>81</lpage>. <pub-id pub-id-type="doi">10.1016/j.aiopen.2021.01.001</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Seo</surname>
<given-names>J.-E.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ashby</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ballard</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>The development of a database for herbal and dietary supplement induced liver toxicity</article-title>. <source>Int. J. Mol. Sci.</source> <volume>19</volume> (<issue>10</issue>), <fpage>2955</fpage>. <pub-id pub-id-type="doi">10.3390/ijms19102955</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>