<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2024.1401810</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Automatic text classification of drug-induced liver injury using document-term matrix and XGBoost</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Chen</surname> <given-names>Minjun</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/304873/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wu</surname> <given-names>Yue</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn0005"><sup>&#x2020;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/437432/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wingerd</surname> <given-names>Byron</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Liu</surname> <given-names>Zhichao</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Xu</surname> <given-names>Joshua</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1277105/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Thakkar</surname> <given-names>Shraddha</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/980190/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Pedersen</surname> <given-names>Thomas J.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2728597/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Donnelly</surname> <given-names>Tom</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Mann</surname> <given-names>Nicholas</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Tong</surname> <given-names>Weida</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/39650/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wolfinger</surname> <given-names>Russell D.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Bao</surname> <given-names>Wenjun</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/842887/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Division of Bioinformatics and Biostatistics, National Center for Toxicological Research, U.S. Food and Drug Administration</institution>, <addr-line>Jefferson, AR</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>JMP Statistical Discovery LLC</institution>, <addr-line>Cary, NC</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>Boehringer Ingelheim Pharmaceuticals, Inc.</institution>, <addr-line>Ridgefield, CT</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Pharmaceutical Sciences, University of Arkansas for Medical Sciences</institution>, <addr-line>Little Rock, AR</addr-line>, <country>United States</country></aff>
<aff id="aff5"><sup>5</sup><institution>Department of Mathematics, The University of North Carolina at Chapel Hill</institution>, <addr-line>Chapel Hill, NC</addr-line>, <country>United States</country></aff>
<author-notes>
<fn id="fn0006" fn-type="edited-by"><p>Edited by: Mohammad Akbari, Amirkabir University of Technology, Iran</p></fn>
<fn id="fn0007" fn-type="edited-by"><p>Reviewed by: Balu Bhasuran, University of California, San Francisco, United States</p>
<p>Hualou Liang, Drexel University, United States</p></fn>
<corresp id="c001">&#x002A;Correspondence: Wenjun Bao, <email>wenjun.bao@jmp.com</email></corresp>
<fn id="fn0005" fn-type="present-address"><p><sup>&#x2020;</sup>Present address: Yue Wu, Verve Therapeutics, Boston, MA, United States</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>06</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>7</volume>
<elocation-id>1401810</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>17</day>
<month>05</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2024 Chen, Wu, Wingerd, Liu, Xu, Thakkar, Pedersen, Donnelly, Mann, Tong, Wolfinger and Bao.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Chen, Wu, Wingerd, Liu, Xu, Thakkar, Pedersen, Donnelly, Mann, Tong, Wolfinger and Bao</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Regulatory agencies generate a vast amount of textual data in the review process. For example, drug labeling serves as a valuable resource for regulatory agencies, such as U.S. Food and Drug Administration (FDA) and Europe Medical Agency (EMA), to communicate drug safety and effectiveness information to healthcare professionals and patients. Drug labeling also serves as a resource for pharmacovigilance and drug safety research. Automated text classification would significantly improve the analysis of drug labeling documents and conserve reviewer resources.</p>
</sec>
<sec>
<title>Methods</title>
<p>We utilized artificial intelligence in this study to classify drug-induced liver injury (DILI)-related content from drug labeling documents based on FDA&#x2019;s DILIrank dataset. We employed text mining and XGBoost models and utilized the Preferred Terms of Medical queries for adverse event standards to simplify the elimination of common words and phrases while retaining medical standard terms for FDA and EMA drug label datasets. Then, we constructed a document term matrix using weights computed by Term Frequency-Inverse Document Frequency (TF-IDF) for each included word/term/token.</p>
</sec>
<sec>
<title>Results</title>
<p>The automatic text classification model exhibited robust performance in predicting DILI, achieving cross-validation AUC scores exceeding 0.90 for both drug labels from FDA and EMA and literature abstracts from the Critical Assessment of Massive Data Analysis (CAMDA).</p>
</sec>
<sec>
<title>Discussion</title>
<p>Moreover, the text mining and XGBoost functions demonstrated in this study can be applied to other text processing and classification tasks.</p>
</sec>
</abstract>
<kwd-group>
<kwd>anatomical therapeutic chemical classification (ATC)</kwd>
<kwd>Matthews correlation coefficient (MCC)</kwd>
<kwd>area under the curve</kwd>
<kwd>XGBoost</kwd>
<kwd>LightGBM</kwd>
<kwd>CatBoost</kwd>
<kwd>TF-IDF</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="3"/>
<equation-count count="5"/>
<ref-count count="29"/>
<page-count count="11"/>
<word-count count="7727"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Machine Learning and Artificial Intelligence</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<title>Introduction</title>
<p>Drug labeling documents are issued by regulatory agencies such as the United States Food and Drug Administration (FDA) and Europe Medical Agency (EMA) to communicate safety and efficacy information for approved drugs available to the public. Comprehensive information such as indications, contraindications, and warnings for adverse drug reactions (ADRs) is included in drug labeling as a reference for healthcare professionals and patients (<xref ref-type="bibr" rid="ref25">Watson and Barash, 2009</xref>; <xref ref-type="bibr" rid="ref17">McMahon and Preskorn, 2014</xref>). The FDA and EMA have approved over 130,000 labeling documents by 2022, creating a vast repository of regulatory text data for regulatory agency reviewers and scientific researchers (<xref ref-type="bibr" rid="ref14">Hoffman et al., 2016</xref>; <xref ref-type="bibr" rid="ref9">Fang et al., 2020</xref>; <xref ref-type="bibr" rid="ref27">Wu et al., 2021</xref>).</p>
<p>Drug-induced liver injury (DILI) is a common adverse drug reaction documented in drug labeling and has been recognized for its significant role in drug failure and withdrawal. <xref ref-type="bibr" rid="ref7">Chen et al. (2011</xref>, <xref ref-type="bibr" rid="ref6">2016)</xref> and <xref ref-type="bibr" rid="ref28">Wu et al. (2022)</xref> used text data from FDA drug labeling to annotate DILI risk in humans for 1,036 drugs. They manually searched the data using a set of predefined keywords and followed with manual curation to generate their annotations. Although manual curation ensures the specificity and usefulness of the information, it requires significant time and effort, including reading, understanding, and classifying the information for each drug, and is subject to the individual judgment and expertise of the reviewer. Drug labeling documents are updated regularly based on new findings from pharmacovigilance studies and case reports in literature (<xref ref-type="bibr" rid="ref13">Food and Drug Administration, 2015a</xref>,<xref ref-type="bibr" rid="ref12">b</xref>). Given the large volume of labeling documents, it is highly challenging to routinely reassess and update safety information manually. An automated, simplified text classification approach for processing text data in labeling and other documents would streamline the process and conserve reviewer resources.</p>
<p>Text mining is a valuable approach for gathering ADR information in drug labeling for comparative analysis during drug evaluation or scientific research (<xref ref-type="bibr" rid="ref9">Fang et al., 2020</xref>). To this end, the standard ADR terms from the Medical Dictionary for Regulatory Activities (MedDRA) and Systematized Nomenclature of Medicine (SNOMED) are commonly used to create a document-term matrix (DTM), which captures the frequency of terms in documents (<xref ref-type="bibr" rid="ref26">Wu et al., 2019</xref>; <xref ref-type="bibr" rid="ref8">Demner-Fushman et al., 2021</xref>; <xref ref-type="bibr" rid="ref28">Wu et al., 2022</xref>). The DTM can then be used as input for machine learning algorithms to classify drug labeling documents based on ADR risk and identify important features that contribute to the classification.</p>
<p>Tree ensemble models such as XGBoost (Extreme Gradient Boosting) (<xref ref-type="bibr" rid="ref5">Chen and Guestrin, 2016</xref>), LightGBM (light gradient-boosting machine) (<xref ref-type="bibr" rid="ref15">Ke et al., 2017</xref>) and CatBoost (Categorical Boosting) (<xref ref-type="bibr" rid="ref19">Prokhorenkova et al., 2018</xref>) are machine learning algorithms commonly used for classification and regression (<xref ref-type="bibr" rid="ref21">Shwartz-Ziv and Armon, 2022</xref>). XGBoost is a decision tree-based method that utilizes the principle of ensemble learning to make predictions. This process combines decisions from multiple models to produce a final prediction, in which the results of one model serve as input for the next, allowing the new model to correct errors made by the previous ones. XGBoost is widely used in a broad range of fields given its advantages in flexibility and interpretability (<xref ref-type="bibr" rid="ref29">Zhang et al., 2018</xref>; <xref ref-type="bibr" rid="ref22">Tahmassebi et al., 2019</xref>; <xref ref-type="bibr" rid="ref16">Li et al., 2020</xref>; <xref ref-type="bibr" rid="ref18">Pandi et al., 2020</xref>; <xref ref-type="bibr" rid="ref4">Chatterjee et al., 2021</xref>).</p>
<p>In this study, natural language processing was employed to create a DTM. The matrix was constructed using MedDRA or FDA Medical Query (FMQ) preferred terms (PTs) retrieved from the drug labeling documents and scientific abstracts. The XGBoost algorithm was then utilized to predict DILI from drug labeling documents and abstracts based on the DTM. The prediction model was evaluated through cross-validation, and the significance of standard terms was ranked and assessed for their relevance to DILI. The efficacy of this automatic text classification approach was verified through testing on FDA drug label data, EMA drug label data, and scientific literature data retrieved from the Critical Assessment of Massive Data Analysis (CAMDA). This approach was finally confirmed using a model generated by the FDA drug label dataset to predict DILI potential in the EMA drug label dataset.</p>
</sec>
<sec sec-type="materials|methods" id="sec2">
<title>Materials and methods</title>
<sec id="sec3">
<title>Datasets</title>
<p>The first dataset was generated from the FDA drug labeling documents of the DILI-rank dataset (<xref ref-type="bibr" rid="ref6">Chen et al., 2016</xref>). The drug labeling documents were retrieved from the FDALabel database<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref> (<xref ref-type="bibr" rid="ref9">Fang et al., 2020</xref>) with chemical structures. The drug labels were annotated for their DILI potential manually (<xref ref-type="bibr" rid="ref7">Chen et al., 2011</xref>). Text of &#x201C;Warnings and Precautions&#x201D; sections of drug labeling documents were extracted for this study. After removing those without labeling (e.g., withdrawn drugs) or chemical structures (e.g., biological products, mixtures), 678 unique prescription drugs approved by the FDA were identified. This data set contained 238 (35%) drug labeling documents that have DILI potential (defined as 1) and 440 (65%) that have no DILI potential (defined as 0). This dataset is hereafter referred to as the &#x201C;FDA dataset.&#x201D;</p>
<p>The second dataset was drug labeling documents of 277 unique prescription drugs from EMA that were retrieved from Electronic Medicines Compendium<xref ref-type="fn" rid="fn0002"><sup>2</sup></xref> (<xref ref-type="bibr" rid="ref2">Annex, 1999</xref>). The drug labels were annotated for their DILI potential manually as for FDA drug labels (<xref ref-type="bibr" rid="ref7">Chen et al., 2011</xref>). Text of &#x201C;Warnings and Precautions&#x201D; sections of EMA drug labeling documents were extracted for this study. This data set contained 132 (48%) drug labeling documents that have DILI potential (defined as 1) and 145 (52%) that have no DILI potential (defined as 0). This dataset is hereafter referred to as the &#x201C;EMA dataset.&#x201D;</p>
<p>The third dataset was retrieved from the CAMDA. This data set was originally published for the &#x201C;Literature AI for Drug Induced Liver Injury&#x201D; challenge in 2021.<xref ref-type="fn" rid="fn0003"><sup>3</sup></xref> This data set contained 12,187 abstracts of published papers from PubMed.<xref ref-type="fn" rid="fn0004"><sup>4</sup></xref> Each of the abstracts were annotated for DILI association by the experts in the NIH LiverTox. This data set contained 5,161 (42%) abstracts that were associated with drugs with DILI potential (defined as 1) and 7,026 (58%) abstracts that were associated with drugs without DILI potential (defined as 0). This dataset is hereafter referred to as the &#x201C;CAMDA dataset.&#x201D;</p>
</sec>
<sec id="sec4">
<title>Software package used</title>
<p>Natural language processing used Text Explorer (TE) function, and model comparisons were conducted via predictive model screening platform in the JMP Pro Statistical Discovery software package (JMP Pro v17, <ext-link xlink:href="https://protect.checkpoint.com/v2/___https://www.jmp.com/___.YzJ1OnNhc2luc3RpdHV0ZTpjOm86MzZjNjE2NDEzNTUyYTcwMjNjYTM4YmI2YjJkOWQ4NzY6Njo0NTQ4OjEzMjg3YjQ2MmZmZmEzZDJjYTI2NzNkMjg2ZDNhZmIwZjdmMTY2OTg1ZDI3MTBkZGZhMjljZDFjOGRlMWQ3NTY6cDpU" ext-link-type="uri">https://www.jmp.com/</ext-link>) in this study. The XGBoost add-in tool can be downloaded from <ext-link xlink:href="https://protect.checkpoint.com/v2/___https://community.jmp.com/t5/JMP-Add-Ins/XGBoost-Add-In-for-JMP-Pro/ta-p/319383___.YzJ1OnNhc2luc3RpdHV0ZTpjOm86MzZjNjE2NDEzNTUyYTcwMjNjYTM4YmI2YjJkOWQ4NzY6NjowOTAyOmQ5N2NmMmVhMjQwMTUxNmUwNjZmMDE3ZWVlOTM5ZDIxNTQ3ODQxMzQyNmU5ZjgzNGEyYjZhZWM0YjU3ZmU1NGY6cDpU" ext-link-type="uri">https://community.jmp.com/t5/JMP-Add-Ins/XGBoost-Add-In-for-JMP-Pro/ta-p/319383</ext-link> and installed into the software.</p>
</sec>
<sec id="sec5">
<title>Construction of document term matrix (DTM) by natural language processing</title>
<p>The FDA, EMA and CAMDA datasets were converted to a DTM using TE separately, the natural language processing function in the software the natural language processing here consisted of the curation of standardized terms, tokenization, and generation of a DTM. (<xref ref-type="fig" rid="fig1">Figure 1</xref>).</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption><p>Overview of text document analysis procedure. Natural Language Processing (NPL) generates term frequency matrices that are used to predict DILI indicator with cross validation. Optimized XGBoost models produce statistical performance metrics, important terms to DILI, and confidences about prediction.</p></caption>
<graphic xlink:href="frai-07-1401810-g001.tif"/>
</fig>
</sec>
<sec id="sec6">
<title>Curation of standardized terms</title>
<p>The medical-related terms were curated by the MedDRA or FMQ (<xref ref-type="bibr" rid="ref1">Andrade et al., 2019</xref>; <xref ref-type="bibr" rid="ref10">FDA, 2022</xref>) PT list. The terms and phrases in FDA, EMA or CAMDA that matched standardized terms were extracted from documents for further analysis. This is achieved by setting all the terms and phrases identified by TE as stop words first, and then adding back terms and phrases and stop word local exception that match MedDRA PT.</p>
</sec>
<sec id="sec7">
<title>Tokenization</title>
<p>Some related terms were combined into a single term by tokenization. For example, allergic hepatitis, autoimmune hepatitis, chronic hepatitis, chronic hepatitis b, chronic hepatitis c, hepatitis b surface antigen, and hepatitis c were all recoded to hepatitis; acute hepatic failure was recoded to hepatic failure. Some terms were processed by stemming; for example, aminotransferases, transaminases, liver enzymes and liver function tests were recoded to aminotransferase, transaminase, liver enzyme and liver function test, respectively.</p>
</sec>
<sec id="sec8">
<title>Term frequency-inverse document frequency (TF-IDF)</title>
<p>The DTM was constructed using weights computed by Term Frequency-Inverse Document Frequency (TF-IDF) for each included word/term/token. Specifically, this weighted TF-IDF is calculated by the <italic>log</italic> value of the frequency of a standardized term vs. the total number of documents in the corpus. This algorithm reduced the weights of the highly frequent terms and added weight to less frequent terms in each document.</p>
</sec>
<sec id="sec9">
<title>Classification modeling by XGBoost</title>
<p>XGBoost decision tree machine learning library uses DTM of standardized terms to predict DILI classification. The XGBoost modeling process with cross validation includes cross validation <italic>k</italic>-fold column creation, XGBoost model setup, and modeling optimization including adjusting iteration number and applying autotune with advanced options (<xref rid="SM1" ref-type="supplementary-material">Supplementary Figure S1</xref>).</p>
</sec>
<sec id="sec10">
<title>XGBoost model setup</title>
<p>The model was set up using DILI indicator as the Y response, DTM as the X regressor and defining the number of cross-validation folds and number of folds within each column for validation. Here, we used minimal term frequency as 1 for FDA and EMA dataset and as 10 for CAMDA data for generating DTM showed in <xref rid="SM1" ref-type="supplementary-material">Supplementary Figure S1</xref> to predict DILI indicator as Y responses and run 10 times of 5-fold validation.</p>
</sec>
<sec id="sec11">
<title>XGBoost model optimization</title>
<p>The different iterations were performed according to the iteration history for the default condition. The optimization of the model can also be achieved by using autotune with advanced options (<xref rid="SM1" ref-type="supplementary-material">Supplementary Figure S1</xref>).</p>
</sec>
<sec id="sec12">
<title>Cross validation</title>
<p>To reduce the risk of overfitting, a cross validation approach was used to evaluate model performance. The data set was divided into 5 roughly equal folds, and each level as a hold-out set. A group of 10 sets of 5 folds were prepared such that each fold was geometrically orthogonal to each fold in the other sets. The folds were stratified by the DILI indicator variable, which assures that approximately the same proportion of DILI occur in each subset.</p>
</sec>
<sec id="sec13">
<title>Model autotune with advanced options</title>
<p>The XGBoost model has a set of built-in hyperparameters that come with default values and suggested ranges for autotune with basic and advanced options (<xref rid="SM1" ref-type="supplementary-material">Supplementary Figure S1</xref>). The value and range of the hyperparameters can be modified.</p>
<p>Autotune offers a list of hyperparameters with default values and range, including max_depth for 3 (1&#x2013;7), subsample for 1 (0.5&#x2013;1), colsample by tree for 1 (0.5&#x2013;1), min child weight for 1 (1&#x2013;3), alpha for 0 (0&#x2013;0.5), lambda for 1 (0&#x2013;2), learning rate for 0.1 (0.05&#x2013;0.2), and iterations for 30 (20&#x2013;50) with number of design points as 10 and number of inner folds as 2. The more advanced options are also available for change. Those hyperparameters can be modified to control or prevent over-fitting and make models reasonably conservative, as overfitting is the most common issue in the machine learning analysis (<xref ref-type="bibr" rid="ref22">Tahmassebi et al., 2019</xref>; <xref ref-type="bibr" rid="ref16">Li et al., 2020</xref>; <xref ref-type="bibr" rid="ref18">Pandi et al., 2020</xref>; <xref ref-type="bibr" rid="ref4">Chatterjee et al., 2021</xref>).</p>
</sec>
<sec id="sec14">
<title>Statistical performance metrics</title>
<p>Six statistical metrics were used to assess XGBoost performance; these are Accuracy (ACC), area under the receiver operating characteristic curve (AUC), Matthews correlation coefficient (MCC), sensitivity, specificity and precision for training and validation sets for DILI indicator as binary Y response.</p>
<disp-formula id="E1"><mml:math id="M1"><mml:mi>A</mml:mi><mml:mi>C</mml:mi><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mfenced open="(" close=")"><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfenced><mml:mo stretchy="true">/</mml:mo><mml:mfenced open="(" close=")"><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfenced></mml:math></disp-formula>
<disp-formula id="E2"><mml:math id="M2"><mml:mi>M</mml:mi><mml:mi>C</mml:mi><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mfenced open="(" close=")"><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x2217;</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x2217;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfenced><mml:msqrt><mml:mrow><mml:mfenced open="(" close=")"><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfenced><mml:mo>&#x2217;</mml:mo><mml:mfenced open="(" close=")"><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfenced><mml:mo>&#x2217;</mml:mo><mml:mfenced open="(" close=")"><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfenced><mml:mo>&#x2217;</mml:mo><mml:mfenced open="(" close=")"><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfenced></mml:mrow></mml:msqrt></mml:mfrac></mml:math></disp-formula>
<disp-formula id="E3"><mml:math id="M3"><mml:mtext mathvariant="italic">Sensitivity</mml:mtext><mml:mo>=</mml:mo><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo stretchy="true">/</mml:mo><mml:mfenced open="(" close=")"><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfenced></mml:math></disp-formula>
<disp-formula id="E4"><mml:math id="M4"><mml:mtext mathvariant="italic">Specificity</mml:mtext><mml:mo>=</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo stretchy="true">/</mml:mo><mml:mfenced open="(" close=")"><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfenced></mml:math></disp-formula>
<disp-formula id="E5"><mml:math id="M5"><mml:mtext mathvariant="italic">Precision</mml:mtext><mml:mo>=</mml:mo><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo stretchy="true">/</mml:mo><mml:mfenced open="(" close=")"><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfenced></mml:math></disp-formula>
<p>Here, TP&#x2009;=&#x2009;True Positives; FN&#x2009;=&#x2009;False Negatives; TN&#x2009;=&#x2009;True Negatives; FP&#x2009;=&#x2009;False Positives.</p>
<p>AUC is the area under the ROC (Receiver Operating Characteristic) curve. The ROC curve plots sensitivity (true positive rate) on the Y axis vs. specificity (true negative rate) on the X axis. Sensitivity shows how well the model detects positives, that is, the ratio of true positives to true positives plus false negatives. Specificity defines how well the model avoids false alarms, which is ratio of true negatives to true negatives plus false positives. The ROC curve is constructed by plotting sensitivity versus specificity over a range of cutoff values applied to predicted probabilities. AUC can be interpreted as a measure of sorting efficiency. An AUC of 1.0 indicates perfect sorting and a value of 0.5 indicates no predictive performance.</p>
</sec>
<sec id="sec15">
<title>Predicting new documents</title>
<p>The prediction formula of the selected model for existing documents can be saved and applied to new documents to generate prediction probability for DILI potential associated with drugs. First, the document term frequency (DTF) for new documents can be created with the same text mining process as described above. Second, concatenate new documents with the DTF to the existing documents. Third, apply the saved prediction formula to the new documents. The prediction probability for DILI potential associated with the drug for each new document will be generated.</p>
</sec>
<sec id="sec16">
<title>Comparing XGBoost with other predictive models</title>
<p>Predictive model screening was employed to compare multiple predictive models. The nested cross validation was used with k as 10 and L as 5 to match the validation in XGBoost.</p>
</sec>
</sec>
<sec sec-type="results" id="sec17">
<title>Results</title>
<sec id="sec18">
<title>Data preprocessing</title>
<p>The text sections of &#x201C;warning and precaution&#x201D; in FDA drug labeling documents for 678 drugs were imported into the TE platform, yielding 14,600 terms and phrases out of 25,916 unique PTs in MedDRA 26.0, with the most frequent terms being common words such as &#x2018;patients&#x2019;, &#x2018;may&#x2019;, &#x2018;treatment&#x2019; (<xref ref-type="fig" rid="fig2">Figure 2</xref>). These common words were not medically related and should be removed. Instead of utilizing the common language processing techniques such as tokenizing, phrasing, terming, and stemming or manually defining the terms and phrases, we took advantage of the known MedDRA standard terms to narrow to medically related terms in three steps (<xref ref-type="fig" rid="fig2">Figure 2</xref>). First, all the terms and phrases (14,600) were removed by adding them as stop words in Text Explorer. Second, terms and phrases were matched with MedDRA PTs by using the software&#x2019;s Manage Phrase and Manage Stop Word functions, resulting in 1443 medical related terms and phrases of standard terms. Finally, each term and phrase were assigned values for each drug document, and a document term matrix weighted with TF-IDF was generated for the use in XGBoost modeling. The same process was applied to EMA and CAMDA datasets.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption><p>Flowchart to generate a document term matrix from FDA drug label documents. The 678 drug labels have 14,600 terms initially, which are reduced to 1,443 terms that match the MedDRA preferred terms. Word clouds are colored by DILI indicator: red for 1 and blue for 0. The EMA and CAMDA data used the same approach.</p></caption>
<graphic xlink:href="frai-07-1401810-g002.tif"/>
</fig>
</sec>
<sec id="sec19">
<title>XGBoost modeling for DILI classification</title>
<p>The parameters for XGBoost modeling were optimized using cross-validation. We explored the effects of term frequency and validation first for the FDA dataset. Using a term frequency of 1, 2 or 30 for drug labeling data yielded no significantly different results. The 10-fold validation resulted in a significantly higher AUC with a smaller variation compared to either 3 or 5 validation folds as measured by ANOVA (data not shown). Therefore, we used a term frequency of 1 (all the terms) for FDA and EMA datasets and a term frequency of 10 for CAMDA dataset to save the model calculation time. The 10-fold validation method for optimization conditions was run for all three datasets. The results include statistical performance metrics, a ranking of importance of terms, and a prediction formula for new data.</p>
<p>The XGBoost generates an iteration history that can assist researchers in choosing the proper number of iterations. For the FDA dataset shown in the plot and table in <xref ref-type="fig" rid="fig3">Figure 3</xref>, the default condition (<xref rid="SM1" ref-type="supplementary-material">Supplementary Figure S1C</xref>) with different iterations shows that validation curves (solid blue line) rapidly decline and then level off between 10 to 20 (<xref ref-type="fig" rid="fig3">Figure 3</xref>). Statistical metrics such as ACC, AUC and MCC for validation are very similar between 10 to 20 iterations with the peak statistical metrics at iteration 10. After 20 iterations, the validation curve started to rise, and the statistical metrics declined. When we applied autotune with default options (<xref rid="SM1" ref-type="supplementary-material">Supplementary Figure S1D</xref>), the validation curve (solid yellow line) leveled off with a different slope. The autotune generates the optimized condition (<xref ref-type="fig" rid="fig3">Figure 3</xref>) with iteration at 35 and almost the same statistical metrics as the default peak condition at iteration 10. The EMA dataset had very similar iteration history curve with peak range in 10&#x2013;20. The autotunes for FDA and EMA datasets have similar iterations at35 and 36. The CAMDA dataset needed 300 iterations to reach the optimized condition with the default condition and 638 iterations for autotune (<xref rid="SM2" ref-type="supplementary-material">Supplementary Figure S2</xref>).</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption><p>Using iteration history curves to optimize XGBoost performance for FDA drug label dataset. The best models lie in the dashed rectangular box of ranges of the validation curves for default parameters (solid blue) or autotune (solid brown). The dashed lines are for the corresponding training curves.</p></caption>
<graphic xlink:href="frai-07-1401810-g003.tif"/>
</fig>
</sec>
<sec id="sec20">
<title>Model performance evaluation</title>
<p>The statistical metrics such as ACC, AUC, MCC and root mean square error (RMSE) are listed in the <xref ref-type="table" rid="tab1">Table 1</xref> for training and validation of three datasets. Three other core concepts that are often used to evaluate the accuracy of models include sensitivity, specificity, and precision from different angles, which are also listed in <xref ref-type="table" rid="tab1">Table 1</xref>. The results in <xref ref-type="table" rid="tab1">Table 1</xref> were generated by 10 iterations for the FDA and EMA datasets and 300 iterations for CAMDA that are optimized default condition, plus the results were generated by autotune. The differences of the statistical metrics between defaults and autotunes for each dataset are ranged from 0 to 0.036, mostly less than 0.01. The differences of the three core concepts between defaults and autotunes for each dataset are ranged from 0.002 to 0.099, mostly less than 0.05.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption><p>Statistical metrics include ACC, AUC, MCC, RMSE, Precision, Sensitivity and Specificity for FDA, EMA and CAMDA datasets.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top" colspan="2">Dataset</th>
<th align="left" valign="top">Parameters</th>
<th align="center" valign="top">ACC</th>
<th align="center" valign="top">AUC</th>
<th align="center" valign="top">MCC</th>
<th align="center" valign="top">RMSE</th>
<th align="center" valign="top">Precision</th>
<th align="center" valign="top">Sensitivity</th>
<th align="center" valign="top">Specificity</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top" rowspan="4">FDA Drug Label</td>
<td align="left" valign="top" rowspan="2">Training</td>
<td align="left" valign="top">Default</td>
<td align="center" valign="top">0.913</td>
<td align="center" valign="top">0.977</td>
<td align="center" valign="top">0.812</td>
<td align="center" valign="top">0.256</td>
<td align="center" valign="top">0.979</td>
<td align="center" valign="top">0.769</td>
<td align="center" valign="top">0.991</td>
</tr>
<tr>
<td align="left" valign="top">Autotune</td>
<td align="center" valign="top">0.931</td>
<td align="center" valign="top">0.983</td>
<td align="center" valign="top">0.848</td>
<td align="center" valign="top">0.236</td>
<td align="center" valign="top">0.970</td>
<td align="center" valign="top">0.828</td>
<td align="center" valign="top">0.986</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Validation</td>
<td align="left" valign="top">Default</td>
<td align="center" valign="top">0.842</td>
<td align="center" valign="top">0.881</td>
<td align="center" valign="top">0.646</td>
<td align="center" valign="top">0.348</td>
<td align="center" valign="top">0.862</td>
<td align="center" valign="top">0.656</td>
<td align="center" valign="top">0.943</td>
</tr>
<tr>
<td align="left" valign="top">Autotune</td>
<td align="center" valign="top">0.839</td>
<td align="center" valign="top">0.886</td>
<td align="center" valign="top">0.639</td>
<td align="center" valign="top">0.344</td>
<td align="center" valign="top">0.856</td>
<td align="center" valign="top">0.651</td>
<td align="center" valign="top">0.941</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="4">EMA Drug Label</td>
<td align="left" valign="top" rowspan="2">Training</td>
<td align="left" valign="top">Default</td>
<td align="center" valign="top">0.971</td>
<td align="center" valign="top">0.996</td>
<td align="center" valign="top">0.942</td>
<td align="center" valign="top">0.177</td>
<td align="center" valign="top">0.971</td>
<td align="center" valign="top">0.955</td>
<td align="center" valign="top">0.986</td>
</tr>
<tr>
<td align="left" valign="top">Autotune</td>
<td align="center" valign="top">0.975</td>
<td align="center" valign="top">0.997</td>
<td align="center" valign="top">0.950</td>
<td align="center" valign="top">0.165</td>
<td align="center" valign="top">0.934</td>
<td align="center" valign="top">0.8560</td>
<td align="center" valign="top">0.945</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Validation</td>
<td align="left" valign="top">Default</td>
<td align="center" valign="top">0.903</td>
<td align="center" valign="top">0.958</td>
<td align="center" valign="top">0.807</td>
<td align="center" valign="top">0.280</td>
<td align="center" valign="top">0.985</td>
<td align="center" valign="top">0.962</td>
<td align="center" valign="top">0.986</td>
</tr>
<tr>
<td align="left" valign="top">Autotune</td>
<td align="center" valign="top">0.903</td>
<td align="center" valign="top">0.957</td>
<td align="center" valign="top">0.805</td>
<td align="center" valign="top">0.281</td>
<td align="center" valign="top">0.920</td>
<td align="center" valign="top">0.871</td>
<td align="center" valign="top">0.931</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="4">CAMDA Abstract</td>
<td align="left" valign="top" rowspan="2">Training</td>
<td align="left" valign="top">Default</td>
<td align="center" valign="top">0.883</td>
<td align="center" valign="top">0.946</td>
<td align="center" valign="top">0.762</td>
<td align="center" valign="top">0.298</td>
<td align="center" valign="top">0.931</td>
<td align="center" valign="top">0.780</td>
<td align="center" valign="top">0.958</td>
</tr>
<tr>
<td align="left" valign="top">Autotune</td>
<td align="center" valign="top">0.864</td>
<td align="center" valign="top">0.927</td>
<td align="center" valign="top">0.722</td>
<td align="center" valign="top">0.338</td>
<td align="center" valign="top">0.905</td>
<td align="center" valign="top">0.757</td>
<td align="center" valign="top">0.942</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Validation</td>
<td align="left" valign="top">Default</td>
<td align="center" valign="top">0.844</td>
<td align="center" valign="top">0.904</td>
<td align="center" valign="top">0.680</td>
<td align="center" valign="top">0.319</td>
<td align="center" valign="top">0.881</td>
<td align="center" valign="top">0.729</td>
<td align="center" valign="top">0.928</td>
</tr>
<tr>
<td align="left" valign="top">Autotune</td>
<td align="center" valign="top">0.837</td>
<td align="center" valign="top">0.897</td>
<td align="center" valign="top">0.666</td>
<td align="center" valign="top">0.344</td>
<td align="center" valign="top">0.871</td>
<td align="center" valign="top">0.722</td>
<td align="center" valign="top">0.921</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The drug label data has default parameters with 10 iterations and autotuned parameters with 35 iterations. The CAMDA data has default parameters with 300 iterations at 300 and autotuned parameters with 638 iterations.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="sec21">
<title>Ranking of term importance</title>
<p>The relative importance of the standardized terms is ranked by three related measurements: splits, gain, and cover. Splits represents the number of times that variable is used to split a branch in a tree. Gain is the average improvement in objective function for splits involving that variable, and Cover is the amount of data covered by splits involving that variable. Here, the gain is used to rank the importance of the selected terms as shown in <xref ref-type="table" rid="tab2">Table 2</xref>. The top five terms of MedDRA are liver damage related keywords classified in <xref ref-type="bibr" rid="ref6">Chen et al. (2016)</xref>, including hepatotoxicity (#1), hepatitis (#2), hepatic failure (#3), liver injury (#4) and jaundice (#5). Another group are immune-related terms, such as toxic epidermal necrolysis (#7), thrombocytopenia (#8), and eosinophilia (#9). Blood urea (#6) and oliguria (#10) that could be related to kidney, are also in the top 10 list.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption><p>Summary and list of top 10 features by gain for FDA drug label data according to MedDRA 26 and FMQ 2.1.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Feature</th>
<th align="center" valign="top">MedDRA26 Gain Rank</th>
<th align="center" valign="top">FMQ 2.1 Gain Rank</th>
<th align="left" valign="top">Terms</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="bottom">Hepatotoxicity</td>
<td align="center" valign="bottom">1</td>
<td align="center" valign="bottom">1</td>
<td align="left" valign="bottom">DILI Keyword</td>
</tr>
<tr>
<td align="left" valign="bottom">Hepatitis</td>
<td align="center" valign="bottom">2</td>
<td align="center" valign="bottom">2</td>
<td align="left" valign="bottom">DILI Keyword</td>
</tr>
<tr>
<td align="left" valign="bottom">Hepatic failure</td>
<td align="center" valign="bottom">3</td>
<td align="center" valign="bottom">3</td>
<td align="left" valign="bottom">DILI Keyword</td>
</tr>
<tr>
<td align="left" valign="bottom">Liver injury</td>
<td align="center" valign="bottom">4</td>
<td align="center" valign="bottom">4</td>
<td align="left" valign="bottom">DILI Keyword</td>
</tr>
<tr>
<td align="left" valign="bottom">Jaundice</td>
<td align="center" valign="bottom">5</td>
<td align="center" valign="bottom">5</td>
<td align="left" valign="bottom">DILI Keyword</td>
</tr>
<tr>
<td align="left" valign="bottom">Blood urea</td>
<td align="center" valign="bottom">6</td>
<td/>
<td align="left" valign="bottom">Renal Dysfunction</td>
</tr>
<tr>
<td align="left" valign="bottom">Bone marrow depression</td>
<td/>
<td align="center" valign="bottom">6</td>
<td align="left" valign="bottom">Immune System</td>
</tr>
<tr>
<td align="left" valign="bottom">Toxic epidermal necrolysis</td>
<td align="center" valign="bottom">7</td>
<td align="center" valign="bottom">8</td>
<td align="left" valign="bottom">Immune System</td>
</tr>
<tr>
<td align="left" valign="bottom">Thrombocytopenia</td>
<td align="center" valign="bottom">8</td>
<td/>
<td align="left" valign="bottom">Immune System</td>
</tr>
<tr>
<td align="left" valign="bottom">Carcinoma</td>
<td/>
<td align="center" valign="bottom">9</td>
<td align="left" valign="bottom">Cancer</td>
</tr>
<tr>
<td align="left" valign="bottom">Eosinophilia</td>
<td align="center" valign="bottom">9</td>
<td align="center" valign="bottom">10</td>
<td align="left" valign="bottom">Immune System</td>
</tr>
<tr>
<td align="left" valign="bottom">Oliguria</td>
<td align="center" valign="bottom">10</td>
<td align="center" valign="bottom">7</td>
<td align="left" valign="bottom">Renal Dysfunction</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The top five features are DILI keywords. Other five features are related to immune system or renal dysfunction.</p>
</table-wrap-foot>
</table-wrap>
<p>We also used FMQ to replace MedDRA as the standard terminology for the construction of the DTM and performed XGBoost with default condition at 10 iterations. FMQ was recently released by the FDA to standardize MedDRA PT groups according to medical concepts such as combining &#x201C;initial insomnia,&#x201D; &#x201C;middle insomnia,&#x201D; &#x201C;early morning awakening,&#x201D; to &#x201C;insomnia&#x201D; (<xref ref-type="bibr" rid="ref10">FDA, 2022</xref>; <xref ref-type="bibr" rid="ref11">FDA, 2023</xref>). The FMQ focuses on safety signal detection in clinical trial datasets. The 10 most important FMQ terms are also listed in <xref ref-type="table" rid="tab2">Table 2</xref>. The top 5 terms were the DILI keywords with the same ranking as with MedDRA. The bone marrow depression (#6) and carcinoma (#9) are different from MedDRA.</p>
<p>The bar charts in <xref ref-type="fig" rid="fig4">Figure 4</xref> showed the relationship between term count and term importance, as measured by gain. The value of gain is on the left Y axis and PT count and DILI keyword (KW) count are on the right Y axis. Only nine out of 58 KWs matched MedDRA PTs. The top plot was sorted by top 40 gain and bottom plot was sorted by top 40 PTs counts. The pink line is used to indicate top 20 terms. The top plot showed that first five plus cholestasis at top 19 are KWs. The bottom plot showed that the top 20 PT were not related to DILI. The top 10 PTs with count between 1,334 to 416 is 5 to 1.5 times greater than the term count of hepatitis (268). Hepatitis has the highest count in the KWs but is ranked #21 in all the terms. That means the importance of terms selected by the XGBoost modeling was not solely based on term count.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption><p>Comparison of the selected preferred terms by term count and term importance, as measured by gain for FDA drug label dataset.</p></caption>
<graphic xlink:href="frai-07-1401810-g004.tif"/>
</fig>
</sec>
<sec id="sec22">
<title>Prediction of DILI for the EMA dataset</title>
<p>The prediction of the DILI indicator in the EMA dataset was based on the presence of terms and phrases that matched the MedDRA PT, where a term frequency of 1 was defined. XGBoost was used with the default conditions, varying the number of iterations, or using autotune (refer to <xref rid="SM5" ref-type="supplementary-material">Supplementary Table S1</xref> for details). The validation curve for the iteration history of the EMA data resembled that of the FDA dataset, leveling off within the range of 10 to 20 iterations (data not shown). The statistical metrics for iterations 10, 15, and 20 were all within a negligible difference of 0.005, similar to those seen for the FDA dataset. Refer to <xref ref-type="table" rid="tab1">Table 1</xref> for a comprehensive overview of the statistical metrics specifically for the EMA data at iteration 10 under default conditions.</p>
</sec>
<sec id="sec23">
<title>Prediction of DILI for the CAMDA dataset</title>
<p>The data set from CAMDA was processed to construct DTM and XGBoost modeling similar to that done for the previous drug label data.</p>
<p>The results from the CAMDA dataset showed a similar pattern to the drug label dataset (<xref rid="SM2" ref-type="supplementary-material">Supplementary Figure S2</xref>), with DILI relevant terms dominating the top 10 important terms selected by XGBoost gain rank. The processing of the CAMDA dataset took 16&#x2009;h to complete and the cross-validation results are shown in <xref ref-type="table" rid="tab1">Table 1</xref>. An ACC of 0.844 or 0.837, AUC of 0.904 or 0.897, and MCC of 0.680 or 0.666 for default condition with iteration at 300 or autotune with iteration at 638. Additionally, immune-related terms such as rash, and rheumatoid arthritis, were also found among the top 10 important terms. These results indicate that the XGBoost modeling was successful in identifying relevant terms for DILI prediction using the CAMDA dataset as well.</p>
</sec>
<sec id="sec24">
<title>Comparison of model predictions for the FDA and EMA datasets</title>
<p>Both the FDA dataset and EMA dataset serve as drug label datasets. Among the 1,443 terms and phrases from the FDA dataset and the 1,190 terms and phrases from the EMA dataset, there were 821 common terms and phrases that matched MedDRA PTs. We employed the FDA dataset model condition, using XGBoost with default parameters and 10 iterations, along with these terms and phrases, to generate prediction probabilities for each document from both FDA and EMA datasets. The prediction probabilities were rounded up to either 1 or 0. We evaluated their consistency by subtracting the rounded-up prediction probabilities from the DILI indicator, which has been defined by experts to assess the agreement between the model results and DILI indicators for each drug. A subtraction result with an absolute value of 0 indicates that both the model and indicator evaluate the drug&#x2019;s DILI potential similarly. On the other hand, a subtraction result with an absolute value of 1 suggests a different classification between the model and indicator regarding the drug&#x2019;s DILI potential. In the FDA dataset model, 16% or 45 out of 277 drugs were inconsistently classified (<xref ref-type="fig" rid="fig5">Figure 5</xref>, Left). Additionally, when comparing the model results for EMA dataset drugs with the EMA terms and phrases, using XGBoost default conditions and iterations, against the DILI indicators, we found that 10% or 27 out of 277 EMA drugs were inconsistently estimated (<xref ref-type="fig" rid="fig5">Figure 5</xref>, Middle). Lastly, there was a 12% inconsistency observed between the 1,190 EMA terms and phrases used to predict EMA data and the 821 common terms and phrases shared between the FDA and EMA datasets (<xref ref-type="fig" rid="fig5">Figure 5</xref>, Right).</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption><p>Comparison of consistency for XGBoost model results using default condition with 10 iteration and DILI indicator. The consistency between model prediction by EMA drug label or FDA drug label and DILI indicator and between EMA and FDA prediction.</p></caption>
<graphic xlink:href="frai-07-1401810-g005.tif"/>
</fig>
</sec>
<sec id="sec25">
<title>Consistency between experts&#x2019; annotation and XGBoost classification</title>
<p>We compared consistency between expert opinion (DILI Indicator) and XGBoost model classification for FDA drug labeling using default condition with 10 iterations. The consistency evaluation method was described above. The 572 out of 678 or 84% FDA drugs label dataset were consistent between expert classification (DILI Indicator) and XGBoost model result (<xref ref-type="fig" rid="fig6">Figure 6A</xref>). To investigate further, we divided the model output probability into 10 ranges between 0 and 1. The lowest range, 0.00 to 0.10 means XGBoost model classifies the possibility to be DILI potential is very low. The highest range, 0.90 to 1.00, means XGBoost model classifies the possibility to be DILI potential is very high. <xref ref-type="fig" rid="fig6">Figure 6B</xref> indicates that for the probability ranges from 0.00 to 0.20 and from 0.70 to 1.00, the consistency rates were above or close to 90%; for probability ranges from 0.30 to 0.60, the consistency rates were about 55%; and for probability ranges from 0.20 to 0.30 and from 0.60 to 0.70, the consistency rates were about 75%.</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption><p>Comparison of consistency between experts&#x2019; opinions and XGBoost model classification for FDA drug labels using default condition with 10 iterations. The plot A showed that FDA drug labels has 84%(Y) consistency between XGBoost model results and experts &#x2018;estimation. The plot B displays the percentage of consistency in different probability range by XGBoost model.</p></caption>
<graphic xlink:href="frai-07-1401810-g006.tif"/>
</fig>
</sec>
<sec id="sec26">
<title>Comparison of XGBoost and other predictive models</title>
<p>We employed a predictive model screening platform to compare multiple predictive models. The nested cross validation was used with k as 10 and L as 5 to match the validation in XGBoost. The validated AUG results are shown in <xref ref-type="table" rid="tab3">Table 3</xref> The validated AUC for XGBoost, boosted tree, bootstrap, decision tree, neural boosted and support vector machines (SVMs) comparison are 0.881, 0.873, 0.872, 0.780, 0.785 and 0.764, respectively.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption><p>Predictive model comparison for FDA drug label dataset.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th/>
<th align="center" valign="top">XGBoost</th>
<th align="center" valign="top">Boosted Tree</th>
<th align="center" valign="top">Bootstrap Forest</th>
<th align="center" valign="top">Decision Tree</th>
<th align="center" valign="top">Neural Boosted</th>
<th align="center" valign="top">SVM</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Validation AUC</td>
<td align="center" valign="top">0.881</td>
<td align="center" valign="top">0.873</td>
<td align="center" valign="top">0.872</td>
<td align="center" valign="top">0.780</td>
<td align="center" valign="top">0.785</td>
<td align="center" valign="top">0.764</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The validated AUC for XGBoost and five predictive models, includes boosted tree, bootstrap, decision tree, neural boosted and support vector machines (SVMs), from the predictive model screening platform were compared.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="sec27">
<title>Prediction errors analysis</title>
<p>Some drugs are found to have inconsistent DILI classifications between the model and expert reviewers. For example, Epirubicin was defined as DILI indicator of 1 by expert reviewers but predicted as 0 by the model with a high confidence (0.867). The text explorer using the selected KWs was run for this drug. Only two keywords, aspartate aminotransferase (AST) and hepatic impairment, showed up twice for Epirubicin in the word cloud (<xref rid="SM3" ref-type="supplementary-material">Supplementary Figure S3</xref> left). Upon reviewing the sentences including those keywords (<xref rid="SM3" ref-type="supplementary-material">Supplementary Figure S3</xref> right), it was found the sentences suggest precondition for hepatic impairment, not drug-induced results. Notably, human experts use the information from the multiple sections in drug label to determine DILI indicators, while our study only used the &#x201C;warning and precaution&#x201D; section in the drug label. This might be another reason for the inconsistencies between the model prediction and DILI indictors.</p>
</sec>
</sec>
<sec id="sec28">
<title>Discussion and conclusions</title>
<p>In this study, we utilize artificial intelligence tools to classify whether contents from drug labeling documents and scientific abstracts are DILI related. We combined text mining and XGBoost models, taking advantage of standard PTs to simplify the elimination of the common/stop words. XGBoost models demonstrate excellent performance in classifying DILI, achieving AUC scores above 0.88 in cross-validation for both drug labels and the CAMDA datasets. Our results show that the DILI-related terms are the most critical contributors to DILI risk classification. Term frequency is not a significant factor, as feature importance did not correlate with term frequency. An important feature of the modeling workflow was to substitute PTs of MedDRA frequency to generate the DTM, suggesting that the results were not dependent on a specific set of standardized terms or corpus terms. Adding chemical descriptors as predictors did not significantly improve model performance (results not shown). Therefore, DILI-related standard terms appear to be the key factors for classifying whether the input text documents are relevant to DILI.</p>
<p>The top 5 important terms (MedDRA or FMQ) to contribute to XGBoost prediction are pre-selected keywords for manually annotating DILI by reviewers (<xref ref-type="bibr" rid="ref7">Chen et al., 2011</xref>). Other top ranked terms, such as toxic epidermal necrolysis, thrombocytopenia, eosinophilia, depression are related to certain DILI mechanisms, as DILI can be linked to immune-mediated ADRs, such as drug reaction with eosinophilia and systemic symptoms, or cutaneous ADRs, including Stevens-Johnson syndrome and toxic epidermal necrolysis (<xref ref-type="bibr" rid="ref1">Andrade et al., 2019</xref>). This suggests that the automatic text classification model can capture the underlying mechanistic relationship between DILI and immune disorders/skin reactions. Furthermore, renal dysfunction is commonly associated with liver diseases, especially in case of direct involvement in multiorgan acute illness or secondary to advanced liver disease (<xref ref-type="bibr" rid="ref3">Betrosian et al., 2007</xref>). The model identifies several terms related to renal damage, such as blood urea and oliguria, as having high importance in model prediction, providing valuable mechanistic information for further investigation.</p>
<p>In terms of the document-term matrix, the FDA dataset uses 1,443 features and EMA used 1,190 features with a term frequency of 1, while the CAMDA dataset uses 740 features with a term frequency of 10. For the drug label datasets, a term frequency of 1 is selected to ensure no terms are excluded that could contribute to the prediction. However, for the CAMDA dataset, a term frequency of 10 is selected to save processing time as the CAMDA dataset is much larger with about 18 times more sample data.</p>
<p>Since both FDA and EMA datasets have the optimized model with the default condition with 10 iterations, using terms and phrases from FDA dataset to estimate the DILI prediction probability for each drug in the EMA dataset has 6% more inconsistency of DILI indicator in comparison with using terms and phrases from EMA dataset, with 84% consistency.</p>
<p>There are two types of classification errors in the model development: mistaking a drug that has DILI potential into non-DILI, and vice versa. From a clinical perspective, misclassifying a drug with DILI potential as non-DILI is a potentially more costly error as it could result in greater harm to patients. However, while misclassifying a non-DILI drug as having DILI potential group would likely cause less harm to patients, it would likely prove more costly for the pharmaceutical company. Like the concept of applicability domain (<xref ref-type="bibr" rid="ref23">Tong et al., 2004</xref>), a Profit Matrix can be used to evaluate prediction performance by assuming different misclassification costs for DILI and non-DILI. The values in a Profit Matrix can be adjusted before or after a model is established to increase confidence in the classification in either direction, as shown in <xref rid="SM4" ref-type="supplementary-material">Supplementary Figure S4</xref>. When the probability threshold is changed from 0.5 to 0.75, the misclassification rate of DILI to non-DILI changes from 62 to 94, and the misclassification of non-DILI to DILI changes from 24 to 15which means more cases are classified to non-DILI cases. When the probability threshold is changed from 0.5 to 0.25, the misclassification rate of DILI to non-DILI cases decreases from 62 to 40, and the misclassification of non-DILI to DILI cases increases from 24 to 49, which means more cases are classified to DILI cases.</p>
<p>Comparing multiple predictive models showed that XGBoost had the best validated AUC, followed closely by boosted trees. XGBoost and boosted trees are powerful ensemble methods that combine multiple decision trees to create accurate predictive models. Bootstrap is a resampling technique used for model validation and uncertainty estimation. Decision trees are intuitive and interpretable but may suffer from overfitting. Neural networks offer the ability to model complex relationships but require large amounts of data and computational resources. SVMs provide robust classification but can be computationally expensive. The choice of predictive model depends on the specific problem, available data, interpretability requirements, and computational resources.</p>
<p>For drug labels we used a basic text mining bag-of-words approach. This method evaluates the content of the document and is blind to ordering, conditional statements, or sentiment. The language used for drug label warning and precautions is straightforward, making these documents ideal for this type of analysis. In contrast, publication abstracts are more complex and written by the authors with varying language skills and cultural backgrounds. Notably, our model is robust and demonstrates similar performance on both the CAMDA dataset and the drug labeling datasets. Other text mining methods that can handle more complicated grammar or negative sentences in natural language processing, such as the transformer neural networks of general domain BERT or specialized BERT (<xref ref-type="bibr" rid="ref20">Shi et al., 2023</xref>; <xref ref-type="bibr" rid="ref24">ValizadehAslani et al., 2023</xref>). The deep learning with Python codes such as BERT, require programming skills and a substantial amount of computing resources. The XGBoost can run much faster (a few minutes) using the document term matrix that were generated by Text Explorer in JMP Pro to predict DILI, while BERT needs much longer time (hours). While an optimal prediction strategy is likely to be an ensemble neural net and boosted trees (<xref ref-type="bibr" rid="ref21">Shwartz-Ziv and Armon, 2022</xref>), we did not explore this in the present study. We also acknowledge potential improvement in performance for drug labelling can be obtained using transformer BERT-style language models we are pursuing this in additional research.</p>
<p>Here, we developed an automatic text classification of DILI using DTM and XGBoost, and it was successfully applied to analyze drug labels from FDA and EMA and literature abstracts from CAMDA. The XGBoost with text-based tabular features approach demonstrated here can be applied to other text processing and classification tasks and offers a non-code solution for scientists and researchers to access AI and ML technologies for natural language processing.</p>
</sec>
<sec sec-type="data-availability" id="sec29">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="sec35">Supplementary material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="sec30">
<title>Author contributions</title>
<p>MC: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. YW: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. BW: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. ZL: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. JX: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. ST: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. TP: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. TD: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. NM: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. WT: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. RW: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. WB: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="sec31">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<ack>
<p>We greatly appreciated the support from the NCTR AI4Tox program, An FDA Artificial Intelligence (AI) Program for Toxicology (<ext-link xlink:href="https://www.fda.gov/about-fda/nctr-research-focus-areas/artificial-intelligence" ext-link-type="uri">https://www.fda.gov/about-fda/nctr-research-focus-areas/artificial-intelligence</ext-link>).</p>
</ack>
<sec sec-type="COI-statement" id="sec32">
<title>Conflict of interest</title>
<p>BW, TP, TD, RW, and WB were employed by JMP Statistical Discovery LLC. ZL was employed by Boehringer Ingelheim Pharmaceuticals, Inc.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec sec-type="disclaimer" id="sec33">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="disclaimer" id="sec34">
<title>Author disclaimer</title>
<p>This article reflects the views of the authors and does not necessarily reflect those of the U.S. Food and Drug Administration. Any mention of commercial products is for clarification only and is not intended as approval, endorsement, or recommendation.</p>
</sec>
<sec sec-type="supplementary-material" id="sec35">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/frai.2024.1401810/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/frai.2024.1401810/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Image_1.tif" id="SM1" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_2.tif" id="SM2" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_3.tif" id="SM3" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_4.tif" id="SM4" mimetype="image/tiff" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_1.pdf" id="SM5" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<title>Abbreviations</title>
<fn fn-type="abbr"><p>ACT, Anatomical Therapeutic Chemical Classification; KW, DILI keywords; PT, Preferred terms of MedDRA; DTM, document term matrixes; ACC, Accuracy; MCC, Matthews Correlation Coefficient; AUC, Area Under the Curve; XGBoost, Extreme Gradient Boosting; LightGBM, light gradient-boosting machine; CatBoost, Categorical Boosting; TF-IDF, Term Frequency-Inverse Document Frequency.</p></fn>
</fn-group>
<fn-group>
<fn id="fn0001"><p><sup>1</sup><ext-link xlink:href="https://nctr-crs.fda.gov/fdalabel/ui/search" ext-link-type="uri">https://nctr-crs.fda.gov/fdalabel/ui/search</ext-link></p></fn>
<fn id="fn0002"><p><sup>2</sup><ext-link xlink:href="https://www.medicines.org.uk" ext-link-type="uri">https://www.medicines.org.uk</ext-link></p></fn>
<fn id="fn0003"><p><sup>3</sup><ext-link xlink:href="http://www.camda.info/" ext-link-type="uri">http://www.camda.info/</ext-link></p></fn>
<fn id="fn0004"><p><sup>4</sup><ext-link xlink:href="https://pubmed.ncbi.nlm.nih.gov/" ext-link-type="uri">https://pubmed.ncbi.nlm.nih.gov/</ext-link></p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Andrade</surname> <given-names>R. J.</given-names></name> <name><surname>Chalasani</surname> <given-names>N.</given-names></name> <name><surname>Bj&#x00F6;rnsson</surname> <given-names>E. S.</given-names></name> <name><surname>Suzuki</surname> <given-names>A.</given-names></name> <name><surname>Kullak-Ublick</surname> <given-names>G. A.</given-names></name> <name><surname>Watkins</surname> <given-names>P. B.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Drug-induced liver injury</article-title>. <source>Nat. Rev. Dis. Prim.</source> <volume>5</volume>:<fpage>58</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41572-019-0105-0</pub-id></citation></ref>
<ref id="ref2"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Annex</surname> <given-names>I.</given-names></name></person-group> (<year>1999</year>). &#x201C;<article-title>Summary of product characteristics</article-title>&#x201D; in <source>Committee for Proprietary Medicinal Products</source> (<publisher-loc>London</publisher-loc>: <publisher-name>The European public assessment report (EPAR). Stocrin. The European Agency for the Evaluation of medicinal products</publisher-name>).</citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Betrosian</surname> <given-names>A. P.</given-names></name> <name><surname>Agarwal</surname> <given-names>B.</given-names></name> <name><surname>Douzinas</surname> <given-names>E. E.</given-names></name></person-group> (<year>2007</year>). <article-title>Acute renal dysfunction in liver diseases</article-title>. <source>World J. Gastroenterol.</source> <volume>13</volume>, <fpage>5552</fpage>&#x2013;<lpage>5559</lpage>. doi: <pub-id pub-id-type="doi">10.3748/wjg.v13.i42.5552</pub-id>, PMID: <pub-id pub-id-type="pmid">17948928</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chatterjee</surname> <given-names>S.</given-names></name> <name><surname>Goyal</surname> <given-names>D.</given-names></name> <name><surname>Prakash</surname> <given-names>A.</given-names></name> <name><surname>Sharma</surname> <given-names>J.</given-names></name></person-group> (<year>2021</year>). <article-title>Exploring healthcare/health-product ecommerce satisfaction: a text mining and machine learning application</article-title>. <source>J. Bus. Res.</source> <volume>131</volume>, <fpage>815</fpage>&#x2013;<lpage>825</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jbusres.2020.10.043</pub-id></citation></ref>
<ref id="ref5"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>T.</given-names></name> <name><surname>Guestrin</surname> <given-names>C.</given-names></name></person-group> <article-title>XGBoost: a scalable tree boosting system</article-title>. in <conf-name>Proceedings of the 22nd acm sigkdd international conference on knowledge discovery and data mining</conf-name>. (<year>2016</year>).</citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>M.</given-names></name> <name><surname>Suzuki</surname> <given-names>A.</given-names></name> <name><surname>Thakkar</surname> <given-names>S.</given-names></name> <name><surname>Yu</surname> <given-names>K.</given-names></name> <name><surname>Hu</surname> <given-names>C.</given-names></name> <name><surname>Tong</surname> <given-names>W.</given-names></name></person-group> (<year>2016</year>). <article-title>DILIrank: the largest reference drug list ranked by the risk for developing drug-induced liver injury in humans</article-title>. <source>Drug Discov. Today</source> <volume>21</volume>, <fpage>648</fpage>&#x2013;<lpage>653</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.drudis.2016.02.015</pub-id>, PMID: <pub-id pub-id-type="pmid">26948801</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>M.</given-names></name> <name><surname>Vijay</surname> <given-names>V.</given-names></name> <name><surname>Shi</surname> <given-names>Q.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Fang</surname> <given-names>H.</given-names></name> <name><surname>Tong</surname> <given-names>W.</given-names></name></person-group> (<year>2011</year>). <article-title>FDA-approved drug labeling for the study of drug-induced liver injury</article-title>. <source>Drug Discov. Today</source> <volume>16</volume>, <fpage>697</fpage>&#x2013;<lpage>703</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.drudis.2011.05.007</pub-id></citation></ref>
<ref id="ref8"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Demner-Fushman</surname> <given-names>D.</given-names></name> <name><surname>Elhadad</surname> <given-names>N.</given-names></name> <name><surname>Friedman</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). &#x201C;<article-title>Natural language processing for health-related texts</article-title>&#x201D; in <source>Biomedical informatics: Computer applications in health care and biomedicine</source>. eds. <person-group person-group-type="editor"><name><surname>Shortliffe</surname> <given-names>E. H.</given-names></name> <name><surname>Cimino</surname> <given-names>J. J.</given-names></name></person-group> (<publisher-name>Springer</publisher-name>), <fpage>241</fpage>&#x2013;<lpage>272</lpage>. Available at: <ext-link xlink:href="https://link.springer.com/chapter/10.1007/978-3-030-58721-5_8" ext-link-type="uri">https://link.springer.com/chapter/10.1007/978-3-030-58721-5_8</ext-link></citation></ref>
<ref id="ref9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fang</surname> <given-names>H.</given-names></name> <name><surname>Harris</surname> <given-names>S.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Thakkar</surname> <given-names>S.</given-names></name> <name><surname>Yang</surname> <given-names>J.</given-names></name> <name><surname>Ingle</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>FDALabel for drug repurposing studies and beyond</article-title>. <source>Nat. Biotechnol.</source> <volume>38</volume>, <fpage>1378</fpage>&#x2013;<lpage>1379</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41587-020-00751-0</pub-id>, PMID: <pub-id pub-id-type="pmid">33235392</pub-id></citation></ref>
<ref id="ref10"><citation citation-type="web"><person-group person-group-type="author"><collab id="coll1">FDA</collab></person-group>. FMQ 2.1 download (<year>2022</year>). Available at: <ext-link xlink:href="https://downloads.regulations.gov/FDA-2022-N-1961-0001/attachment_1.xlsm" ext-link-type="uri">https://downloads.regulations.gov/FDA-2022-N-1961-0001/attachment_1.xlsm</ext-link></citation></ref>
<ref id="ref11"><citation citation-type="web"><person-group person-group-type="author"><collab id="coll2">FDA</collab></person-group>. Biomedical Informatics and Regulatory Review Science (BIRRS). (<year>2023</year>). Available at: <ext-link xlink:href="https://www.fda.gov/about-fda/center-drug-evaluation-and-research-cder/biomedical-informatics-and-regulatory-review-science-birrs#:~:text=FDA%20Medical%20Queries%20(FMQ)%3A,clinical%20trial%20adverse%20event%20data" ext-link-type="uri">https://www.fda.gov/about-fda/center-drug-evaluation-and-research-cder/biomedical-informatics-and-regulatory-review-science-birrs#:~:text=FDA%20Medical%20Queries%20(FMQ)%3A,clinical%20trial%20adverse%20event%20data</ext-link></citation></ref>
<ref id="ref12"><citation citation-type="other"><person-group person-group-type="author"><collab id="coll3">Food and Drug Administration</collab></person-group>, <source>Warnings and precautions, contraindications, and boxed warning sections of labeling for human prescription drug and biological products&#x2013;content and format. 2011</source>. (<year>2015b</year>).</citation></ref>
<ref id="ref13"><citation citation-type="other"><person-group person-group-type="author"><collab id="coll4">Food and Drug Administration</collab></person-group>, <source>Adverse reactions section of labeling for human prescription drug and biological products&#x2014;Content and format</source>. (<year>2015a</year>).</citation></ref>
<ref id="ref14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hoffman</surname> <given-names>K. B.</given-names></name> <name><surname>Dimbil</surname> <given-names>M.</given-names></name> <name><surname>Tatonetti</surname> <given-names>N. P.</given-names></name> <name><surname>Kyle</surname> <given-names>R. F.</given-names></name></person-group> (<year>2016</year>). <article-title>A pharmacovigilance signaling system based on FDA regulatory action and post-marketing adverse event reports</article-title>. <source>Drug Saf.</source> <volume>39</volume>, <fpage>561</fpage>&#x2013;<lpage>575</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s40264-016-0409-x</pub-id>, PMID: <pub-id pub-id-type="pmid">26946292</pub-id></citation></ref>
<ref id="ref15"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Ke</surname> <given-names>G.</given-names></name> <name><surname>Meng</surname> <given-names>Q.</given-names></name> <name><surname>Finley</surname> <given-names>T.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Chen</surname> <given-names>W.</given-names></name> <name><surname>Ma</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2017</year>). &#x201C;<article-title>LightGBM: a highly efficient gradient boosting decision tree</article-title>&#x201D; in <source>Advances in neural information processing systems</source>. eds. <person-group person-group-type="editor"><name><surname>Guyon</surname> <given-names>I.</given-names></name> <name><surname>Von Luxburg</surname> <given-names>U.</given-names></name> <name><surname>Bengio</surname> <given-names>S.</given-names></name> <name><surname>Wallach</surname> <given-names>H.</given-names></name> <name><surname>Fergus</surname> <given-names>R.</given-names></name> <name><surname>Vishwanathan</surname> <given-names>S.</given-names></name> <etal/></person-group>. <volume>30</volume>.</citation></ref>
<ref id="ref16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>W. T.</given-names></name> <name><surname>Ma</surname> <given-names>J.</given-names></name> <name><surname>Shende</surname> <given-names>N.</given-names></name> <name><surname>Castaneda</surname> <given-names>G.</given-names></name> <name><surname>Chakladar</surname> <given-names>J.</given-names></name> <name><surname>Tsai</surname> <given-names>J. C.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Using machine learning of clinical data to diagnose COVID-19: a systematic review and meta-analysis</article-title>. <source>BMC Med. Inform. Decis. Mak.</source> <volume>20</volume>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s12911-020-01266-z</pub-id></citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>McMahon</surname> <given-names>D.</given-names></name> <name><surname>Preskorn</surname> <given-names>S. H.</given-names></name></person-group> (<year>2014</year>). <article-title>The package insert: who writes it and why, what are its implications, and how well does medical school explain it?</article-title> <source>J. Psychiatr. Pract.</source> <volume>20</volume>, <fpage>284</fpage>&#x2013;<lpage>290</lpage>. doi: <pub-id pub-id-type="doi">10.1097/01.pra.0000452565.83039.20</pub-id>, PMID: <pub-id pub-id-type="pmid">25036584</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pandi</surname> <given-names>M.-T.</given-names></name> <name><surname>van der Spek</surname> <given-names>P. J.</given-names></name> <name><surname>Koromina</surname> <given-names>M.</given-names></name> <name><surname>Patrinos</surname> <given-names>G. P.</given-names></name></person-group> (<year>2020</year>). <article-title>A novel text-mining approach for retrieving pharmacogenomics associations from the literature</article-title>. <source>Front. Pharmacol.</source> <volume>11</volume>:<fpage>602030</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fphar.2020.602030</pub-id>, PMID: <pub-id pub-id-type="pmid">33343371</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Prokhorenkova</surname> <given-names>L.</given-names></name> <name><surname>Gusev</surname> <given-names>G.</given-names></name> <name><surname>Vorobev</surname> <given-names>A.</given-names></name> <name><surname>Dorogush</surname> <given-names>A. V.</given-names></name> <name><surname>Gulin</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>CatBoost: unbiased boosting with categorical features</article-title>&#x201D; in <source>Advances in neural information processing systems</source>, vol. <volume>31</volume>.</citation></ref>
<ref id="ref20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Ren</surname> <given-names>P.</given-names></name> <name><surname>ValizadehAslani</surname> <given-names>T.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Hu</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Fine-tuning BERT for automatic ADME semantic labeling in FDA drug labeling to enhance product-specific guidance assessment</article-title>. <source>J. Biomed. Inform.</source> <volume>138</volume>:<fpage>104285</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jbi.2023.104285</pub-id>, PMID: <pub-id pub-id-type="pmid">36632860</pub-id></citation></ref>
<ref id="ref21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shwartz-Ziv</surname> <given-names>R.</given-names></name> <name><surname>Armon</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>Tabular data: deep learning is not all you need</article-title>. <source>Inf. Fusion</source> <volume>81</volume>, <fpage>84</fpage>&#x2013;<lpage>90</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.inffus.2021.11.011</pub-id></citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tahmassebi</surname> <given-names>A.</given-names></name> <name><surname>Wengert</surname> <given-names>G. J.</given-names></name> <name><surname>Helbich</surname> <given-names>T. H.</given-names></name> <name><surname>Bago-Horvath</surname> <given-names>Z.</given-names></name> <name><surname>Alaei</surname> <given-names>S.</given-names></name> <name><surname>Bartsch</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Impact of machine learning with multiparametric magnetic resonance imaging of the breast for early prediction of response to neoadjuvant chemotherapy and survival outcomes in breast cancer patients</article-title>. <source>Investig. Radiol.</source> <volume>54</volume>, <fpage>110</fpage>&#x2013;<lpage>117</lpage>. doi: <pub-id pub-id-type="doi">10.1097/RLI.0000000000000518</pub-id>, PMID: <pub-id pub-id-type="pmid">30358693</pub-id></citation></ref>
<ref id="ref23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tong</surname> <given-names>W.</given-names></name> <name><surname>Xie</surname> <given-names>Q.</given-names></name> <name><surname>Hong</surname> <given-names>H.</given-names></name> <name><surname>Shi</surname> <given-names>L.</given-names></name> <name><surname>Fang</surname> <given-names>H.</given-names></name> <name><surname>Perkins</surname> <given-names>R.</given-names></name></person-group> (<year>2004</year>). <article-title>Assessment of prediction confidence and domain extrapolation of two structure&#x2013;activity relationship models for predicting estrogen receptor binding activity</article-title>. <source>Environ. Health Perspect.</source> <volume>112</volume>, <fpage>1249</fpage>&#x2013;<lpage>1254</lpage>. doi: <pub-id pub-id-type="doi">10.1289/txg.7125</pub-id>, PMID: <pub-id pub-id-type="pmid">15345371</pub-id></citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>ValizadehAslani</surname> <given-names>T.</given-names></name> <name><surname>Shi</surname> <given-names>Y.</given-names></name> <name><surname>Ren</surname> <given-names>P.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Hu</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>PharmBERT: a domain-specific BERT model for drug labels</article-title>. <source>Brief. Bioinform.</source> <volume>24</volume>:<fpage>bbad226</fpage>. doi: <pub-id pub-id-type="doi">10.1093/bib/bbad226</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Watson</surname> <given-names>K. T.</given-names></name> <name><surname>Barash</surname> <given-names>P. G.</given-names></name></person-group> (<year>2009</year>). <article-title>The new Food and Drug Administration drug package insert: implications for patient safety and clinical care</article-title>. <source>Anesth. Analg.</source> <volume>108</volume>, <fpage>211</fpage>&#x2013;<lpage>218</lpage>. doi: <pub-id pub-id-type="doi">10.1213/ane.0b013e31818c1b27</pub-id>, PMID: <pub-id pub-id-type="pmid">19095852</pub-id></citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>L.</given-names></name> <name><surname>Ingle</surname> <given-names>T.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Zhao-Wong</surname> <given-names>A.</given-names></name> <name><surname>Harris</surname> <given-names>S.</given-names></name> <name><surname>Thakkar</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Study of serious adverse drug reactions using FDA-approved drug labeling and MedDRA</article-title>. <source>BMC Bioinformatics</source> <volume>20</volume>, <fpage>129</fpage>&#x2013;<lpage>139</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s12859-019-2628-5</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Wu</surname> <given-names>L.</given-names></name> <name><surname>Chen</surname> <given-names>M.</given-names></name> <name><surname>Tong</surname> <given-names>W.</given-names></name></person-group> (<year>2021</year>). <article-title>BERT-based natural language processing of drug labeling documents: a case study for classifying drug-induced liver injury risk</article-title>. <source>Front Artif Intell</source> <volume>4</volume>:<fpage>729834</fpage>. doi: <pub-id pub-id-type="doi">10.3389/frai.2021.729834</pub-id></citation></ref>
<ref id="ref28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Xiao</surname> <given-names>W.</given-names></name> <name><surname>Tong</surname> <given-names>W.</given-names></name> <name><surname>Borlak</surname> <given-names>J.</given-names></name> <name><surname>Chen</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>A systematic comparison of hepatobiliary adverse drug reactions in FDA and EMA drug labeling reveals discrepancies</article-title>. <source>Drug Discov. Today</source> <volume>27</volume>, <fpage>337</fpage>&#x2013;<lpage>346</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.drudis.2021.09.009</pub-id></citation></ref>
<ref id="ref29"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>A. Y.</given-names></name> <name><surname>Lam</surname> <given-names>S. S. W.</given-names></name> <name><surname>Liu</surname> <given-names>N.</given-names></name> <name><surname>Pang</surname> <given-names>Y.</given-names></name> <name><surname>Chan</surname> <given-names>L. L.</given-names></name> <name><surname>Tang</surname> <given-names>P. H.</given-names></name></person-group> <article-title>Development of a radiology decision support system for the classification of MRI brain scans</article-title>. in <conf-name>2018 IEEE/ACM 5th International Conference on Big Data Computing Applications and Technologies (BDCAT)</conf-name>. (<year>2018</year>). <publisher-name>IEEE</publisher-name>.</citation></ref>
</ref-list>
</back>
</article>
