<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Ind. Microbiol.</journal-id>
<journal-title>Frontiers in Industrial Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Ind. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">2813-7809</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/finmi.2024.1404729</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Industrial Microbiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Predicting antimicrobial properties of lignin derivatives through combined data driven and experimental approach</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kalinoski</surname>
<given-names>Ryan M.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">*</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Shao</surname>
<given-names>Qing</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Shi</surname>
<given-names>Jian</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">*</xref>
<uri xlink:href="https://loop.frontiersin.org/people/438870"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Biosystems and Agricultural Engineering, University of Kentucky</institution>, <addr-line>Lexington, KY</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Chemical and Materials Engineering, University of Kentucky</institution>, <addr-line>Lexington, KY</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Minliang Yang, North Carolina State University, United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Maciej Szaleniec, Polish Academy of Sciences, Poland</p>
<p>Sandra Vojnovic, University of Belgrade, Serbia</p>
<p>Luan Luong Chu, Vietnam National University, Vietnam</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Ryan M. Kalinoski, <email xlink:href="mailto:rmkalinoski4@gmail.com">rmkalinoski4@gmail.com</email>; Jian Shi, <email xlink:href="mailto:j.shi@uky.edu">j.shi@uky.edu</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>12</day>
<month>08</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>2</volume>
<elocation-id>1404729</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>09</day>
<month>07</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Kalinoski, Shao and Shi</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Kalinoski, Shao and Shi</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Meta-analysis, experimental and data-driven quantitative structure&#x2013;activity relationship (QSAR) models were developed to predict the antimicrobial properties of lignin derivatives. Five machine learning algorithms were applied to develop QSAR models based on the ChEMBL, a public non-lignin specific database. QSAR models were refined using ordinary-least-square regressions with a meta-analysis dataset extracted from literature and an experimental dataset. The minimum inhibition concentration (MIC) values of compounds in the meta-analysis dataset correlate to classification-based descriptors and the number of aliphatic carboxylic acid groups (R<sup>2</sup>&#xa0;=&#xa0;0.759). Comparatively, QSARs derived from the experimental datasets suggest that the number of aromatic hydroxyl groups were better predictors of Bacterial Load Difference (BLD, R<sup>2</sup>&#xa0;=&#xa0;0.831) for <italic>Bacillus subtilis</italic>, while the number of alkyl aryl groups were the strongest correlation in predicting the BLD (R<sup>2</sup>&#xa0;=&#xa0;0.682) of <italic>Escherichia coli.</italic> This study provides insights into the type of descriptors that correlate to antimicrobial activity and guides the valorization of lignin into sustainable antimicrobials for potential applications in food preservation, fermentation, and other industrial sectors.</p>
</abstract>
<kwd-group>
<kwd>quantitative structure&#x2013;activity relationship</kwd>
<kwd>machine learning</kwd>
<kwd>open-source database</kwd>
<kwd>meta-analysis</kwd>
<kwd>lignin valorization</kwd>
</kwd-group>
<contract-num rid="cn001">1355438, 1632854</contract-num>
<contract-num rid="cn002">1018315</contract-num>
<contract-sponsor id="cn001">National Science Foundation<named-content content-type="fundref-id">10.13039/100000001</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">National Institute of Food and Agriculture<named-content content-type="fundref-id">10.13039/100005825</named-content>
</contract-sponsor>
<counts>
<fig-count count="5"/>
<table-count count="5"/>
<equation-count count="1"/>
<ref-count count="50"/>
<page-count count="13"/>
<word-count count="8787"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Fuels and Chemicals</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<title>Introduction</title>
<p>Due to the overuse of antibiotics in our society, there has been a renewed interest in natural compounds for antimicrobial discovery amongst the scientific community (<xref ref-type="bibr" rid="B47">WHO, 2014</xref>; <xref ref-type="bibr" rid="B19">Harvey et&#xa0;al., 2015</xref>). Plant-based phenolics have a wide spectrum of antimicrobial activity and a variety of ring structures with low ecotoxicity that makes them an promising source of potential antimicrobial replacements (<xref ref-type="bibr" rid="B43">Upadhyay et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B19">Harvey et&#xa0;al., 2015</xref>). To this end, lignin is one of the most abundant naturally occurring sources of phenolic polymers on earth and is currently considered a major waste product in the paper and pulp industries and industrial lignocellulosic biorefineries (<xref ref-type="bibr" rid="B25">Mathew et&#xa0;al., 2018</xref>). Lignin is known to have antimicrobial properties against different microorganisms, which is due to the phenolic subunits that comprise lignin&#x2019;s polyphenolic structure (<xref ref-type="bibr" rid="B41">Telysheva et&#xa0;al., 2005</xref>; <xref ref-type="bibr" rid="B12">Cazacu et&#xa0;al., 2013</xref>). Lignin&#x2019;s antimicrobial properties are dictated by the source of the lignin, its extraction methods and chemical structure (i.e., monomers, oligomers, and functional groups) (<xref ref-type="bibr" rid="B12">Cazacu et&#xa0;al., 2013</xref>; <xref ref-type="bibr" rid="B9">Calvo-Flores et&#xa0;al., 2015</xref>). In general, it is believed that lignin phenolics have a mode of action that centers around their ability to increase the ion permeability of cell membranes or by causing direct cell membrane damage resulting in cell lysis (<xref ref-type="bibr" rid="B3">Barber et&#xa0;al., 2000</xref>; <xref ref-type="bibr" rid="B14">Dong et&#xa0;al., 2011</xref>; <xref ref-type="bibr" rid="B15">Espinoza-Acosta et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B48">Yang et&#xa0;al., 2018</xref>). However, lignin&#x2019;s inhomogeneity and complex structure greatly reduces its capacity to be used in industrial and commercial sectors.</p>
<p>While a variety of technical lignin (i.e., Kraft lignin and organosolv lignin) with large undefined structures have had notable antimicrobial properties, there remains inconsistencies in different batches, across different lignin sources and extraction methods (<xref ref-type="bibr" rid="B12">Cazacu et&#xa0;al., 2013</xref>; <xref ref-type="bibr" rid="B9">Calvo-Flores et&#xa0;al., 2015</xref>). Conversely, when lignin is depolymerized into smaller more defined structures, these smaller oligomers and phenolic monomers have shown increased antimicrobial activity and higher specificity (<xref ref-type="bibr" rid="B49">Zemek et&#xa0;al., 1979</xref>). Thus, to increase the effectiveness and selectivity of lignin&#x2019;s antimicrobial properties, it is necessary to depolymerize the polyphenolic structure of technical lignins into smaller units.</p>
<p>There are a plethora of lignin depolymerization techniques including: pyrolysis, acid/base/metal catalyzed hydrolysis, hydrogenolysis and oxidation (<xref ref-type="bibr" rid="B28">Pandey and Kim, 2011</xref>; <xref ref-type="bibr" rid="B45">Wang et&#xa0;al., 2013</xref>; <xref ref-type="bibr" rid="B38">Sun et&#xa0;al., 2018</xref>). Depending on the lignin source each depolymerization method will produce a variety of different phenolic compounds (monomers and oligomers) with potential antimicrobial properties in the form of a bio-oil. Pyrolysis oils, liquid smoke and wood vinegars are derived from the liquid fraction obtained from the incomplete combustion of wood and other lignocellulosic materials. These products have been used extensively in human history to preserve food by smoking and creating a protective barrier on wood for building applications (<xref ref-type="bibr" rid="B24">Louren&#xe7;on et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B32">Sari et&#xa0;al., 2019</xref>). More recently, pyroligneous acid from the slow pyrolysis of hardwood has shown significant antimicrobial activity against multi-antibiotic resistant strains of <italic>E. coli</italic>, <italic>Pseudomonas aeruginosa</italic>, <italic>Staphylococcus aureus, Candida albicans</italic> and <italic>Cryptococcus neoformans</italic>, based on agar diffusion tests (<xref ref-type="bibr" rid="B13">de Souza Araujo et&#xa0;al., 2018</xref>). The pyrolysis oil from pine trees has also been shown to have antimicrobial properties against the foodborne pathogens, <italic>Bacillus cereus</italic> and <italic>Listeria monocytogenese</italic>, at concentrations ranging from 500 to 1,000 ug/ml (<xref ref-type="bibr" rid="B29">Patra et&#xa0;al., 2015</xref>). The main antimicrobial components of these products have been attributed to phenolics, furans, formaldehyde, and organic acids. Wood vinegars from sapwood were found to have significant antimicrobial activity against <italic>Ralstonia solanacearum</italic>, <italic>Phytophthora capsici</italic>, <italic>Fusarium oxysporum</italic>, and <italic>Pythium splendens</italic> (<xref ref-type="bibr" rid="B21">Hwang et&#xa0;al., 2005</xref>).</p>
<p>While lignin bio-oils have shown promising antimicrobial properties for a variety of industrial applications, questions remain as to what individual compounds are responsible for their diverse antimicrobial properties (<xref ref-type="bibr" rid="B24">Louren&#xe7;on et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B32">Sari et&#xa0;al., 2019</xref>). In practice, when antimicrobials are developed, they are usually composed of a single compound or combination of a few compounds. When considering the use of lignin-based bio-oils it would be incredibly difficult to attribute a single compound to its antimicrobial properties, as it is too complex of a mixture. Therefore, methods need to be developed that can predict the antimicrobial potential of lignin derivatives, so that the search for lignin depolymerization products with enhanced antimicrobial properties can be expedited.</p>
<p>Quantitative structure&#x2013;activity relationship (QSAR) models are an indispensable tool in drug design and discovery including predicting antimicrobial properties. They work by finding relationships between the variations in calculated molecular descriptors (properties) or fingerprints (functional groups) with the biological activity of a group of compounds, so that biological activity of new chemical entities can be assessed more quickly (<xref ref-type="bibr" rid="B34">Shahlaei, 2013</xref>). QSAR modeling for predicting antimicrobial properties of polyphenols typically utilizes experimentally derived datasets with a limited number of compounds (&lt;50) and selected descriptors for developing a predictive regression type model, such as multiple linear regressions (MLR) (<xref ref-type="bibr" rid="B2">Araya-Cloutier et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B8">Bouarab-Chibane et&#xa0;al., 2019b</xref>). While this increases the specificity of the model to predict the identified target compounds, it simultaneously limits the model&#x2019;s ability to predict the activity of new compounds with a wider variety of structures. One of the ways to circumvent this issue would be to increase dataset size and compound variability. However, due to the lengthy experimental procedures used to measure antimicrobial activity, and the fact that many lignin oligomers after depolymerization are currently unidentifiable, it would be difficult to drastically increase the number of compounds tested in an efficient manner. Given the recent advances in machine learning and the increase in the amount of chemical and biological activity data available in the public domain in recent years (<xref ref-type="bibr" rid="B10">Camacho et&#xa0;al., 2018</xref>), QSAR models that can explore a large chemical space (thousands of compounds) can now be more widely applied (<xref ref-type="bibr" rid="B23">Lenselink et&#xa0;al., 2017</xref>).</p>
<p>In this context, the aim of this study was to develop and compare QSAR models that can predict the antimicrobial properties of lignin derivatives against representative Gram-positive (<italic>Bacillus subtilis</italic>) and negative bacteria (<italic>Escherichia coli</italic>). The compounds used to construct the models were selected from 1) a large public access database that were non-specific to lignin, 2) a database created from a meta-analysis of available lignin compounds with activity measurements, and 3) an experimentally derived dataset of lignin monomers and dimers. ChEMBL was used as the open access database, which contains over 1.9 million distinct bioactive molecules with drug-like properties and 16 million activity measurements (<xref ref-type="bibr" rid="B16">Gaulton et&#xa0;al., 2012</xref>). Since minimum inhibitory concentration (MIC) is one the most widely used antimicrobial activity measurements (<xref ref-type="bibr" rid="B1">Andrews, 2001</xref>), both the ChEMBL and meta-analyses datasets used MIC to describe the compounds activity. For both <italic>B. subtilis</italic> and <italic>E. coli</italic>, three distinct datasets from ChEMBL were obtained by first choosing all the available compounds with MIC measurements against both organisms, selecting a subset of compounds with only C, H, and O atoms (the only atoms present in lignin), and then an additional subset of compounds with at least one phenolic ring. By enhancing the quality of these datasets and making them more lignin specific to improve the accuracy of our models we are applying a more data-centric approach to developing our models, which has become an emerging trend in data science (<xref ref-type="bibr" rid="B42">Tsai et&#xa0;al., 2020</xref>). Due to the large sizes of these ChEMBL datasets, five different regression-based machine learning algorithms were used to create their QSAR models: support vector machine, random forest, k-nearest neighbor, decision tree, and neural networks.</p>
<p>Next, a meta-analysis of the available literature with MIC activity measurements for lignin derivatives against both <italic>B. subtilis</italic> and <italic>E. coli</italic> was conducted. Not only was this dataset used to develop a QSAR model using ordinary least square (OLS) regressions, but it was also used as a validation set for determining the ChEMBL-based model&#x2019;s performance for predicting lignin specific compounds.</p>
<p>Finally, a variety of commercially available lignin monomers and dimers were screened for antimicrobial properties against <italic>B. subtilis</italic> and a subsequent OLS regression based QSAR was developed. The activity measurement used in the experimental set was the Bacterial Load Difference (BLD) (percent inhibition of growth) as it more easily measured, encompasses the low antibacterial activity, absence of antibacterial activity, and potential growth promoting effect sometimes observed with phenolics compared to MIC (<xref ref-type="bibr" rid="B8">Bouarab-Chibane et&#xa0;al., 2019b</xref>). The results from this study will provide insights into using different types of databases (open access, meta-analysis, experimentally derived, and lignin specific/non-specific) to develop QSAR models with the potential to predict the antibacterial activity of lignin derivatives.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<title>Materials and methods</title>
<sec id="s2_1">
<title>ChEMBL datasets</title>
<p>Antimicrobial data for both <italic>B. subtilis</italic> and <italic>E. coli</italic>, used as representative Gram-positive and -negative bacteria, were obtained from the ChEMBL database (version 27) (<xref ref-type="bibr" rid="B16">Gaulton et&#xa0;al., 2012</xref>). Using the ChEMBL web server, a dataset was created for each bacteria type by selecting minimum inhibitory concentration (MIC) as the biological/antimicrobial activity measurement. The datasets were then downloaded, and further filtering was performed in the Python environment.</p>
<p>Firstly, compounds with &#x2018;non standard unit for type&#x2019; or &#x2018;outside typical range&#x2019; in the data validity comments were removed. Then compounds with standard relation values of &#x2018;&lt;&#x2019; or &#x2018;&gt;&#x2019; were also removed, and duplicates based on compound &#x2018;Molecule ChEMBL ID&#x2019; were averaged into one value. At this point the <italic>B. subtilis</italic> dataset had 9,828 compounds and <italic>E. coli</italic> had 21,657 compounds, which are hereafter referred to as &#x2018;B-All&#x2019; and &#x2018;E-All&#x2019;, respectively. Since lignin has a chemical composition that only contains carbon (C), hydrogen (H), and oxygen (O) atoms, the datasets were further filtered by keeping compounds with only those atoms. This was performed by searching for compounds with a canonical simplified molecular-input line-entry system (SMILES) string with only C, H, and O atoms (<xref ref-type="bibr" rid="B46">Weininger et&#xa0;al., 1989</xref>). The resulting filtering produced a <italic>B. subtilis</italic> dataset with 768 compounds and an <italic>E. coli</italic> dataset with 703 compounds, which are hereafter referred to as &#x2018;B-Sorted&#x2019; and &#x2018;E-Sorted&#x2019;, respectively. Finally, to increase the datasets specificity for predicting lignin phenolics, the previously SMILE sorted dataset was filtered for compounds with at least one phenolic ring. This resulted in a <italic>B. subtilis</italic> dataset with 309 compounds and an <italic>E. coli</italic> dataset with 278 compounds, which are hereafter referred to as &#x2018;B-Phenolic&#x2019; and &#x2018;E-Phenolic&#x2019;, respectively. Therefore, three datasets for both <italic>B. subtilis</italic> and <italic>E. coli</italic> were created with MIC data. Furthermore, MIC values originally determined in &#xb5;g/ml were converted to micromolar values (&#xb5;M/ml) and then converted to pMIC (i.e. -logMIC, in molar) for all datasets (<xref ref-type="bibr" rid="B2">Araya-Cloutier et&#xa0;al., 2018</xref>). One could consider preprocessing the dataset differently for different ML models, especially for ANN as ANN models can bear more noise than some others. However, the purpose of this study is to compare performance across different ML models using the same dataset. We decided to use the same preprocessing methods across all ML models.</p>
</sec>
<sec id="s2_2">
<title>Lignin monomers meta-analysis dataset</title>
<p>A new dataset of MIC biological activity measurements for lignin monomers against both <italic>B. subtilis</italic> and <italic>E. coli</italic> were compiled from published sources. Multidisciplinary databases such as Academic OneFile, Academic Search Complete, EBSCO, and Google Scholar for terms including combinations such as &#x2018;lignin,&#x2019; &#x2018;antimicrobial,&#x2019; &#x2018;phenolic,&#x2019; &#x2018;MIC,&#x2019; &#x2018;monomer,&#x2019; &#x2018;antibacterial,&#x2019; as well as authors with previous work containing appropriate data, were used to find journal articles that contained MIC antimicrobial data for phenolics that can be derived from lignin. In total, 16 compounds were found with MIC data for <italic>B. subtilis</italic> and 27 compounds for <italic>E. coli</italic> (listed in Section 3.2). MIC values originally determined in &#xb5;g/ml were converted to micromolar values (&#xb5;M/ml) and then converted to pMIC (i.e. -logMIC, in molar) prior to modeling (<xref ref-type="bibr" rid="B2">Araya-Cloutier et&#xa0;al., 2018</xref>). The resulting datasets for <italic>B. subtilis</italic> and <italic>E. coli</italic> are hereafter referred to as &#x2018;B-Meta&#x2019; and &#x2018;E-Meta&#x2019;, respectively.</p>
</sec>
<sec id="s2_3">
<title>Experimental dataset</title>
<p>The antibacterial activity of 25 lignin derived monomers and three dimers were assessed by monitoring the cell growth (as represented by the optical density at 600 nm, OD<sub>600</sub>) of <italic>B. subtilis</italic> (NRRL B-354) and <italic>E. coli</italic> using a spectrophotometry. The full list of compounds and subsequent antimicrobial activity measurements are listed in Section 3.3. The monomers were of analytical quality and purchased from either Sigma Aldrich (St. Louis, MO, USA) or TCI America. The guaiacylglycerol-beta-guaiacyl ether dimer was purchased from TCI America, while 2-(2-methoxyphenoxy)-1-(4-methoxyphenyl)ethanol and 3-hydroxy-2-(2-methoxyphenoxy)-1-(4-methoxyphenyl)-1-propanone dimers were kindly provided by Dr. Mark Crocker at the Center for Applied Energy, University of Kentucky (<xref ref-type="bibr" rid="B36">Song et&#xa0;al., 2018</xref>).</p>
<p>Briefly, frozen cultures were first revived in liquid growth media (LB broth, Fisher BioReagents&#x2122;, BP9723) and allowed to grow at 180 rpm shaking speed for 12&#xa0;h at 37&#xb0;C. Afterwards the cells were pelletized, washed, and resuspended in fresh liquid media. To test for the antimicrobial properties, each microbe was cultivated in 96-well plates (run in triplicate) and the OD<sub>600</sub> was monitored for 24&#xa0;h with time points taken every 10&#xa0;min. All wells were brought to an OD<sub>600</sub> of 0.2 prior to growth, and the phenolics were added to treatment wells to create a final concentration of 1 g/L. To facilitate the solubility of the phenolics in media, all cultures had a final ethanol concentration of 5% (v/v). Two controls were used, one having the 5% ethanol concentration, and one having just microbes and the media. To determine how the phenolics affected microbial growth, the percent change in OD<sub>600</sub> of the ethanol control during the exponential phase of growth was compared to the growth of the phenolic treatments. This resulted in the percent decrease in growth or Bacterial Load Difference (BLD) for each phenolic treatment (<xref ref-type="bibr" rid="B8">Bouarab-Chibane et&#xa0;al., 2019b</xref>), with the formula described in <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mtext>BLD&#xa0;</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>%</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>Max&#xa0;OD</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>600</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mtext>Min&#xa0;OD</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>600</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;with&#xa0;phenolic</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>Max&#xa0;OD</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>600</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mtext>Min&#xa0;OD</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>600</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;of&#xa0;Ethanol&#xa0;Control</mml:mtext>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xd7;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mn>100</mml:mn>
</mml:mrow>
</mml:math>
</disp-formula>
<p>After obtaining the BLD values for each phenolic, the structures of each compound were converted to canonical SMILES strings using PubChem for use in descriptor calculations. The final experimental datasets for <italic>B. subtilis</italic> and <italic>E. coli</italic> are here after referred to as &#x2018;B-Experimental&#x2019; and &#x2018;E-Experimental&#x2019;, respectively.</p>
</sec>
<sec id="s2_4">
<title>Descriptor calculations and QSAR modeling</title>
<p>To calculate the various molecular descriptors, all the compound&#x2019;s structures in each dataset were converted into canonical SMILES strings (<xref ref-type="bibr" rid="B46">Weininger et&#xa0;al., 1989</xref>), if not already provided. These SMILES were then entered into an open-access molecular descriptor calculator software package for Python, RDKit (<ext-link ext-link-type="uri" xlink:href="http://www.rdkit.org">http://www.rdkit.org</ext-link>). RDKit has a variety of calculatable descriptors that describe a molecule&#x2019;s lipophilicity (i.e., LogP, LogD), topological indices (i.e., fragment complexity, size, polarity), connectivity indices and different molecular fingerprints (i.e., number of hydroxyl groups, phenolic rings, carboxylic acids etc.). A full list of descriptors and their description is provided in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1.</bold>
</xref> While it is possible to create models with all the calculatable descriptors, a variety of descriptor selection methods were utilized to improve model accuracy by reducing dimensionality of input space without losing important information.</p>
<p>For the B-All, E-ALL, B-Sorted, E-Sorted, B-Phenolic, and E-Phenolic datasets 200 of RDKit&#x2019;s available descriptors were calculated. Highly correlated (|r| &#x2265; 0.8) and constant descriptors were eliminated from the list for each individual dataset. To further reduce the dimensionality of the predictors (descriptors) a principal component analysis (PCA) was performed using scikit-learn (<xref ref-type="bibr" rid="B30">Pedregosa et&#xa0;al., 2011</xref>). The number of new principal components to be used was assessed by plotting the number of components vs the percent explained variance, and the number of components that explained 99% of the variance were chosen for each dataset. After the optimal number of principal components were chosen and calculated these values were used as the independent variables for predicting the pMIC values in the subsequent QSAR models. Before modeling, each of the above dataset&#x2019;s with pMIC and PCA data were randomly split into training (80%) and test (20%) sets three times for cross-validation. We compared and utilized five machine learning algorithms to build the QSAR models for the B-All, E-ALL, B-Sorted, E-Sorted, B-Phenolic, and E-Phenolic datasets. They were the support vector machine (Epsilon-Support Vector Regression), random forest regressor, k-nearest neighbors regressor, decision tree regressor, and neural network regressor (Multi-layer Perceptron regressor) algorithms provided by scikit-learn. The specific settings and parameters used to build each machine learning algorithm are provided below. QSAR models were assessed based on their average coefficient of determination (R<sup>2</sup>) and root mean squared error (RMSE) based on the predictions made for the three training and test sets.</p>
<p>The best QSAR models constructed from the ChEMBL datasets were further tested for prediction accuracy, by using the meta-analysis datasets as a test set for predicting lignin-specific compounds. Kernel density estimate (KDE) plots using the Seaborn plugin for python were constructed to determine the distribution of each dataset&#x2019;s pMIC values. Furthermore, the applicability domain (AD) for estimating the reliability in the prediction of new compounds from the ChEMBL datasets were evaluated against the meta-analysis datasets, according to previous work (<xref ref-type="bibr" rid="B17">Golbraikh et&#xa0;al., 2003</xref>).</p>
<p>For the B-Meta, E-Meta, B-Experimental, and E-Experimental datasets all 200 of RDKit&#x2019;s available descriptors were calculated. Pearson&#x2019;s correlation coefficient (|r| &#x2265; 0.5) was used to select a fixed subset of predictors (descriptors) best able to predict the antimicrobial activities (either pMIC or BLD) using the ordinary least squares (OLS) regression analysis (<xref ref-type="bibr" rid="B20">Hira and Gillies, 2015</xref>). The OLS regressions were performed using Statsmodels (<xref ref-type="bibr" rid="B33">Seabold and Perktold, 2010</xref>). As the number of compounds for each of these datasets were very low (less than 30 compounds), the datasets were not separated into training and test sets due to higher risks of chance correlation and overfitting (<xref ref-type="bibr" rid="B2">Araya-Cloutier et&#xa0;al., 2018</xref>). For each dataset, the selected descriptors were fed into an OLS regression and backwards elimination was used until the significance of each descriptor coefficient in the model (<italic>p</italic>-value) was less than 0.05, which identified the best fitting model.</p>
</sec>
<sec id="s2_5">
<title>Machine learning algorithms</title>
<p>All machine learning models were created using scikit-learn and either the default hyper parameters were used or a number of different parameters through a grid search based exploration of model parameter space was utilized (<xref ref-type="bibr" rid="B23">Lenselink et&#xa0;al., 2017</xref>). The final parameters used for the machine learning algorithms that used grid search for QSAR model development are reported in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Each dataset&#x2019;s final number of compounds, descriptors, and hyper parameters for machine learning algorithms that used grid search parameterization.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Dataset</th>
<th valign="bottom" colspan="2" align="center">Data Processing</th>
<th valign="bottom" align="center">PCA</th>
<th valign="bottom" align="center">k-Nearest Neighbor</th>
<th valign="bottom" colspan="2" align="center">Decision Tree</th>
<th valign="bottom" colspan="3" align="center">Neural Network</th>
</tr>
<tr>
<th valign="bottom" align="center">Compounds</th>
<th valign="bottom" align="center">Descriptors</th>
<th valign="bottom" align="center">Components</th>
<th valign="bottom" align="center">Neighbors</th>
<th valign="bottom" align="center">Depth</th>
<th valign="bottom" align="center">Sample Leaves</th>
<th valign="bottom" align="center">Hidden Layers</th>
<th valign="bottom" align="center">Activation</th>
<th valign="bottom" align="center">Learning Rate</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">B-All</td>
<td valign="top" align="center">9,828</td>
<td valign="top" align="center">118</td>
<td valign="top" align="center">80</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">14</td>
<td valign="top" align="center">50</td>
<td valign="top" align="center">(50, 100, 50)</td>
<td valign="top" align="center">tanh</td>
<td valign="top" align="center">constant</td>
</tr>
<tr>
<td valign="top" align="left">B-Sort</td>
<td valign="top" align="center">768</td>
<td valign="top" align="center">62</td>
<td valign="top" align="center">40</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">50</td>
<td valign="top" align="center">(50, 50, 50)</td>
<td valign="top" align="center">relu</td>
<td valign="top" align="center">constant</td>
</tr>
<tr>
<td valign="top" align="left">B-Phenol</td>
<td valign="top" align="center">309</td>
<td valign="top" align="center">61</td>
<td valign="top" align="center">40</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">10</td>
<td valign="top" align="center">(50, 100, 50)</td>
<td valign="top" align="center">tanh</td>
<td valign="top" align="center">constant</td>
</tr>
<tr>
<td valign="top" align="left">E-All</td>
<td valign="top" align="center">21,657</td>
<td valign="top" align="center">114</td>
<td valign="top" align="center">80</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">13</td>
<td valign="top" align="center">20</td>
<td valign="top" align="center">(100)</td>
<td valign="top" align="center">tanh</td>
<td valign="top" align="center">constant</td>
</tr>
<tr>
<td valign="top" align="left">E-Sort</td>
<td valign="top" align="center">703</td>
<td valign="top" align="center">67</td>
<td valign="top" align="center">40</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">(100)</td>
<td valign="top" align="center">tanh</td>
<td valign="top" align="center">constant</td>
</tr>
<tr>
<td valign="top" align="left">E-Phenol</td>
<td valign="top" align="center">278</td>
<td valign="top" align="center">67</td>
<td valign="top" align="center">40</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">20</td>
<td valign="top" align="center">(50, 50, 50)</td>
<td valign="top" align="center">relu</td>
<td valign="top" align="center">constant</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The datasets denoted with &#x2018;B&#x2019; and &#x2018;E&#x2019; represent the data utilized from ChEMBL for <italic>B. subtilis</italic> and <italic>E. coli</italic>, respectively.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The support vector machine (SVM) or Epsilon-Support Vector regression is a non-linear regression that calculates an optimal hyper-plane where the distance and error between each data points is minimized (<xref ref-type="bibr" rid="B26">Mei et&#xa0;al., 2005</xref>). The SVM performed here used the default parameters provided by scikit-learn. These included a radial basis function kernel, gamma of 1/number of descriptors, parameter cost of 1, and epsilon of 0.1.</p>
<p>Decision tree regressors (DT) are a non-parametric learning method that works by creating a set of binary rules to calculate the target value by dividing the data into subsets that contain data with similar values (<xref ref-type="bibr" rid="B4">Basant et&#xa0;al., 2016</xref>). The DT used a grid search to select the optimal maximum depth from 1 to 21 and minimum number of sample leaves from 1 to 100 for each dataset, by fitting the training set and using five cross-fold validations and RMSE to choose the best values. All other parameters utilized the scikit-learn default settings.</p>
<p>The random forest regressor (RF) is an ensemble learning method for non-linear regression analysis, that operates by constructing a multitude of decision trees and outputting the mean prediction of the individual trees (<xref ref-type="bibr" rid="B40">Svetnik et&#xa0;al., 2003</xref>). We used all the default parameters provided by scikit-learn, but the number of estimators was increased from the default 100 to 500.</p>
<p>K-nearest neighbor regressions (KNN) are a non-parametric method that stores all available cases and predicts a continuous target based on the similarity measure (distance function) between different features in the same neighborhood (<xref ref-type="bibr" rid="B50">Zheng and Tropsha, 2000</xref>). The KNN used a grid search to select the optimal number of neighbors from 2 to 15 for each dataset, by fitting the training set using five cross-fold validations and RMSE to choose the best number of neighbors. The rest of the parameters including the weight function and leaf size utilized scikit-learns default settings.</p>
<p>Neural networks (NN) are brain-inspired algorithms where input features are fed into an input layer, and after a number of nonlinear transformations are performed in a hidden layer, the predictions are generated in an output layer to produce a regression (<xref ref-type="bibr" rid="B23">Lenselink et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B10">Camacho et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B44">Vamathevan et&#xa0;al., 2019</xref>). The ANN relied on most of the default settings provided by scikit-learn&#x2019;s MLPRegressor neural network. In order to optimize the model&#x2019;s hyper-parameters, the GridSearchCV function was utilized, with the best parameters being selected based on the best five cross-fold validations and RMSE score. The hyper-parameters chosen to be optimized were the hidden layer sizes, activation type, and learning rate. Specifically, for the hidden layers and number of neurons in each layer was either three layers with 50&#x2013;100 neurons [(50,50,50), (50,100,50)], or the default setting of a single layer and 100 neurons (100). The activation functions used for the hidden layers were either the rectified linear unit function &#x2018;relu&#x2019; or the hyperbolic tan function &#x2018;tanh&#x2019;. The learning rate schedule for weight updates was either the constant or adaptive learning rates. For further documentation and explanation of the other default settings used to create this model please refer to the scikit-learn version 0.23.2 documentation.</p>
</sec>
<sec id="s2_6">
<title>Software used</title>
<p>Python (version 3.7.7) was used with the following libraries: RDKit (version 2020.03.6) for the calculation of fingerprints and descriptors, Scikit-learn (version 0.23.2) for all machine learning algorithms and descriptor selection techniques, seaborn (version 0.11.0) with Matplotlib (version 3.3.2) for all figure visualizations, and Pandas (version 1.1.2) for all dataset analysis and manipulation.</p>
</sec>
</sec>
<sec id="s3" sec-type="results|discussion">
<title>Results and discussion</title>
<sec id="s3_1">
<title>ChEMBL dataset models</title>
<p>The open access database, ChEMBL, was used to develop datasets of compounds with antimicrobial activity measurements (MIC) against both <italic>B. subtilis</italic> and <italic>E. coli</italic>. These datasets were used alongside machine learning algorithms to develop QSAR models with the potential to predict the antimicrobial activity of lignin derived phenolics, from compounds that are not lignin specific.</p>
<p>The initial ChEMBL datasets created for <italic>B. subtilis</italic> (B-All) and <italic>E. coli</italic> (E-All) contained 9,628 and 21,657 compounds, respectively. These datasets were filtered into two additional subsets, that contained compounds having more similar structures to that of lignin derivatives. The first subset was created by selecting compounds with only C, H, and O atoms, resulting in a <italic>B. subtilis</italic> dataset with 768 compounds (B-Sort) and an <italic>E. coli</italic> dataset with 703 compounds (E-Sort). By removing compounds with nitrogenous, chlorine, or fluorine based functional groups, the remaining compounds could have more similar chemical characteristics to that of lignin derivatives. Then those subsets were further filtered by selecting compounds with at least one phenolic ring, resulting in a <italic>B. subtilis</italic> dataset with 309 compounds (B-Phenolic) and an <italic>E. coli</italic> dataset with 278 compounds (E-Phenolic). Lignin&#x2019;s antimicrobial properties have been reported to attribute to its phenolic structures, so it was important to include a subset of compounds that contained only phenolic-based structures. These were the final six datasets used for QSAR model development from the ChEMBL database.</p>
<p>The QSAR models used antimicrobial activities measured in pMIC (-log MIC, in &#xb5;M/mL) values as the target and the molecular descriptors calculated from RDKit as the variables. <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1</bold>
</xref> lists the descriptors and molecular fingerprints used in this work. These descriptors represent the lipophilicity (i.e., LogP, LogD), topological indices (i.e., fragment complexity, size, polarity), connectivity indices and functional groups. They are selected based on previous work (<xref ref-type="bibr" rid="B37">Speck-Planche et&#xa0;al., 2012</xref>; <xref ref-type="bibr" rid="B39">Svensson et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B2">Araya-Cloutier et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B8">Bouarab-Chibane et&#xa0;al., 2019b</xref>). <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold></xref> <xref ref-type="supplementary-material" rid="SM2">
<bold>(Additional File 2)</bold>
</xref> lists the specific descriptors used for each dataset after pre-processing. <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> summarizes the number of descriptors.</p>
<p>Principal component analysis (PCA) was used to reduce the number of descriptors and the dimensionality of the feature space. PCA reconstructs features of a dataset into a new set of uncorrelated features called principal components (PCs). The optimal number of new PCs for each dataset was selected by the number of components that explained 99% of the variance in the dependent variable. <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> shows the number of PCs vs the percent explained variance, and <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> summarizes the number that explained 99% of the variance. Since this feature extraction technique creates new independent variables that are less interpretable, the ability to examine how each descriptor influences pMIC is no longer easily obtainable. This is actually beneficial when using the ChEMBL datasets, as we are attempting to predict the antimicrobial properties of lignin with non-lignin based compounds from a data driven perspective, and do not need to understand the exact relationship between these compound&#x2019;s descriptors and pMIC values. Therefore, the QSAR models were developed from the pMIC and PC values from each dataset using five popular regression-based machine learning algorithms: support vector machine (SVM), random forest (RF), k-nearest neighbor (KNN), decision tree (DT), and neural networks (NN).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Plots showing the number of components from the principal component analysis performed on each datasets descriptor set against the explained variance (%). The ChEMBL datasets for <italic>B. subtilis</italic> are B-All <bold>(A)</bold>, B-Sort <bold>(B)</bold>, and B-Phenol <bold>(C)</bold>, while the <italic>E. coli</italic> sets are E-All <bold>(D)</bold>, E-Sort <bold>(E)</bold>, and E-Phenol <bold>(F)</bold>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="finmi-02-1404729-g001.tif"/>
</fig>
<p>The performance summary of five machine learning QSAR models for each ChEMBL dataset is provided in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Tables S3&#x2013;S5</bold>
</xref>, respectively. Each dataset was split randomly into three different training (80%) and test (20%) sets, with each of these same sets being applied to the different model types. The training sets were used to build each machine learning model and the test sets were used for model validation. The metrics used for measuring model performance was the average coefficient of determination (R<sup>2</sup>) and root mean square error (RMSE) for the three training and test sets. The better performing model is identified as having a high R<sup>2</sup> and low RMSE value for the average test scores and training scores; thus, <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> provides a summary of the best fitting models for each ChEMBL dataset. When comparing models, if one model had a higher R<sup>2</sup> and lower RMSE for the test sets, but not the training sets, the model with better performance for the test set was chosen, as it is ultimately the more important metric (<xref ref-type="bibr" rid="B5">Bengio et&#xa0;al., 2017</xref>). For example, B-All&#x2019;s best performing QSAR model was the KNN algorithm (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S3</bold>
</xref>), as it had the highest R<sup>2</sup> of 0.69 for the test set, despite a slightly lower R<sup>2</sup> for the training sets (0.86) compared to the RF algorithm (0.95). Accordingly, the E-All, B-Sort, E-Sort, B-Phenol, and E-phenol datasets had the most robust QSAR models using the RF, NN, KNN, RF, and KNN algorithms, respectively (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>). Not surprisingly the table shows that only NN model has sensible generalization capabilities as there is almost no difference between RMSE for training and test set. The only similar model is E-phenol KNN but it has low prediction quality, so this is most probably effect of high error and low prediction capability of the model in general. The stark difference between training and test group suggests that the model is over-trained or has only interpolation capabilities (like B-ALL KNN and E-ALL RF &#x2013; huge differences in R<sup>2</sup>). Future work could test an ensemble system with different techniques to get averaged predications; a possible way to improve the model capability.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>QSAR model performance for the best fitting machine learning models for each ChEMBL dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Dataset</th>
<th valign="middle" rowspan="2" align="center">Best Fitting Model</th>
<th valign="bottom" colspan="2" align="center">Test</th>
<th valign="bottom" colspan="2" align="center">Train</th>
</tr>
<tr>
<th valign="bottom" align="center">R<sup>2</sup>
</th>
<th valign="bottom" align="center">RMSE</th>
<th valign="bottom" align="center">R<sup>2</sup>
</th>
<th valign="bottom" align="center">RMSE</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="bottom" align="center">B-All</td>
<td valign="top" align="center">KNN</td>
<td valign="bottom" align="center">0.69 &#xb1; 0.008</td>
<td valign="bottom" align="center">0.58 &#xb1; 0.009</td>
<td valign="bottom" align="center">0.86 &#xb1; 0.001</td>
<td valign="bottom" align="center">0.39 &#xb1; 0.002</td>
</tr>
<tr>
<td valign="bottom" align="center">E-All</td>
<td valign="top" align="center">RF</td>
<td valign="bottom" align="center">0.69 &#xb1; 0.004</td>
<td valign="bottom" align="center">0.62 &#xb1; 0.002</td>
<td valign="bottom" align="center">0.95 &#xb1; 0.000</td>
<td valign="bottom" align="center">0.24 &#xb1; 0.001</td>
</tr>
<tr>
<td valign="bottom" align="center">B-Sort</td>
<td valign="top" align="center">NN</td>
<td valign="top" align="center">0.71 &#xb1; 0.014</td>
<td valign="top" align="center">0.49 &#xb1; 0.005</td>
<td valign="top" align="center">0.79 &#xb1; 0.032</td>
<td valign="top" align="center">0.41 &#xb1; 0.036</td>
</tr>
<tr>
<td valign="bottom" align="center">E-Sort</td>
<td valign="top" align="center">KNN</td>
<td valign="top" align="center">0.49 &#xb1; 0.067</td>
<td valign="top" align="center">0.79 &#xb1; 0.017</td>
<td valign="top" align="center">0.69 &#xb1; 0.007</td>
<td valign="top" align="center">0.42 &#xb1; 0.007</td>
</tr>
<tr>
<td valign="bottom" align="center">B-Phenol</td>
<td valign="top" align="center">RF</td>
<td valign="top" align="center">0.57 &#xb1; 0.007</td>
<td valign="top" align="center">0.59 &#xb1; 0.036</td>
<td valign="top" align="center">0.63 &#xb1; 0.005</td>
<td valign="top" align="center">0.42 &#xb1; 0.007</td>
</tr>
<tr>
<td valign="bottom" align="center">E-Phenol</td>
<td valign="top" align="center">KNN</td>
<td valign="top" align="center">0.38 &#xb1; 0.019</td>
<td valign="top" align="center">0.75 &#xb1; 0.041</td>
<td valign="top" align="center">0.53 &#xb1; 0.015</td>
<td valign="top" align="center">0.76 &#xb1; 0.002</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The datasets denoted with &#x2018;B&#x2019; and &#x2018;E&#x2019; represent the data utilized from ChEMBL for <italic>B. subtilis</italic> and <italic>E. coli</italic>, respectively. Measured by average coefficient of determination (R<sup>2</sup>) and root mean square error (RMSE) for both the training and test sets, where values are mean &#xb1; SE (n=3). Each dataset was split into random test and train sets three different times to obtain the average performance score. The number of compounds, selected descriptors, and number of principal components used to develop models can be found in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> and <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>A common theme with all the models in each dataset, was that the R<sup>2</sup> for the test set was always lower than the training set. This could be a sign of model overfitting or unrepresentative data between the training and test sets (<xref ref-type="bibr" rid="B5">Bengio et&#xa0;al., 2017</xref>). However, all the models had very low SE values when averaging the R<sup>2</sup> values of the three different test/train splits for cross-validation, which would suggest compounds are not being underrepresented. The number of independent variables (ICs) used for each dataset were also rather large (80 or 40), which could contribute to overfitting, but they explained 99% of the dependent variable&#x2019;s variation and when smaller numbers of ICs were used the model&#x2019;s performance drastically decreased (data not shown). Coupled with the fact that most models used a grid search parametrization technique to fine tune the hyperparameters, these discrepancies may just be a function of the data itself and not with how the models were evaluated or fit. Furthermore, the E-Sort, B-Phenol, and E-Phenol datasets did not have any QSAR models with a R<sup>2</sup> &gt; 0.6, which is usually needed to describe a truly predictive model (<xref ref-type="bibr" rid="B34">Shahlaei, 2013</xref>). Yet, since these datasets are not lignin-specific, the true measure of these model&#x2019;s performance needs to be evaluated with an additional test set of actual lignin derived compounds.</p>
<p>To this end, the available literature was searched for lignin derived monomers that had reported MIC values against <italic>B. subtilis</italic> and <italic>E. coli</italic>. The results from this meta-analysis are reported in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S6</bold>
</xref>, where 16 compounds were found with MIC data for <italic>B. subtilis</italic> (B-Meta) and 27 compounds for <italic>E. coli</italic> (E-Meta). These two datasets were then evaluated as an additional test set for each of the best performing QSAR models found for each ChEMBL dataset, described above. The data is summarized in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>, where the predicted vs actual pMIC values of the lignin monomers are plotted. It can immediately be seen that none of the ChEMBL QSAR models could accurately predict the lignin monomers. All the models predicted the lignin compounds as having pMIC values roughly less than 2.5, when they are reported as actually having pMIC values greater than 2.5. This suggests these models are grossly underpredicting the pMIC values for the lignin compounds, which would correlate to them having a lower MIC and subsequently greater antimicrobial activity. To understand this, a kernel density estimate (KDE) plot for the ChEMBL and meta-analysis datasets were constructed to visualize the distribution of their pMIC values (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>), and their applicability domains were evaluated (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Tables S7, S8</bold>
</xref>, respectively).</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Plots of predicted versus actual pMIC values for the B-Meta <bold>(A&#x2013;C)</bold> and E-Meta <bold>(D&#x2013;F)</bold> datasets by utilizing the best QSAR models developed from the ChEMBL datasets. The ChEMBL datasets used to predict pMIC of the meta-analysis datasets for <italic>B. subtilis</italic> are B-All <bold>(A)</bold>, B-Sort <bold>(B)</bold>, and B-Phenol <bold>(C)</bold>, while the <italic>E. coli</italic> sets are E-All <bold>(D)</bold>, E-Sort <bold>(E)</bold>, and E-Phenol <bold>(F)</bold>. The best QSAR models used in each prediction are as follows: RF <bold>(A)</bold>, NN <bold>(B)</bold>, RF <bold>(C)</bold>, RF <bold>(D)</bold>, SVM <bold>(E)</bold>, and KNN <bold>(F)</bold>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="finmi-02-1404729-g002.tif"/>
</fig>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Kernel density estimates describing the distribution of pMIC values for the <italic>B. subtilis</italic> <bold>(A)</bold> and <italic>E. coli</italic> <bold>(B)</bold> ChEMBL/meta-analysis datasets.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="finmi-02-1404729-g003.tif"/>
</fig>
<p>The KDE plots show that the meta-analysis datasets for both <italic>E. coli</italic> and <italic>B. subtilis</italic> have pMIC distributions centered around 3&#x2013;4, while the ChEMBL datasets are centered between 0&#x2013;2.5. Even though the ChEMBL datasets clearly contain compounds with pMIC values within the distribution of the meta-analysis datasets, they did not lie within the applicability domains (AD) of the ChEMBL datasets. The AD is a useful measure for determining the reliability of a model&#x2019;s prediction for a new set of compounds. Based on the PCA for each ChEMBL dataset, their ADs were calculated based on the Euclidean distances among all their compounds and a final threshold value is determined (<xref ref-type="bibr" rid="B17">Golbraikh et&#xa0;al., 2003</xref>). Then, the same measure is calculated for each of the compounds in the meta-analysis dataset to test if they lie within the threshold of the ChEMBL dataset&#x2019;s AD. We can see in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Tables S7, S8</bold>
</xref> that none of the B-Meta or E-Meta compounds fall within the AD of the ChEMBL datasets. The data show that our data-centric approach to creating datasets of traditional pharmacological compounds that are most similar to potential lignin structures still cannot accurately be used to predict the antimicrobial activity of true lignin derivates. Even though these results are not what the authors had hoped, these data create a more concrete conclusion that a comprehensive dataset of lignin derivatives with antimicrobial measurements needs to be developed. Considering this, QSAR models using actual lignin compounds from the meta-analysis datasets, and an experimentally derived dataset were developed and are discussed in the further sections.</p>
</sec>
<sec id="s3_2">
<title>Meta-analysis dataset models</title>
<p>The meta-analysis datasets, used for validating the ChEMBL QSAR models, were applied to develop their own QSARs using ordinary least square (OLS) regressions. Instead of using PCA as a feature extraction technique, univariate feature selection relying on Pearson&#x2019;s correlation coefficient (<italic>r</italic>) was employed. Since OLS regressions rely on linear relationships, it made more sense to utilize Pearson&#x2019;s correlation as it measures the strength of the linear correlation between the independent (descriptors) and dependent variables (pMIC). Therefore, the same 200 molecular descriptors were calculated for the B-Meta and E-Meta datasets, and the descriptors with a <italic>r</italic>&gt;0.5 were selected to develop the OLS regressions. Subsequently, the selected descriptors were fed into an OLS regression and backwards elimination was used until the significance of each descriptor coefficient in the model (<italic>p</italic>-value) was less than 0.05, which indicated the best fitting model. It should be noted that this approach is biased for linear relationships and it undercasts models that have non-linear capabilities.</p>
<p>No reliable QSARs using OLS was obtained for the E-Meta dataset (results not shown). This result was not surprising considering the pMIC distribution in E-Meta dataset had three different centers, as shown in the KDE plot (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>). Its variable distribution and small sample size could prevent the QSAR model from capturing any relevant relationships in the feature space (<xref ref-type="bibr" rid="B8">Bouarab-Chibane et&#xa0;al., 2019b</xref>). Conversely, even though the B-Meta (16 compounds) dataset was smaller than E-Meta (27 compounds), a more successful QSAR model was developed. To improve model capacity, one could use ANN MLP with two regression outputs &#x2013; one for E-Meta and the other for B-Meta. Provided the intrinsic relations are not entirely different this would increase dataset and allow training more sophisticated models.</p>
<p>The model for best predicting the antibacterial activity (pMIC) of the lignin monomers in the B-Meta dataset is summarized in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> and <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>. As observed, the selected OLS model showed good predictive power with a R<sup>2</sup> of 0.759. Three descriptors, SLogP_VSA3, SLogP_VSA5, and fr_AL_COO, were used to develop the best fitting OLS regression model. The SLogP_VSA3 and SLogP_VSA5 descriptors are Molecular Operation Environment (MOE)-type descriptors that bin the output from other descriptor types (i.e., SLogP) and calculate the van der Waals (VDWs) surface area (VSA) of atoms contributing to any specified bin of that output. Thus, SLogP_VSA3 and SLogP_VSA5 calculate the sum of VSA contributions to the lipophilicity measurement SLogP (partition coefficient of compound in two immiscible solvent) within &#x2212;0.2&#x2013;0 and 0.1&#x2013;0.15 bin ranges, respectively. While SLogP and VSA are &#x2018;primary&#x2019; descriptors that have a more-or-less interpretable contributions to a compounds mechanism of action, the MOE-type descriptors are intended to be used as model predictors and are not as interpretable (<xref ref-type="bibr" rid="B22">Labute, 2000</xref>). Therefore, the negative and positive relationships SLogP_VSA3 and SLogP_VSA5 contribute to the OLS regression, can only be used as a data driven identifier for predicting the pMIC values of lignin compounds. On the other hand, the fr_AL_COO descriptor represents the number of aliphatic carboxylic acid groups in each compound and can directly be used to infer the mechanism of action.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Statistical performance of the best OLS models obtained through backwards elimination of descriptors, for predicting pMIC values of lignin phenolics against <italic>B. subtilis</italic> in the B-Meta dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Dataset</th>
<th valign="top" align="left">N</th>
<th valign="top" align="left">R<sup>2</sup>
</th>
<th valign="top" align="left">Descriptor</th>
<th valign="top" align="left">Coefficient</th>
<th valign="top" align="left">Standard Error</th>
<th valign="top" align="left">
<italic>p</italic>-value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="4" align="left">B-Meta</td>
<td valign="middle" rowspan="4" align="left">16</td>
<td valign="middle" rowspan="4" align="left">0.759</td>
<td valign="top" align="left">SLogP_VSA3</td>
<td valign="top" align="left">&#x2212;0.2951</td>
<td valign="top" align="left">0.108</td>
<td valign="top" align="left">0.041</td>
</tr>
<tr>
<td valign="top" align="left">SLogP_VSA5</td>
<td valign="top" align="left">0.6025</td>
<td valign="top" align="left">0.129</td>
<td valign="top" align="left">0.003</td>
</tr>
<tr>
<td valign="top" align="left">fr_AL_COO</td>
<td valign="top" align="left">&#x2212;0.2588</td>
<td valign="top" align="left">0.164</td>
<td valign="top" align="left">0.047</td>
</tr>
<tr>
<td valign="top" align="left">Intercept</td>
<td valign="top" align="left">3.5442</td>
<td valign="top" align="left">0.117</td>
<td valign="top" align="left">0.000</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The compounds used and their pMIC values can be found in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> and the descriptor meaning can be found in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1</bold>
</xref>. N, number of compounds; R<sup>2</sup>, coefficient of determination.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Predicted vs actual pMIC regression from the OLS QSAR model for the B-Meta dataset, whose parameters can be found in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>. The shaded region represents the 95% confidence interval for the regression.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="finmi-02-1404729-g004.tif"/>
</fig>
<p>Caffeic, ferulic, sinapic, and p-coumaric acid were the only compounds with an aliphatic carboxylic acid group present in this dataset and they had the lowest observed pMIC values (~3.3). They also represent hydroxycinnamic acid derivatives that are known to have increased antimicrobial properties compared to their more polar hydroxybenzoic acid counterparts (<xref ref-type="bibr" rid="B6">Borges et&#xa0;al., 2013</xref>). This is confirmed here by the fact that gallic and protocatechuic acids, with aromatic carboxylic acid groups, had higher MIC values that corresponds to lower antimicrobial activity. Previous work has suggested that hydroxycinnamic acid&#x2019;s propenoic side chain is responsible for its increased antimicrobial properties, as it facilitates the transport of the molecule through the cell membrane of Gram-positive bacteria (<xref ref-type="bibr" rid="B11">Campos et&#xa0;al., 2003</xref>; <xref ref-type="bibr" rid="B27">Nohynek et&#xa0;al., 2006</xref>; <xref ref-type="bibr" rid="B6">Borges et&#xa0;al., 2013</xref>). Therefore, this explains why an increase in aliphatic carboxylic acid groups correlated to an increase in antimicrobial activity (lower pMIC) for this dataset. Nonetheless, the B-Meta dataset only represents a very small number of lignin monomers and more compounds need to be examined to truly understand or predict the properties that influence their antimicrobial activity.</p>
</sec>
<sec id="s3_3">
<title>Experimental dataset models</title>
<p>The antibacterial activity of 25 lignin derived monomers and three relevant dimers were assessed by measuring their BLD or percent inhibition against <italic>B. subtilis</italic> and <italic>E. coli</italic> at concentrations of 1 g/L. The BLD values are presented in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>, and they ranged from 2% up to 100%, indicating compounds can be completely inhibitory to both organisms. The 3-hydroxy-2-(2-methoxyphenoxy)-1-(4-methoxyphenyl)-1-propanone lignin dimer was the only compound to show complete inhibition against both <italic>B. subtilis</italic> and <italic>E. coli</italic>. Interestingly, the 2-(2-methoxyphenoxy)-1-(4-methoxyphenyl)ethanol lignin dimer only had a BLD value of 66% and 55% for both <italic>B. subtilis</italic> and <italic>E. coli</italic>, but its chemical structure differs only by an absence of a methoxy group on &#x3b2;-carbon compared to 3-hydroxy-2-(2-methoxyphenoxy)-1-(4-methoxyphenyl)-1-propanone. Therefore, the presence of this one methoxy group seems to increase the molecules BLD by ~34&#x2013;45%. Moreover, we can also see from <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> that that alkyl chains on the phenolic subunit (4-ethylphenol) and the lignin dimers themselves play an important role in these lignin derivatives antimicrobial properties (i.e., higher BLD values). However, the development of a QSAR model for both organisms will provide an actual statistical relationship between these molecules BLD values and descriptors for more predictive purposes.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Experimental antimicrobial activity of lignin monomers and dimers against <italic>B. subtilis</italic> and <italic>E. coli</italic> (BLD %), where experimental values are mean &#xb1; SE (n=3).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Type</th>
<th valign="top" align="left">Compound</th>
<th valign="top" align="left">
<italic>B. subtilis</italic>
<break/>(BLD %)</th>
<th valign="top" align="left">
<italic>E. coli</italic>
<break/>(BLD %)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" rowspan="25" align="left">Monomers</td>
<td valign="top" align="left">2-6-dimethoxyphenol</td>
<td valign="top" align="left">42.44 &#xb1; 6.05</td>
<td valign="top" align="left">53.53 &#xb1; 5.28</td>
</tr>
<tr>
<td valign="top" align="left">4-ethyl phenol</td>
<td valign="top" align="left">62.43 &#xb1; 1.11</td>
<td valign="top" align="left">81.75 &#xb1; 1.30</td>
</tr>
<tr>
<td valign="top" align="left">4-propyl phenol</td>
<td valign="top" align="left">73.34 &#xb1; 0.04</td>
<td valign="top" align="left">100.00 &#xb1; 0.00</td>
</tr>
<tr>
<td valign="top" align="left">acetovanillone</td>
<td valign="top" align="left">46.13 &#xb1; 3.69</td>
<td valign="top" align="left">58.07 &#xb1; 0.20</td>
</tr>
<tr>
<td valign="top" align="left">coniferyl alcohol</td>
<td valign="top" align="left">35.74 &#xb1; 3.22</td>
<td valign="top" align="left">40.85 &#xb1; 2.48</td>
</tr>
<tr>
<td valign="top" align="left">coniferyl aldehyde</td>
<td valign="top" align="left">36.89 &#xb1; 13.35</td>
<td valign="top" align="left">81.76 &#xb1; 0.88</td>
</tr>
<tr>
<td valign="top" align="left">ethyl 3,4 hydroxy propionate</td>
<td valign="top" align="left">64.33 &#xb1; 0.60</td>
<td valign="top" align="left">81.03 &#xb1; 2.57</td>
</tr>
<tr>
<td valign="top" align="left">eugenol</td>
<td valign="top" align="left">60.77 &#xb1; 2.27</td>
<td valign="top" align="left">60.06 &#xb1; 2.06</td>
</tr>
<tr>
<td valign="top" align="left">ferulic acid</td>
<td valign="top" align="left">36.89 &#xb1; 13.35</td>
<td valign="top" align="left">29.86 &#xb1; 3.42</td>
</tr>
<tr>
<td valign="top" align="left">gallic acid</td>
<td valign="top" align="left">31.13 &#xb1; 0.54</td>
<td valign="top" align="left">4.75 &#xb1; 3.01</td>
</tr>
<tr>
<td valign="top" align="left">guaiacol</td>
<td valign="top" align="left">23.24 &#xb1; 2.10</td>
<td valign="top" align="left">35.18 &#xb1; 4.34</td>
</tr>
<tr>
<td valign="top" align="left">homosyringic acid</td>
<td valign="top" align="left">29.94 &#xb1; 3.81</td>
<td valign="top" align="left">3.68 &#xb1; 8.23</td>
</tr>
<tr>
<td valign="top" align="left">homovanillic acid</td>
<td valign="top" align="left">37.73 &#xb1; 2.09</td>
<td valign="top" align="left">2.13 &#xb1; 1.23</td>
</tr>
<tr>
<td valign="top" align="left">hydroquinone</td>
<td valign="top" align="left">35.06 &#xb1; 0.73</td>
<td valign="top" align="left">8.25 &#xb1; 5.05</td>
</tr>
<tr>
<td valign="top" align="left">p-coumaric acid</td>
<td valign="top" align="left">46.05 &#xb1; 3.60</td>
<td valign="top" align="left">76.88 &#xb1; 0.65</td>
</tr>
<tr>
<td valign="top" align="left">p-coumaryl alcohol</td>
<td valign="top" align="left">43.51 &#xb1; 5.88</td>
<td valign="top" align="left">71.74 &#xb1; 2.87</td>
</tr>
<tr>
<td valign="top" align="left">p-creosol</td>
<td valign="top" align="left">64.68 &#xb1; 3.67</td>
<td valign="top" align="left">84.33 &#xb1; 0.55</td>
</tr>
<tr>
<td valign="top" align="left">syringaldehyde</td>
<td valign="top" align="left">44.29 &#xb1; 4.75</td>
<td valign="top" align="left">24.07 &#xb1; 2.08</td>
</tr>
<tr>
<td valign="top" align="left">syringic acid</td>
<td valign="top" align="left">26.64 &#xb1; 1.88</td>
<td valign="top" align="left">22.27 &#xb1; 4.51</td>
</tr>
<tr>
<td valign="top" align="left">syringyl alcohol</td>
<td valign="top" align="left">37.86 &#xb1; 3.41</td>
<td valign="top" align="left">19.77 &#xb1; 4.49</td>
</tr>
<tr>
<td valign="top" align="left">syringyl propane</td>
<td valign="top" align="left">48.07 &#xb1; 0.43</td>
<td valign="top" align="left">52.14 &#xb1; 3.71</td>
</tr>
<tr>
<td valign="top" align="left">vanillic acid</td>
<td valign="top" align="left">43.82 &#xb1; 4.09</td>
<td valign="top" align="left">37.40 &#xb1; 7.46</td>
</tr>
<tr>
<td valign="top" align="left">vanillin</td>
<td valign="top" align="left">16.10 &#xb1; 3.86</td>
<td valign="top" align="left">21.02 &#xb1; 10.78</td>
</tr>
<tr>
<td valign="top" align="left">protocatechuic acid</td>
<td valign="top" align="left">10.08 &#xb1; 2.36</td>
<td valign="top" align="left">28.02 &#xb1; 2.09</td>
</tr>
<tr>
<td valign="top" align="left">Catechol</td>
<td valign="top" align="left">19.22 &#xb1; 6.99</td>
<td valign="top" align="left">16.39 &#xb1; 17.45</td>
</tr>
<tr>
<td valign="top" rowspan="3" align="left">Dimers</td>
<td valign="top" align="left">2-(2-methoxyphenoxy)-1-(4-methoxyphenyl)ethanol</td>
<td valign="top" align="left">66.00 &#xb1; 13.79</td>
<td valign="top" align="left">55.93 &#xb1; 1.08</td>
</tr>
<tr>
<td valign="top" align="left">3-hydroxy-2-(2-methoxyphenoxy)-1-(4-methoxyphenyl)-1-propanone</td>
<td valign="top" align="left">100.00 &#xb1; 0.00</td>
<td valign="top" align="left">100.00 &#xb1; 0.00</td>
</tr>
<tr>
<td valign="top" align="left">Guaiacylglycerol-beta-guaiacyl ether</td>
<td valign="top" align="left">30.97 &#xb1; 1.03</td>
<td valign="top" align="left">18.55 &#xb1; 4.12</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The same methods used to the develop the QSAR models for the B-Meta dataset were used for both the B-Experimental and E-Experimental datasets. Where RDKit&#x2019;s calculated descriptors were chosen based on univariate feature selection (r&gt;0.5) and an OLS regression with backwards elimination was performed until all descriptors had a <italic>p</italic>-value less than 0.05. The best fitting OLS regressions are summarized in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref> and the predicted vs actual BLD values are plotted in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>. Firstly, the B-Experimental dataset&#x2019;s OLS regression had greater predictive power with a R<sup>2</sup> of 0.831 than that of the B-Meta dataset. Four descriptors were used to develop the best fitting OLS regression model: MinABSEStateIndex, PEOE_VSA13, VSA_EState8, and fr_Ar_OH. As stated previously, PEOE_VSA13 and VSA_EState8 are MOE-type descriptors that are intended to be used as model predictors and are not as interpretable for describing the compounds mechanism of action (<xref ref-type="bibr" rid="B22">Labute, 2000</xref>). The MinABSEStateIndex is the minimum absolute electrotopological state (E-state) of a skeletal atom, formulated as an intrinsic value plus a perturbation term arising from the electronic interaction and modified by the molecular topological environment of each atom in the molecule (<xref ref-type="bibr" rid="B18">Hall et&#xa0;al., 1991</xref>). This descriptor, like the MOE-type descriptors, is used as more of a classification tool for identifying similar compounds instead of describing a feature that could relate to the compounds mode of action. Therefore, while the MinABSEStateIndex, PEOE_VSA13 and VSA_EState8 descriptors show a positive relationship to the lignin compound&#x2019;s BLD value against <italic>B. subtilis</italic>. Comparatively, fr_Ar_OH represents the number of aromatic hydroxyl groups in each compound and is better at elucidating their potential antibacterial mechanisms.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Statistical performance of the best OLS models obtained through backwards elimination of descriptors, for predicting BLD (%) values of lignin phenolics against <italic>B. subtilis</italic> in the B-Experimental dataset and <italic>E. coli</italic> in the E-Experimental dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Dataset</th>
<th valign="top" align="left">N</th>
<th valign="top" align="left">R<sup>2</sup>
</th>
<th valign="top" align="left">Descriptor</th>
<th valign="top" align="left">Coefficient</th>
<th valign="top" align="left">Standard Error</th>
<th valign="top" align="left">
<italic>p</italic>-value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="5" align="left">B-Experimental</td>
<td valign="middle" rowspan="5" align="left">28</td>
<td valign="middle" rowspan="5" align="left">0.831</td>
<td valign="top" align="left">MinABSEStateIndex</td>
<td valign="top" align="left">24.7939</td>
<td valign="top" align="left">7.229</td>
<td valign="top" align="left">0.002</td>
</tr>
<tr>
<td valign="top" align="left">PEOE_VSA13</td>
<td valign="top" align="left">32.0858</td>
<td valign="top" align="left">11.202</td>
<td valign="top" align="left">0.009</td>
</tr>
<tr>
<td valign="top" align="left">VSA_EState8</td>
<td valign="top" align="left">25.1929</td>
<td valign="top" align="left">7.769</td>
<td valign="top" align="left">0.004</td>
</tr>
<tr>
<td valign="top" align="left">fr_Ar_OH</td>
<td valign="top" align="left">&#x2212;43.8297</td>
<td valign="top" align="left">10.158</td>
<td valign="top" align="left">0.000</td>
</tr>
<tr>
<td valign="top" align="left">Intercept</td>
<td valign="top" align="left">43.9883</td>
<td valign="top" align="left">4.545</td>
<td valign="top" align="left">0.000</td>
</tr>
<tr>
<td valign="middle" rowspan="4" align="left">E-Experimental</td>
<td valign="middle" rowspan="4" align="left">28</td>
<td valign="middle" rowspan="4" align="left">0.682</td>
<td valign="top" align="left">EState_VSA6</td>
<td valign="top" align="left">1.1295</td>
<td valign="top" align="left">0.279</td>
<td valign="top" align="left">0.000</td>
</tr>
<tr>
<td valign="top" align="left">VSA_EState3</td>
<td valign="top" align="left">&#x2212;1.5204</td>
<td valign="top" align="left">0.567</td>
<td valign="top" align="left">0.013</td>
</tr>
<tr>
<td valign="top" align="left">fr_aryl_methyl</td>
<td valign="top" align="left">30.5588</td>
<td valign="top" align="left">9.770</td>
<td valign="top" align="left">0.005</td>
</tr>
<tr>
<td valign="top" align="left">Intercept</td>
<td valign="top" align="left">47.8742</td>
<td valign="top" align="left">11.123</td>
<td valign="top" align="left">0.000</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The compounds used and their BLD values can be found in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> and the descriptor meaning can be found in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1</bold>
</xref>. N, number of compounds; R<sup>2</sup>, coefficient of determination.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Predicted vs actual BLD (%) regression from the OLS QSAR model for the B-Experimental <bold>(A)</bold> and E-Experimental <bold>(B)</bold> datasets, whose parameters can be found in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>. The shaded region represents the 95% confidence interval for the regression.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="finmi-02-1404729-g005.tif"/>
</fig>
<p>The number of aromatic hydroxyl groups can be seen to have a negative relationship with BLD for the <italic>B. subtilis</italic> data (<xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>). Where gallic acid, catechol, and protocatechuic acid had more than one aromatic hydroxyl group and the lowest BLD values compared to all the other compounds. So, with an increase in the number of aromatic hydroxyl groups there will be a decrease in BLD, correlating to a decrease in the compound&#x2019;s antibacterial properties against <italic>B. subtilis</italic>. <xref ref-type="bibr" rid="B7">Bouarab-Chibane et al. (2019a)</xref> found a negative relationship between the number of hydrogen donors and the BLD of plant-based polyphenols screened against <italic>B. subtilis</italic>. Since the number of aromatic hydroxyl groups and the number of hydrogen donors have a direct positive relationship (<xref ref-type="bibr" rid="B7">Bouarab-Chibane et&#xa0;al., 2019a</xref>), we can see that in general phenolics with higher overall polarity will have a decrease in antimicrobial properties. This is supported by the experimental data seen here, where highly lipophilic compounds like 4-ethylphenol had high BLD values. However, this model does not provide an explanation for the 3-hydroxy-2-(2-methoxyphenoxy)-1-(4-methoxyphenyl)-1-propanone lignin dimers high BLD value compared to the monomers, highlighting the issue QSAR models can have with limited data sizes and breadth of compound variability.</p>
<p>When examining the OLS regression for <italic>E. coli</italic>, we can see that the EState_VSA6, VSA_EState3, and fr_aryl_methyl descriptors were used to create the best fitting model. Again, EState_VSA6 and VSA_EState3 are MOE type descriptors that are used for classification-based purposes and cannot be used to infer an influence on the molecules antimicrobial activity. Comparatively, the fr_aryl_methyl descriptor represents the number of aryl methyl groups or an alkyl chain (i.e., ethyl or propyl) on the molecule and it shows a positive correlation with BLD. Thus, as the number of methyl or alkyl chain groups on the phenolic ring increase there is an increase in the BLD or antimicrobial activity of the compounds against <italic>E. coli</italic> (<xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>). The compounds described by the fr_aryl_methyl descriptor such as 4-ethyl phenol, 4-propyl phenol, and p-creosol are seen to have BLD values of 80&#x2013;100% for <italic>E. coli</italic> (<xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>). These compounds are also considered to be more lipophilic given their alkyl groups on the phenolic ring, of which, lipophilicity is already considered an important factor in increasing antibacterial activity against <italic>E. coli</italic> in the literature (<xref ref-type="bibr" rid="B35">Sikkema et&#xa0;al., 1995</xref>; <xref ref-type="bibr" rid="B8">Bouarab-Chibane et&#xa0;al., 2019b</xref>). Furthermore, since Gram-negative bacteria have a substantially higher lipid content in their cell wall compared to Gram-positive bacteria (<xref ref-type="bibr" rid="B31">Salton, 1953</xref>), it makes sense that the alkylated phenolics were seen to have greater BLD values for <italic>E. coli</italic> compared to <italic>B. subtilis</italic>. Additionally, like the <italic>B. subtilis</italic> OLS regression, <italic>E. coli&#x2019;s</italic> model also did not provide inferences as to why the lignin dimer 3-hydroxy-2-(2-methoxyphenoxy)-1-(4-methoxyphenyl)-1-propanone had such a high BLD value.</p>
<p>Overall, when comparing the results from the QSAR models for the meta-analysis and experimental datasets, we can see that the presence of certain compounds and how antimicrobial activity was measured will influence which descriptors play the most important role in describing antimicrobial activity. We saw that the hydroxycinnamic derivatives in the B-Meta dataset drove the negative relationship between the number of aliphatic carboxylic acid groups and pMIC. Comparatively, a higher number of aromatic hydroxyl groups were shown to decrease the BLD in the B-Experimental dataset. Additionally, the E-Meta dataset did not provide an accurate OLS regression model, while E-Experimental&#x2019;s model showed increasing alkyl groups on the phenolic ring increases BLD values against <italic>E. coli</italic>. This emphasizes the fact that using different measures of antimicrobial properties and different lignin compounds to develop QSARs for predicting the antimicrobial properties of lignin may lead to different conclusions. It is important to understand the origin of the strains and the cultivation parameters. Meanwhile one could introduce reference compounds across datasets to ensure a more reproducible response. While this is intuitive, the data here provide support for the need in developing a comprehensive and cohesive dataset with lignin derivatives and their antimicrobial properties. Without such a dataset, the ability to accurately predict the antimicrobial potential of lignin and the variety of derivatives that are produced from depolymerization schemes for biorefinery waste stream &#x2013; lignin valorization is limited.</p>
</sec>
</sec>
<sec id="s4" sec-type="conclusions">
<title>Conclusions</title>
<p>Based on meta-analysis, MOE-type descriptors and the number of aliphatic carboxylic acid groups were the descriptors that showed strong correlations to the pMIC values. Comparatively, experimentally based QSAR found that MOE-type descriptors and the number of aromatic hydroxyl groups were better predictors of BLD for <italic>B. subtilis</italic>, while MOE-type descriptors and the number of aryl methyl groups were predictors of BLD for <italic>E. coli</italic>. This study represents one of the first steps towards expediting the search for highly active lignin derivatives that can be produced from depolymerization reactions for valorizing lignin into a sustainable source of antimicrobial compounds.</p>
</sec>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are publicly available. This data can be found here: Harvard Dataverse repository, <uri xlink:href="https://doi.org/10.7910/DVN/LYX4TS">https://doi.org/10.7910/DVN/LYX4TS</uri>.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>RK: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Validation, Visualization, Writing &#x2013; original draft. QS: Conceptualization, Methodology, Resources, Supervision, Writing &#x2013; review &amp; editing. JS: Conceptualization, Funding acquisition, Project administration, Supervision, Writing &#x2013; review &amp; editing.</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by the National Science Foundation under Cooperative Agreement No. 1355438 and 1632854. In addition, this work was also supported by the National Institute of Food and Agriculture, U.S. Department of Agriculture, Hatch-Multistate project under accession number 1018315.</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/finmi.2024.1404729/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/finmi.2024.1404729/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet_1.pdf" id="SM1" mimetype="application/pdf"/>
<supplementary-material xlink:href="DataSheet_2.pdf" id="SM2" mimetype="application/pdf"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andrews</surname> <given-names>J. M.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Determination of minimum inhibitory concentrations</article-title>. <source>J. Antimicrob. Chemother.</source> <volume>48 Suppl 1</volume>, <fpage>5</fpage>&#x2013;<lpage>16</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/jac/48.suppl_1.5</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Araya-Cloutier</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Vincken</surname> <given-names>J.-P.</given-names>
</name>
<name>
<surname>van de Schans</surname> <given-names>M. G. M.</given-names>
</name>
<name>
<surname>Hageman</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Schaftenaar</surname> <given-names>G.</given-names>
</name>
<name>
<surname>den Besten</surname> <given-names>H. M. W.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>QSAR-based molecular signatures of prenylated (iso)flavonoids underlying antimicrobial potency against and membrane-disruption in Gram positive and Gram negative bacteria</article-title>. <source>Sci. Rep.</source> <volume>8</volume>, <fpage>9267</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-018-27545-4</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barber</surname> <given-names>M. S.</given-names>
</name>
<name>
<surname>McConnell</surname> <given-names>V. S.</given-names>
</name>
<name>
<surname>DeCaux</surname> <given-names>B. S.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Antimicrobial intermediates of the general phenylpropanoid and lignin specific pathways</article-title>. <source>Phytochemistry</source> <volume>54</volume>, <fpage>53</fpage>&#x2013;<lpage>56</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S0031-9422(00)00038-8</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Basant</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Gupta</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>K. P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>QSAR modeling for predicting reproductive toxicity of chemicals in rats for regulatory purposes</article-title>. <source>Toxicol. Res.</source> <volume>5</volume>, <fpage>1029</fpage>&#x2013;<lpage>1038</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1039/C6TX00083E</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bengio</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Goodfellow</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Courville</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <source>Deep learning.</source> (<publisher-loc>Massachusetts, USA</publisher-loc>: <publisher-name>MIT press</publisher-name>).</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Borges</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Ferreira</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Saavedra</surname> <given-names>M. J.</given-names>
</name>
<name>
<surname>Sim&#xf5;es</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Antibacterial activity and mode of action of ferulic and gallic acids against pathogenic bacteria</article-title>. <source>Microb. Drug Resist.</source> <volume>19</volume>, <fpage>256</fpage>&#x2013;<lpage>265</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1089/mdr.2012.0244</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bouarab-Chibane</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Forquet</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Lant&#xe9;ri</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Cl&#xe9;ment</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>L&#xe9;onard-Akkari</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Oulahal</surname> <given-names>N.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>a). <article-title>Antibacterial properties of polyphenols: characterization and QSAR (quantitative structure-activity relationship) models</article-title>. <source>Front. Microbiol.</source> <volume>10</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fmicb.2019.00829</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bouarab-Chibane</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Forquet</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Lant&#xe9;ri</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Cl&#xe9;ment</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>L&#xe9;onard-Akkari</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Oulahal</surname> <given-names>N.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>b). <article-title>Antibacterial properties of polyphenols: characterization and QSAR (Quantitative structure&#x2013;activity relationship) models</article-title>. <source>Front. Microbiol.</source> <volume>10</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fmicb.2019.00829</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Calvo-Flores</surname> <given-names>F. G.</given-names>
</name>
<name>
<surname>Dobado</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Isac-Garc&#xed;a</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Mart&#xed;n-Mart&#xed;Nez</surname> <given-names>F. J.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Applications of modified and unmodified lignins</article-title>,&#x201d; in <source>Lignin and lignans as renewable raw materials.</source> (<publisher-loc>Hoboken, NJ, USA</publisher-loc>: <publisher-name>John Wiley &amp; Sons, Ltd</publisher-name>), <fpage>247</fpage>&#x2013;<lpage>288</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Camacho</surname> <given-names>D. M.</given-names>
</name>
<name>
<surname>Collins</surname> <given-names>K. M.</given-names>
</name>
<name>
<surname>Powers</surname> <given-names>R. K.</given-names>
</name>
<name>
<surname>Costello</surname> <given-names>J. C.</given-names>
</name>
<name>
<surname>Collins</surname> <given-names>J. J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Next-generation machine learning for biological networks</article-title>. <source>Cell</source> <volume>173</volume>, <fpage>1581</fpage>&#x2013;<lpage>1592</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cell.2018.05.015</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Campos</surname> <given-names>F. M.</given-names>
</name>
<name>
<surname>Couto</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Hogg</surname> <given-names>T. A.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Influence of phenolic acids on growth and inactivation of Oenococcus oeni and Lactobacillus hilgardii</article-title>. <source>J. Appl. Microbiol.</source> <volume>94</volume>, <fpage>167</fpage>&#x2013;<lpage>174</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1046/j.1365-2672.2003.01801.x</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cazacu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Capraru</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Popa</surname> <given-names>V. I.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Advances concerning lignin utilization in new materials</article-title>,&#x201d; in <source>Advances in natural polymers.</source> (<publisher-name>Springer</publisher-name>, <publisher-loc>Berlin, Heidelberg</publisher-loc>), <fpage>255</fpage>&#x2013;<lpage>312</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>de Souza Araujo</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Pimenta</surname> <given-names>A. S.</given-names>
</name>
<name>
<surname>Feijo</surname> <given-names>F. M. C.</given-names>
</name>
<name>
<surname>Castro</surname> <given-names>R. V. O.</given-names>
</name>
<name>
<surname>Fasciotti</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Monteiro</surname> <given-names>T. V. C.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>Antibacterial and antifungal activities of pyroligneous acid from wood of Eucalyptus urograndis and Mimosa tenuiflora</article-title>. <source>J. Appl. Microbiol.</source> <volume>124</volume>, <fpage>85</fpage>&#x2013;<lpage>96</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/jam.13626</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Turley</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Antimicrobial and antioxidant activities of lignin from residue of corn stover to ethanol production</article-title>. <source>Ind. Crops Products</source> <volume>34</volume>, <fpage>1629</fpage>&#x2013;<lpage>1634</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.indcrop.2011.06.002</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Espinoza-Acosta</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Torres-Ch&#xe1;vez</surname> <given-names>P. I.</given-names>
</name>
<name>
<surname>Ram&#xed;rez-Wong</surname> <given-names>B.</given-names>
</name>
<name>
<surname>L&#xf3;pez-Saiz</surname> <given-names>C. M.</given-names>
</name>
<name>
<surname>Monta&#xf1;o-Leyva</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Antioxidant, antimicrobial, and antimutagenic properties of technical lignins and their applications</article-title>. <source>BioResources</source> <volume>11</volume>, <fpage>5452</fpage>&#x2013;<lpage>5481</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.15376/biores</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gaulton</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Bellis</surname> <given-names>L. J.</given-names>
</name>
<name>
<surname>Bento</surname> <given-names>A. P.</given-names>
</name>
<name>
<surname>Chambers</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Davies</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Hersey</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2012</year>). <article-title>ChEMBL: a large-scale bioactivity database for drug discovery</article-title>. <source>Nucleic Acids Res.</source> <volume>40</volume>, <fpage>D1100</fpage>&#x2013;<lpage>D1107</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkr777</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Golbraikh</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>Y.-D.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>K.-H.</given-names>
</name>
<name>
<surname>Tropsha</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Rational selection of training and test sets for the development of validated QSAR models</article-title>. <source>J. Computer-Aided Mol. Design</source> <volume>17</volume>, <fpage>241</fpage>&#x2013;<lpage>253</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1023/A:1025386326946</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hall</surname> <given-names>L. H.</given-names>
</name>
<name>
<surname>Mohney</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Kier</surname> <given-names>L. B.</given-names>
</name>
</person-group> (<year>1991</year>). <article-title>The electrotopological state: structure information at the atomic level for molecular graphs</article-title>. <source>J. Chem. Inf. Comput. Sci.</source> <volume>31</volume>, <fpage>76</fpage>&#x2013;<lpage>82</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/ci00001a012</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Harvey</surname> <given-names>A. L.</given-names>
</name>
<name>
<surname>Edrada-Ebel</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Quinn</surname> <given-names>R. J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>The re-emergence of natural products for drug discovery in the genomics era</article-title>. <source>Nat. Rev. Drug Discovery</source> <volume>14</volume>, <fpage>111</fpage>&#x2013;<lpage>129</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nrd4510</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hira</surname> <given-names>Z. M.</given-names>
</name>
<name>
<surname>Gillies</surname> <given-names>D. F.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>A review of feature selection and feature extraction methods applied on microarray data</article-title>. <source>Adv. Bioinf.</source> <volume>282</volume>, <fpage>111</fpage>&#x2013;<lpage>135</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1155/2015/198363</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hwang</surname> <given-names>Y. H.</given-names>
</name>
<name>
<surname>Matsushita</surname> <given-names>Y. I.</given-names>
</name>
<name>
<surname>Sugamoto</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Matsui</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Antimicrobial effect of the wood vinegar from Cryptomeria japonica sapwood on plant pathogenic microorganisms</article-title>. <source>J. Microbiol. Biotechnol.</source> <volume>15</volume>(<issue>5</issue>), <fpage>1106</fpage>&#x2013;<lpage>1109</lpage>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Labute</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>A widely applicable set of descriptors</article-title>. <source>J. Mol. Graphics Model.</source> <volume>18</volume>, <fpage>464</fpage>&#x2013;<lpage>477</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S1093-3263(00)00068-1</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lenselink</surname> <given-names>E. B.</given-names>
</name>
<name>
<surname>ten Dijke</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Bongers</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Papadatos</surname> <given-names>G.</given-names>
</name>
<name>
<surname>van Vlijmen</surname> <given-names>H. W. T.</given-names>
</name>
<name>
<surname>Kowalczyk</surname> <given-names>W.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). <article-title>Beyond the hype: deep neural networks outperform established methods using a ChEMBL bioactivity benchmark set</article-title>. <source>J. Cheminformatics</source> <volume>9</volume>, <fpage>45</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13321-017-0232-0</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Louren&#xe7;on</surname> <given-names>T. V.</given-names>
</name>
<name>
<surname>Mattos</surname> <given-names>B. D.</given-names>
</name>
<name>
<surname>Cademartori</surname> <given-names>P. H. G.</given-names>
</name>
<name>
<surname>Magalh&#xe3;es</surname> <given-names>W. L. E.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Bio-oil from a fast pyrolysis pilot plant as antifungal and hydrophobic agent for wood preservation</article-title>. <source>J. Analytical Appl. Pyrolysis</source> <volume>122</volume>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jaap.2016.11.004</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mathew</surname> <given-names>A. K.</given-names>
</name>
<name>
<surname>Abraham</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Mallapureddy</surname> <given-names>K. K.</given-names>
</name>
<name>
<surname>Sukumaran</surname> <given-names>R. K.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Chapter 9 - lignocellulosic biorefinery wastes, or resources</article-title>?,&#x201d; in <source>
<italic>Waste biorefinery</italic>,</source>. Eds. <person-group person-group-type="editor">
<name>
<surname>Bhaskar</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Pandey</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Mohan</surname> <given-names>S. V.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>D.-J.</given-names>
</name>
<name>
<surname>Khanal.</surname> <given-names>S. K.</given-names>
</name>
</person-group> (<publisher-loc>B.V., Amsterdam, Netherlands</publisher-loc>: <publisher-name>Elsevier</publisher-name>), <fpage>267</fpage>&#x2013;<lpage>297</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mei</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Support vector machine applied in QSAR modelling</article-title>. <source>Chin. Sci. Bull.</source> <volume>50</volume>, <fpage>2291</fpage>&#x2013;<lpage>2296</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/BF03183737</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nohynek</surname> <given-names>L. J.</given-names>
</name>
<name>
<surname>Alakomi</surname> <given-names>H. L.</given-names>
</name>
<name>
<surname>K&#xe4;hk&#xf6;nen</surname> <given-names>M. P.</given-names>
</name>
<name>
<surname>Heinonen</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Helander</surname> <given-names>I. M.</given-names>
</name>
<name>
<surname>Oksman-Caldentey</surname> <given-names>K. M.</given-names>
</name>
<etal/>
</person-group>. (<year>2006</year>). <article-title>Berry phenolics: antimicrobial properties and mechanisms of action against severe human pathogens</article-title>. <source>Nutr. Cancer</source> <volume>54</volume>, <fpage>18</fpage>&#x2013;<lpage>32</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1207/s15327914nc5401_4</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pandey</surname> <given-names>M. P.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>C. S.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Lignin depolymerization and conversion: A review of thermochemical methods</article-title>. <source>Chem. Eng. Technol.</source> <volume>34</volume>, <fpage>29</fpage>&#x2013;<lpage>41</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/ceat.201000270</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Patra</surname> <given-names>J. K.</given-names>
</name>
<name>
<surname>Hwang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Choi</surname> <given-names>J. W.</given-names>
</name>
<name>
<surname>Baek</surname> <given-names>K. H.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Bactericidal Mechanism of Bio-oil Obtained from Fast Pyrolysis of Pinus densiflora Against Two Foodborne Pathogens, Bacillus cereus and Listeria monocytogenes</article-title>. <source>Foodborne Pathog. Dis.</source> <volume>12</volume>, <fpage>529</fpage>&#x2013;<lpage>535</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1089/fpd.2014.1914</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedregosa</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Varoquaux</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Gramfort</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Michel</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Thirion</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Grisel</surname> <given-names>O.</given-names>
</name>
<etal/>
</person-group>. (<year>2011</year>). <article-title>Scikit-learn: machine learning in python</article-title>. <source>J. Mach. Learn. Res.</source> <volume>12</volume>, <fpage>2825</fpage>&#x2013;<lpage>2830</lpage>. doi: <pub-id pub-id-type="doi">10.5555/1953048.2078195</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Salton</surname> <given-names>M. R.</given-names>
</name>
</person-group> (<year>1953</year>). <article-title>Studies of the bacterial cell wall. IV. The composition of the cell walls of some Gram-positive and Gram-negative bacteria</article-title>. <source>Biochim. Biophys. Acta</source> <volume>10</volume>, <fpage>512</fpage>&#x2013;<lpage>523</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/0006-3002(53)90296-0</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sari</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Khatab</surname> <given-names>U.</given-names>
</name>
<name>
<surname>Burmawi</surname>
</name>
<name>
<surname>Rahman</surname> <given-names>E. D.</given-names>
</name>
<name>
<surname>Afriza</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Maulidita</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Production of liquid smoke from the process of carbonization of durian skin biomass, coconut shell and palm shell for preservation of tilapia fish</article-title>. <source>IOP Conf. Series: Materials Sci. Eng.</source> <volume>543</volume>, <elocation-id>12075</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1088/1757-899X/543/1/012075</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Seabold</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Perktold</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>Statsmodels: Econometric and statistical modeling with python</article-title>,&#x201d; in <source>Proceedings of the 9th python in science conference</source>(<publisher-loc>Austin, TX</publisher-loc>), <fpage>61</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.25080/issn.2575-9752</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shahlaei</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Descriptor selection methods in quantitative structure&#x2013;activity relationship studies: A review study</article-title>. <source>Chem. Rev.</source> <volume>113</volume>, <fpage>8093</fpage>&#x2013;<lpage>8103</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/cr3004339</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sikkema</surname> <given-names>J.</given-names>
</name>
<name>
<surname>de Bont</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Poolman</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>1995</year>). <article-title>Mechanisms of membrane toxicity of hydrocarbons</article-title>. <source>Microbiological Rev.</source> <volume>59</volume>, <fpage>201</fpage>&#x2013;<lpage>222</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1128/mr.59.2.201-222.1995</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Mobley</surname> <given-names>J. K.</given-names>
</name>
<name>
<surname>Motagamwala</surname> <given-names>A. H.</given-names>
</name>
<name>
<surname>Isaacs</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Dumesic</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Ralph</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>Gold-catalyzed conversion of lignin to low molecular weight aromatics</article-title>. <source>Chem. Sci.</source> <volume>9</volume>, <fpage>8127</fpage>&#x2013;<lpage>8133</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1039/C8SC03208D</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Speck-Planche</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Kleandrova</surname> <given-names>V. V.</given-names>
</name>
<name>
<surname>Luan</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Cordeiro</surname> <given-names>M. N.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Predicting multiple ecotoxicological profiles in agrochemical fungicides: a multi-species chemoinformatic approach</article-title>. <source>Ecotoxicol Environ. Saf.</source> <volume>80</volume>, <fpage>308</fpage>&#x2013;<lpage>313</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoenv.2012.03.018</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Fridrich</surname> <given-names>B.</given-names>
</name>
<name>
<surname>de Santi</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Elangovan</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Barta</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Bright side of lignin depolymerization: toward new platform chemicals</article-title>. <source>Chem. Rev.</source> <volume>118</volume>, <fpage>614</fpage>&#x2013;<lpage>678</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/acs.chemrev.7b00588</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Svensson</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Norinder</surname> <given-names>U.</given-names>
</name>
<name>
<surname>Bender</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Modelling compound cytotoxicity using conformal prediction and PubChem HTS data</article-title>. <source>Toxicol. Res.</source> <volume>6 1</volume>, <fpage>73</fpage>&#x2013;<lpage>80</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1039/C6TX00252H</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Svetnik</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Liaw</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Tong</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Culberson</surname> <given-names>J. C.</given-names>
</name>
<name>
<surname>Sheridan</surname> <given-names>R. P.</given-names>
</name>
<name>
<surname>Feuston</surname> <given-names>B. P.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Random forest:&#x2009; A classification and regression tool for compound classification and QSAR modeling</article-title>. <source>J. Chem. Inf. Comput. Sci.</source> <volume>43</volume>, <fpage>1947</fpage>&#x2013;<lpage>1958</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/ci034160g</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Telysheva</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Dizhbite</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Lebedeva</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Niokolaeva</surname> <given-names>V.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Lignin products for decontamination of environment objects from pathogenic microorganisms and pollutants</article-title>. <source>Proc. 7th ILI Forum Barcelona Spain</source>, <fpage>71</fpage>&#x2013;<lpage>74</lpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tsai</surname> <given-names>W.-P.</given-names>
</name>
<name>
<surname>Fang</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Ji</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Lawson</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Revealing causal controls of storage-streamflow relationships with a data-centric bayesian framework combining machine learning and process-based modeling</article-title>. <source>Front. Water</source> <volume>2</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/frwa.2020.583000</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Upadhyay</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Upadhyaya</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Kollanoor-Johny</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Venkitanarayanan</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Combating pathogenic microorganisms using plant-derived antimicrobials: A minireview of the mechanistic basis</article-title>. <source>BioMed. Res. Int.</source> <volume>2014</volume>, <fpage>761741</fpage>. doi: <pub-id pub-id-type="doi">10.1155/2014/761741</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vamathevan</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Clark</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Czodrowski</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Dunham</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Ferran</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>G.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Applications of machine learning in drug discovery and development</article-title>. <source>Nat. Rev. Drug Discovery</source> <volume>18</volume>, <fpage>463</fpage>&#x2013;<lpage>477</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41573-019-0024-5</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Tucker</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Ji</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Recent development in chemical depolymerization of lignin: A review</article-title>. <source>J. Appl. Chem.</source> <volume>2013</volume>, <fpage>9</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1155/2013/838645</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weininger</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Weininger</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Weininger</surname> <given-names>J. L.</given-names>
</name>
</person-group> (<year>1989</year>). <article-title>SMILES. 2. Algorithm for generation of unique SMILES notation</article-title>. <source>J. Chem. Inf. Comput. Sci.</source> <volume>29</volume>, <fpage>97</fpage>&#x2013;<lpage>101</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/ci00062a008</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>WHO</surname> <given-names>W. H. O.</given-names>
</name>
</person-group> (<year>2014</year>). <source>Antimicrobial resistance: global report on surveillance</source> (<publisher-loc>Geneva, Switzerland</publisher-loc>: <publisher-name>World Health Organization</publisher-name>).</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Fortunati</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Balestra</surname> <given-names>G. M.</given-names>
</name>
<name>
<surname>Giovanale</surname> <given-names>G.</given-names>
</name>
<name>
<surname>He</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>Valorization of acid isolated high yield lignin nanoparticles as innovative antioxidant/antimicrobial organic materials</article-title>. <source>ACS Sustain. Chem. Eng.</source> <volume>6</volume>, <fpage>3502</fpage>&#x2013;<lpage>3514</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/acssuschemeng.7b03782</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zemek</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ko&#x161;&#xed;kov&#xe1;</surname> <given-names>B.</given-names>
</name>
<name>
<surname>August&#xed;n</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Joniak</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>1979</year>). <article-title>Antibiotic properties of lignin components</article-title>. <source>Folia Microbiologica</source> <volume>24</volume>, <fpage>483</fpage>&#x2013;<lpage>486</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/BF02927180</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Tropsha</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Novel variable selection quantitative structure&#x2013;property relationship approach based on the k-nearest-neighbor principle</article-title>. <source>J. Chem. Inf. Comput. Sci.</source> <volume>40</volume>, <fpage>185</fpage>&#x2013;<lpage>194</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/ci980033m</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>