<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Toxicol.</journal-id>
<journal-title>Frontiers in Toxicology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Toxicol.</abbrev-journal-title>
<issn pub-type="epub">2673-3080</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1640612</article-id>
<article-id pub-id-type="doi">10.3389/ftox.2025.1640612</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Toxicology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A deep-learning approach to predict reproductive toxicity of chemicals using communicative message passing neural network</article-title>
<alt-title alt-title-type="left-running-head">He et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/ftox.2025.1640612">10.3389/ftox.2025.1640612</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>He</surname>
<given-names>Owen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3086921/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Daoxing</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Li</surname>
<given-names>Yimei</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/635409/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Deerfield Academy</institution>, <addr-line>Deerfield</addr-line>, <addr-line>MA</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Pharmaceutical Sciences</institution>, <institution>Wenzhou Medical University</institution>, <addr-line>Wenzhou</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Biostatistics</institution>, <institution>St. Jude Children&#x2019;s Research Hospital</institution>, <addr-line>Memphis</addr-line>, <addr-line>TN</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/124113/overview">Ruili Huang</ext-link>, National Center for Advancing Translational Sciences (NIH), United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3095005/overview">Xi Luo</ext-link>, National Institutes of Health (NIH), United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3095535/overview">Sohaib Habiballah</ext-link>, Colorado State University, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Owen He, <email>ohe26@deerfield.edu</email>; Yimei Li, <email>yimei.li@stjude.org</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>22</day>
<month>07</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>7</volume>
<elocation-id>1640612</elocation-id>
<history>
<date date-type="received">
<day>04</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>10</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 He, Chen and Li.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>He, Chen and Li</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Reproductive toxicity is a concern critical to human health and chemical safety assessment. Recently, the U.S. Food and Drug Administration announced plans to assess toxicity with artificial intelligence-based computational models instead of animal studies in &#x201c;a win-win for public health and ethics.&#x201d; In this study, we used a reproductive toxicity dataset using Simplified Molecular Input Line Entry Specifications (SMILES) to represent 1091 reproductively toxic and 1063 non-toxic small-molecule compounds. A repeated nested cross-validation procedure was applied, in which the dataset was randomly partitioned into five distinct folds in the outer loop, each time, one fold serving as the test set. In the inner loop, a similar procedure was also repeated five times, with 12.5% each time serving as the validation set. We first evaluated the performance of classical machine learning (ML) methods such as Random Forest and Extreme Gradient Boosting on predicting reproductive toxicity, using standard model evaluation metrics including accuracy score (ACC), the area under the curve (AUC) of the receiver operating characteristics curve (ROC) and F1 score. Our analyses indicate that these methods&#x2019; overall results were mediocre and insufficient for high-throughput screening. To overcome these limitations, we adopted the Communicative Message Passing Neural Network (CMPNN) framework, which incorporates a communicative kernel and a message booster module. Our results show that our ReproTox-CMPNN model outperforms the current best baselines in both embedding quality and predictive accuracy. In independent test sets, ReproTox-CMPNN achieved a mean AUC of 0.946, ACC of 0.857 and F1 score of 0.846, surpassing traditional algorithms to establish itself as a new state-of-the-art model in this field. These findings demonstrate that CMPNN&#x2019;s deep capture of multi-level molecular relationships offers an efficient and reliable computational tool for rapid chemical safety screening and risk assessment.</p>
</abstract>
<kwd-group>
<kwd>reproductive</kwd>
<kwd>artificial intelligence (AI)</kwd>
<kwd>deep learning</kwd>
<kwd>graph neural network</kwd>
<kwd>CMPNN</kwd>
<kwd>
<italic>in silico</italic>
</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Toxicology and Informatics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Reproductive toxicity, referring to the ability to disturb reproductive competence through structural and functional alterations (<xref ref-type="bibr" rid="B40">U.S. Food and Drug Administration, 2011</xref>), remains a concern critical to chemical safety assessment, human health and development of novel drugs. Such toxicity can lead to a wide range of adverse effects (<xref ref-type="bibr" rid="B39">U.S. Environmental Protection Agency, 1996</xref>), including reduced fertility due to hormonal disruptions, decreased sperm count and motility or impaired ovarian reserves, harm to embryonic or fetal development leading to elevated rates of failure to reach term (about 50% for conceptions and 32%&#x2013;34% for early pregnancies), congenital malformations or growth or neurobehavioral disorders like low birth weight, hypospadias or irregular puberty onset. These adverse outcomes arise from diverse mechanisms at multiple stages. Endocrine disruption occurs when toxins such as phthalates mimic or block endogenous hormones, destabilizing endocrine systems; oxidative stress can disrupt cellular signaling and induce apoptosis, impairing spermatogenesis and oocyte maturation. Genotoxic effects on germline or embryonic cells, including mutations and chromosomal abnormalities, elevate risks of developmental defects and early pregnancy loss. During embryogenesis, toxic exposures may interfere with implantation, organ formation, and morphogenesis, culminating in teratogenic outcomes and postnatal developmental deficits. Exposure-triggered epigenetic alterations, such as DNA methylation, histone modifications, and miRNA expression changes, may persist through embryonic development, potentially affecting multiple generations of offspring. During important windows of development like gametogenesis, pregnancy or early childhood, exposure to reproductive toxicants can cause irreversible adverse outcomes (<xref ref-type="bibr" rid="B1">Archibong et al., 2018</xref>; <xref ref-type="bibr" rid="B19">Iavarone and Dasmahapatra, 2025</xref>; <xref ref-type="bibr" rid="B36">Shi et al., 2021</xref>; <xref ref-type="bibr" rid="B47">Yang et al., 2018</xref>).</p>
<p>Through the aforementioned mechanisms, environmental and industrial chemicals from everyday plastics to persistent pollutants have been increasingly implicated in reproductive complications. Phthalates, widely used as plasticizers, interfere with androgen signaling, impairing spermatogenesis and ovary function; in animal studies, exposure was linked to short anogenital distances and reproductive tract malformations. Bisphenol A (BPA), a ubiquitous xenoestrogen, disrupts the hypothalamic-pituitary-gonadal axis, decreasing fertility in both sexes, and impairs embryonic implantation. Pesticides, particularly those with endocrine-modulating or genotoxic properties, are known to delay puberty, reduce gamete quality and increase miscarriage risk. Heavy metals like lead and cadmium induce oxidative stress in gonadal tissues, leading to diminished sperm production, menstrual irregularities, and fetal growth restriction. Per- and polyfluoroalkyl substances (PFAS)&#x2014;persistent &#x201c;forever chemicals&#x201d;&#x2014;cross the placenta, disrupt hormone pathways, impair ovarian and testicular development, and have been associated with reduced birth weight and infertility (<xref ref-type="bibr" rid="B49">Yesildemir and Celik, 2024</xref>; <xref ref-type="bibr" rid="B26">M&#xed;nguez-Alarc&#xf3;n et al., 2023</xref>).</p>
<p>Recent research has shown that many chemicals used in workplaces have not been adequately studied on their possible reproductive toxicity despite previous reports that exposure to these chemicals increases risk of endocrine disruption, impaired fertility and adverse reproductive outcomes (<xref ref-type="bibr" rid="B34">Rim, 2017</xref>). In electronics factories, workers routinely handle solvents such as trichloroethylene (TCE) and perchloroethylene (PCE), volatile organic compounds, as well as phthalate-rich plasticizers and flame retardants like polybrominated diphenyl ethers (PBDEs) to which they are exposed via inhalation and skin contact with cables and casings. In textile manufacturing, workrooms are often laden with bleaching agents, azo dyes, formaldehyde, and heavy metals released as airborne dust or absorbed dermally, leading to potential miscarriages, menstrual disturbances, and hormone disruption. Moreover, in poorly ventilated settings, microfibers and volatile processing chemicals that increase oxidative and endocrine stress can be inhaled (<xref ref-type="bibr" rid="B42">Wang and Qian, 2021</xref>). Even low levels of exposure to toxicants during pregnancy can lead to maternal complications, birth defects and delays or disorders in childhood development (<xref ref-type="bibr" rid="B10">Di Renzo et al., 2015</xref>). A well-known example is thalidomide, which has caused thousands of miscarriages and stillbirths in addition to almost 10,000 severe limb malformations at birth (<xref ref-type="bibr" rid="B25">Miller, 1991</xref>). Therefore, thorough evaluation of reproductive toxicity is required both for public health policy safeguarding current and future generations&#x2019; wellbeing as well as for the development of novel drugs and regulatory requirements.</p>
<p>Besides health impairments to immediately affected individuals, widespread exposure to reproductive toxicants brings substantial increases in healthcare costs and long-term burdens on public health systems. <xref ref-type="bibr" rid="B27">Njagi et al. (2023)</xref> reported that approximately 17.5% of adults globally experience infertility and that a recent Global Burden of Disease analysis estimated over 110 million cases of female infertility in 2021&#x2014;a rise of 84% since 1990. Exposure to endocrine-disrupting chemicals, heavy metals, and persistent organic pollutants has additionally been linked to increased incidence of reproductive cancers, such as testicular and ovarian cancer, and a higher prevalence of birth defects (e.g., neural tube defects, hypospadias, congenital heart anomalies). Growing evidence also supports that prenatal or parental exposure to toxicants like methoxychlor and polychlorinated biphenyls has been linked to impaired fertility and reproductive health across multiple generations (<xref ref-type="bibr" rid="B23">Liu et al., 2025</xref>; <xref ref-type="bibr" rid="B31">Pan et al., 2023</xref>; <xref ref-type="bibr" rid="B6">Brehm and Flaws, 2019</xref>). According to a comprehensive analysis (<xref ref-type="bibr" rid="B2">Attina et al., 2016</xref>), by 2016, the estimated annual cost of healthcare in the United States had amounted to over 340 billion USD, or more than 2.3% of GDP, due to low-level daily exposure to endocrine-disrupting chemicals potentially hazardous to reproduction. Together, these trends underscore an urgent need for improved reproductive health surveillance and regulatory intervention.</p>
<p>With a global cost of approximately 10.6 billion USD in 2022, a figure expected to rise to $25.7 billion by 2032 (<xref ref-type="bibr" rid="B14">Faizullabhoy and Kamthe, 2022</xref>), reproductive toxicity testing is essential to the development of novel drugs. This cost indicates that traditional <italic>in vitro</italic> and <italic>in vivo</italic> tests of toxicity remain expensive and time-consuming; they additionally raise ethical issues regarding animal use. On 10 April 2025, the FDA announced plans to replace animal studies with artificial intelligence (AI)-based computational models to assess drug toxicity (<xref ref-type="bibr" rid="B30">Office of the Commissioner, 2025</xref>), in &#x201c;a win-win for public health and ethics.&#x201d; The European Union&#x2019;s Registration, Evaluation, Authorisation, and Restriction of Chemicals regulation (REACH) and the U.S. Environmental Protection Agency (EPA)&#x2019;s <ext-link ext-link-type="uri" xlink:href="https://www.epa.gov/tsca-screening-tools">Toxic Substances Control Act</ext-link> (TSCA) now require not only extensive hazard assessments but also explicit justification for the use of animal testing, effectively making computational models a regulatory necessity. Under REACH&#x2019;s &#x201c;last-resort&#x201d; provision, animal testing can only be pursued when alternative methods, including <italic>in silico</italic> approaches, have been exhausted. Similarly, TSCA encourages the use of predictive exposure and fate models to fill data gaps in chemical assessments to reduce reliance on new animal studies (<xref ref-type="bibr" rid="B13">European Commission, 2022</xref>; <xref ref-type="bibr" rid="B38">United States Environmental Protection Agency, 2025</xref>). As these regulations continue to expand their scope, development of robust computational toxicology models is no longer optional but essential to meet global compliance while minimizing ethical and financial burdens.</p>
<p>Quantitative Structure&#x2013;Activity Relationship (QSAR) models use mathematical methods to model relationships between chemical structures&#x2019; properties and biological activities in order to predict biological activities of novel chemicals before formal experiments. These models provide a more ethical, cost-effective, rapid and efficient alternative to traditional <italic>in vitro</italic> and <italic>in vivo</italic> tests. Until the 1990s, QSAR models used simple linear and partial least squares regressions along with simple 1-D descriptors representing chemical structures. Since the early 2000s, QSAR models have evolved to incorporate machine learning (ML) methods such as Random Forest (RF), Support Vector Machines (SVM), and Extreme Gradient Boosting (XGBoost). These methods employ 2-D and 3-D descriptors, which more accurately capture molecular properties and non-linear relationships between properties and bioactivities (<xref ref-type="bibr" rid="B8">Cortes and Vapnik, 1995</xref>; <xref ref-type="bibr" rid="B7">Breiman, 2001</xref>; <xref ref-type="bibr" rid="B35">Sheridan et al., 2016</xref>; <xref ref-type="bibr" rid="B22">Li et al., 2025</xref>; <xref ref-type="bibr" rid="B3">Bahia et al., 2023</xref>; <xref ref-type="bibr" rid="B46">Xu et al., 2012</xref>; <xref ref-type="bibr" rid="B4">Ballester and Mitchell, 2010</xref>; <xref ref-type="bibr" rid="B43">Wu and Wang, 2018</xref>). These advancements resulted in more robust models and increased accuracy of predictions.</p>
<p>However, classical ML models rely on pre-computed descriptors that remain fixed throughout the training process, possibly limiting their performance. In the last 15 years, Deep Learning (DL) methods, particularly graph convolution-based Graph Neural Network (GNN), have been introduced to QSAR modeling (<xref ref-type="bibr" rid="B41">Wang et al., 2023</xref>; <xref ref-type="bibr" rid="B11">Duvenaud et al., 2015</xref>; <xref ref-type="bibr" rid="B21">Kearnes et al., 2016</xref>). In GNN models or in general, Message Passing Neural Network (MPNN), molecules are presented as undirected graphs with atoms as nodes and bonds as edges. The message passing phase of MPNN captures dynamic interactions between atoms and bonds; aggregated messages that represent whole molecules are then used to predict bioactivities through readout functions (<xref ref-type="bibr" rid="B17">Gilmer et al., 2017</xref>). Different from the node-based message passing phase of MPNN, Directed MPNN (DMPNN) considers directions of edges that better differentiate the influence between nodes, reducing redundancy in message passing (<xref ref-type="bibr" rid="B9">Dai et al., 2016</xref>; <xref ref-type="bibr" rid="B48">Yang et al., 2019</xref>; <xref ref-type="bibr" rid="B18">Han et al., 2022</xref>; <xref ref-type="bibr" rid="B45">Xia et al., 2023</xref>). Yang et al. showed that DMPNN outperformed most other deep neural network methods in predicting molecular bioactivities (<xref ref-type="bibr" rid="B48">Yang et al., 2019</xref>). More recently, the Communicative Message Passing Neural Network (CMPNN) framework, employing a communicative kernel to reinforce message exchange between nodes and edges and incorporates a message booster module during message passing to enrich molecular graph embeddings has demonstrated an enhanced predictive performance compared to DMPNN (<xref ref-type="bibr" rid="B37">Song et al., 2021</xref>). While all GNN methods offer a more dynamic and data-driven approach to developing bioactivity prediction models that allows for automatic extraction of features from molecular graphs, thus requiring less expertise for more accurate predictions of bioactivity, helping to enhance high-throughput screening processes and accelerate drug development timelines, recent research has reported significant gains of CMPNN in different molecular property prediction tasks (<xref ref-type="bibr" rid="B32">Rao et al., 2022</xref>; <xref ref-type="bibr" rid="B37">Song et al., 2021</xref>; <xref ref-type="bibr" rid="B24">Liu et al., 2024</xref>).</p>
<p>Recently in the specific area of predicting reproductive toxicity, Basant et al. (<xref ref-type="bibr" rid="B5">Basant et al., 2016</xref>) utilized two ensemble machine learning models, Decision Tree Forest and Decision Tree Boost, based on 334 chemicals of which toxicity to rats is known, to demonstrate the effectiveness of ML-based QSAR models. Further work based on larger datasets with more than 1,500 chemicals and methods including frequentist approaches (<xref ref-type="bibr" rid="B20">Jiang et al., 2019</xref>; <xref ref-type="bibr" rid="B15">Feng et al., 2021</xref>), Bayesian methods (<xref ref-type="bibr" rid="B50">Zhang et al., 2020</xref>) and Graph Transformer Networks (<xref ref-type="bibr" rid="B33">Ren et al., 2024</xref>), achieved areas under the receiver operating characteristic curve (AUCs) close or greater than 0.900 and accuracy scores (ACC) of at least 0.830.</p>
<p>Our study is to develop better-performing <italic>in silico</italic> predictive models on reproductive toxicity using a larger dataset of 2,154 chemicals that contains Simplified Molecular Input Line Entry Specifications (SMILES) and a binary classification of reproductively toxic or non-toxic. Considering the performance of CMPNN method in other areas, we compared it with 11&#xa0;ML models using standard model evaluation metrics including accuracy score (ACC), the area under the curve (AUC) of the receiver operating characteristics curve (ROC), F1 score, balanced accuracy (BA), Cohen&#x2019;s Kappa and Matthews correlation coefficient (MCC).</p>
<p>The rest of the manuscript is organized as follows. In <xref ref-type="sec" rid="s2">Section 2</xref>, the reproductive toxicity dataset and the overall process are described first, followed by a brief description of different models and metrics of model evaluation. In <xref ref-type="sec" rid="s3">Section 3</xref>, information about hyperparameters, results, and comparisons are presented. The manuscript ends with a conclusion and description of future work in <xref ref-type="sec" rid="s4">Section 4</xref>.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Toxicity data preparation</title>
<p>The reproductive toxicity dataset assembled in this study includes 2,154 small-molecule compounds from the ECHA-C&#x26;L Inventory, OECD-eChemPortal and previous literature (<xref ref-type="bibr" rid="B33">Ren et al., 2024</xref>; <xref ref-type="bibr" rid="B15">Feng et al., 2021</xref>; <xref ref-type="bibr" rid="B50">Zhang et al., 2020</xref>, Jiang et cal. 2019). The lists of chemicals are shown in <xref ref-type="sec" rid="s11">Supplementary Table S1</xref>. This dataset includes a varied mix of chemical types&#x2014;industrial substances, environmental pollutants, and pharmaceuticals. In the <ext-link ext-link-type="uri" xlink:href="https://one.oecd.org/document/ENV/JM/MONO%282018%2918/en/pdf?utm_source=chatgpt.com">European Chemicals Agency</ext-link>(ECHA) database, substances receive Category 1A (known human toxicant), 1B (presumed), or 2 (suspected) based on a weight-of-evidence approach incorporating epidemiological, <italic>in vivo</italic>, <italic>in vitro</italic>, and structural-activity data&#x2014;excluding effects merely secondary to general toxicity (<xref ref-type="bibr" rid="B12">European Chemicals Agency, 2017</xref>). Likewise, The Organisation for Economic Co-operation and Development (OECD)- eChemPortal aggregates reproductive hazard classifications (1A/1B/2 and lactation effects) from national and international regulatory sources (<xref ref-type="bibr" rid="B28">OECD, 2008</xref>; <xref ref-type="bibr" rid="B29">2023</xref>). Most of the underlying data were derived from rodent studies measuring endpoints like fertility rates, implantation success, offspring development, and congenital malformations, with human evidence primarily influencing Category 1A. Therefore, this dataset provides a good basis for distinguishing reproductively toxic from non-reproductively-toxic compounds to develop computational methods.</p>
</sec>
<sec id="s2-2">
<title>2.2 Calculation and construction of molecular fingerprints</title>
<p>This study was conducted in a Python 3.12.2 environment installed with RDKit 2025.3.2, Chemprop 2.1.0, PyTorch 2.7.1, scikit-learn 1.6.1, Pandas 2.2.2, and NumPy 1.26.4. SMILES strings of the 2,154 chemicals were converted with the standard SMILES parser into RDKit molecular objects. Morgan fingerprints (ECFP4) were generated with a radius of 2 and a total length of 2,048 bits in binary (presence/absence) mode, producing ExplicitBitVect-type representations that provided a reliable data foundation for subsequent classification modeling.</p>
</sec>
<sec id="s2-3">
<title>2.3 Classical machine learning methods</title>
<p>This study used eleven classical machine learning methods to predict chemical compounds&#x2019; reproductive toxicity. Decision Tree constructs an interpretable tree by recursively splitting on feature thresholds, k-Nearest Neighbors is a &#x201c;lazy&#x201d; learner that assigns class based on the majority vote of nearest training samples in Euclidean space, Linear SVM finds a maximum-margin hyperplane for linear separation, Naive Bayes applies Bayes&#x2019; theorem under a conditional feature-independence assumption, Logistic Regression fits a sigmoid function to estimate class probabilities, Random Forest ensembles multiple decision trees built on random features and sample subsets to reduce overfitting, Adaptive Boosting (AdaBoost) trains weighted weak learners iteratively and combines them into a strong classifier, Gradient Boosted Decision Tree (GBDT) builds learners sequentially by fitting residual errors, Extra Trees further randomizes split thresholds and features to increase model diversity, Light Gradient Boosting Machine (LightGBM) employs a leaf-wise tree growth strategy for faster training, and XGBoost uses second-order derivative information and regularization to optimize both speed and accuracy. To ensure fair comparison and reproducibility, all models were tuned and evaluated within a unified pipeline. Algorithm implementation and key hyperparameters are shown in <xref ref-type="sec" rid="s11">Supplementary Table S2</xref>.</p>
</sec>
<sec id="s2-4">
<title>2.4 Communicative message passing neural network (CMPNN)</title>
<p>In this study, we selected the CMPNN framework (<xref ref-type="fig" rid="F1">Figure 1</xref>), given its architecture of enhanced message exchange and boosting, to develop our reproductive toxicity prediction model (ReproTox-CMPNN) and compared it with machine learning methods.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Architecture of CMPNN model.</p>
</caption>
<graphic xlink:href="ftox-07-1640612-g001.tif">
<alt-text content-type="machine-generated">Diagram showing a molecular graph model process. It begins with a molecular graph labeled G, leading to node and edge states. The process includes message passing with boosters, node-edge communication, and state updates across multiple iterations. The final outputs are an atom representation and bioactivity prediction.</alt-text>
</graphic>
</fig>
<p>To provide a more formal and precise description, we decomposed CMPNN into five steps, each with its governing equations.</p>
<sec id="s2-4-1">
<title>2.4.1 Graphical representation</title>
<p>Each molecule is represented as a directed graph<disp-formula id="equ1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the set of atoms (nodes) and <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the set of directed bonds (edges). Each atom <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> has an initial feature vector <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> &#x200b;, and each directed edge (<inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x2192;<inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) has an initial embedding <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, both obtained from atom- and bond-level descriptors.</p>
</sec>
<sec id="s2-4-2">
<title>2.4.2 Message passing</title>
<p>Over K iterations, node states are updated by aggregating incoming edge messages:<disp-formula id="equ2">
<mml:math id="m9">
<mml:mrow>
<mml:msubsup>
<mml:mi>m</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>:</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>.</mml:mo>
<mml:mo>.</mml:mo>
<mml:mo>.</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>The new node hidden state is then<disp-formula id="equ3">
<mml:math id="m10">
<mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>m</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where &#x201c;&#x2016;&#x201d; denotes concatenation and <inline-formula id="inf9">
<mml:math id="m12">
<mml:mrow>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>&#x200b; is the message-booster (<xref ref-type="sec" rid="s2-4-3">Section 2.4.3</xref>).</p>
</sec>
<sec id="s2-4-3">
<title>2.4.3 Message booster</title>
<p>To amplify the most informative bond contributions, for each node <inline-formula id="inf10">
<mml:math id="m13">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> we compute<disp-formula id="equ4">
<mml:math id="m14">
<mml:mrow>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi mathvariant="italic">max</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>:</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>(element-wise maximum over its incoming edge embeddings). We then scale the aggregated message by element-wise multiplication:<disp-formula id="equ5">
<mml:math id="m15">
<mml:mrow>
<mml:msubsup>
<mml:mi>m</mml:mi>
<mml:mi>v</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mi>m</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2299;</mml:mo>
<mml:msubsup>
<mml:mi>b</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>and use <inline-formula id="inf11">
<mml:math id="m16">
<mml:mrow>
<mml:msubsup>
<mml:mi>m</mml:mi>
<mml:mi>v</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>&#x200b; in place of <inline-formula id="inf12">
<mml:math id="m17">
<mml:mrow>
<mml:msubsup>
<mml:mi>m</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> in the node update above.</p>
</sec>
<sec id="s2-4-4">
<title>2.4.4 Node-edge communication</title>
<p>Edge embeddings are updated based on the newly computed node states. For each directed bond (<inline-formula id="inf13">
<mml:math id="m18">
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x2192;<inline-formula id="inf14">
<mml:math id="m19">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>),<disp-formula id="equ6">
<mml:math id="m20">
<mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="" separators="|">
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf15">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a learned weight matrix and ReLU is a rectified linear unit used at the next iteration. This residual formulation allows atom and bond representations to co-evolve.</p>
</sec>
<sec id="s2-4-5">
<title>2.4.5 Readout and prediction head</title>
<p>After the final iteration t &#x3d; K, we obtain node embeddings <inline-formula id="inf16">
<mml:math id="m22">
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>. We then apply a gated recurrent unit (GRU) readout to each node (to capture ordering effects), and sum over all nodes to produce a fixed-length molecular vector:<disp-formula id="equ7">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi>G</mml:mi>
<mml:mi>R</mml:mi>
<mml:mi>U</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Finally, a two-layer perceptron with dropout maps <inline-formula id="inf17">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to the toxicity probability:<disp-formula id="equ8">
<mml:math id="m25">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf18">
<mml:math id="m26">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the sigmoid activation.</p>
</sec>
</sec>
<sec id="s2-5">
<title>2.5 Model training, hyperparameter optimization and evaluation</title>
<p>Using Python and RDKit, we generated classical molecular fingerprints and physicochemical descriptors as input features for various machine learning models like Random Forest (RF) and Support Vector Machines (SVM). To overcome the limitations of fingerprints, such as difficulty with activity cliffs and neglecting 3D conformational details, we constructed molecular graphs G and applied ReproTox-CMPNN to learn end-to-end embeddings based on automatic extraction of rich topological and chemical context from the molecular structure (<xref ref-type="fig" rid="F2">Figure 2</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Machine learning and deep learning process to predict reproductive toxicity.</p>
</caption>
<graphic xlink:href="ftox-07-1640612-g002.tif">
<alt-text content-type="machine-generated">Diagram showing a machine learning workflow for evaluating reproductive toxicity using chemical data from ECHA's eChemPortal. It includes data preprocessing with Python and RDKit, classical models (linear regression, decision tree, support vector machines), and a CMPNN for molecular embedding. Model evaluation involves repeated nested cross-validation. Evaluation metrics listed include sensitivity, specificity, accuracy, balanced accuracy, F1 score, Cohen&#x2019;s Kappa, MCC, and AUC. The process aims to predict reproductive toxicity.</alt-text>
</graphic>
</fig>
<p>Model training and evaluation consisted of a rigorous repeated nested cross-validation scheme. In the outer loop, the full dataset was randomly partitioned into five distinct folds, with one fold each time serving as the test set for performance evaluation. In the inner loop, a similar procedure was also repeated five times, with 12.5% each time (10% of the total data) serving for validation/hyperparameter tuning. Nested cross-validation is more robust than 1-layer five-fold cross-validation as it uses inner loop for tuning and a separate set in outer loop for unbiased testing. The 80/20 split in the outer loop is a widely adopted guideline that balances the need for ample training data to learn patterns while retaining a sufficiently large test set to assess generalization; <xref ref-type="bibr" rid="B16">Gholamy et al. (2018)</xref> concluded that &#x201c;p &#x2248; 80% is empirically the best division into the training and the testing sets.&#x201d; This system ensured our results&#x2019; robustness and reproducibility, laying a solid foundation for future virtual screening and toxicological risk assessment.</p>
<p>Binary cross-entropy loss was minimized using the Adam optimizer, batch size &#x3d; 50, maximum epochs &#x3d; 60, with early stopping patience &#x3d; 10 on validation AUC. Hyperparameters (hidden dimension &#x3d; 256) were tuned via grid search within each training fold. All experiments were implemented in PyTorch and run on a Linux server equipped with 6 NVIDIA 4090 GPUs.</p>
</sec>
<sec id="s2-6">
<title>2.6 Model evaluation metrics</title>
<p>The comparison of machine learning and ReproTox-CMPNN models was based on accuracy score (ACC), balanced accuracy (BA), Cohen&#x2019;s Kappa, Matthews correlation coefficient (MCC), F1 score, and the area under the curve (AUC) of the receiver operating characteristic curve (ROC). The designations TP, TN, FP, and FN in calculations refer to the number of true positives, true negatives, false positives, and false negatives, respectively. Compared to accuracy, Cohen&#x2019;s kappa accounts for the possibility of agreements due to randomness, with <inline-formula id="inf19">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> defined as the observed agreement between two classifiers and <inline-formula id="inf20">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> as the expected agreement by chance.<disp-formula id="equ9">
<mml:math id="m29">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ10">
<mml:math id="m30">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ11">
<mml:math id="m31">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ12">
<mml:math id="m32">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.5</mml:mn>
<mml:mo>&#x2217;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ13">
<mml:math id="m33">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#x2217;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2217;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ14">
<mml:math id="m34">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:msup>
<mml:mi>n</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mi>s</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>K</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>a</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ15">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mtext>&#x2002;</mml:mtext>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mi>e</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ16">
<mml:math id="m36">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>AUC is calculated as the following with FPR as X-axis and TPR as Y-axis. A higher AUC value indicates a greater ability of the model to distinguish between positive and negative cases, i.e., better performance:<disp-formula id="equ17">
<mml:math id="m37">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x222b;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mn>1</mml:mn>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>or using trapezoidal rule,<disp-formula id="equ18">
<mml:math id="m38">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>U</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>&#x2248;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2217;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mfrac>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Similarly, for other metrics, a value closer to 1 indicates superior model performance.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Key molecular descriptors</title>
<p>The dataset used in this study contains Simplified Molecular Input Line Entry Specifications (SMILES) and a binary classification of reproductive toxicity (1 &#x3d; Yes, 0 &#x3d; No). Of the 2,154 chemicals, mainly organic compounds, 1,091 (51%) and 1,063 (49%) are classified as reproductively toxic and non-toxic, respectively. This balanced distribution provides a solid foundation for subsequent QSAR model training and validation.</p>
<p>To better understand the physicochemical features that distinguish reproductive toxicants from non-toxic compounds, we first assessed six common molecular descriptors across our dataset. These six descriptors reflect basic physicochemical characteristics of molecules as well as widely used in QSAR models and drug discovery. As shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, we compared them between toxic and non-toxic compounds. Overall, toxicants exhibit higher median values of and larger dispersion in molecular weight (Weight) and topological polar surface area (TPSA) compared to non-toxic molecules, suggesting that larger and more polar structures are more likely reproductive toxicants. Specifically, the median molecular weight in the non-toxic group is 200&#xa0;Da compared to 314&#xa0;Da in the toxic group; similarly, the median non-toxic TPSA is 37&#xa0;&#xc5;<sup>2</sup> compared to 58&#xa0;&#xc5;<sup>2</sup> for toxicants with a more right-skewed distribution with more extreme outliers exceeding 200&#xa0;&#xc5;<sup>2</sup>. In terms of lipophilicity, the median of logarithm of the partition coefficient (Log P) is 2.43 in the non-toxic class, compared to 2.76 in the toxic class, although the non-toxic class displays more outliers beyond a very high value (&#x3e;10), reflecting that enhanced lipophilicity may facilitate membrane permeation and toxic bioaccumulation.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Distribution of key molecular descriptors by reproductive toxicity classification.</p>
</caption>
<graphic xlink:href="ftox-07-1640612-g003.tif">
<alt-text content-type="machine-generated">Box plots and histograms comparing non-toxic and toxic samples. The box plots show distributions for weight, topological polar surface area, and log P. Histograms display percentages of acceptors, bonds, and donors, with data segregated into non-toxic (blue) and toxic (red) categories.</alt-text>
</graphic>
</fig>
<p>For hydrogen bond donors and acceptors, a slightly more right-skewed distribution in the toxic class indicates potential involvement of hydrogen-bonding interactions in mediating reproductive toxicity. However, comparable distributions between the two classes suggest that these descriptors alone are insufficient for clear-cut classification. Similarly, distributions of rotatable bonds imply limited impact of molecular flexibility on toxicity risk. Taken together, while several molecular descriptors show discernible trends between toxic and non-toxic compounds, these distributions&#x2019; overall comparability indicates a need for multivariate modeling. Integrating these descriptors within a comprehensive machine learning framework would be key to robust and generalizable reproductive toxicity predictions.</p>
</sec>
<sec id="s3-2">
<title>3.2 Hyperparameter search and model training</title>
<p>ReproTox-CMPNN uses a hyperparameter configuration carefully designed to improve predictive performance by balancing convergence and efficiency (<xref ref-type="table" rid="T1">Table 1</xref>). We used 60 training epochs with a batch size of 50, random enough to avoid local optima while effectively taking advantage of GPU memory to handle molecules of different sizes. We additionally employed an adaptive learning rate through a two-epoch warm-up phase followed by a decay phase.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Hyperparameters of ReproTox-CMPNN.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Hyperparameter</th>
<th align="center">Value</th>
<th align="center">Description</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">epochs</td>
<td align="center">60</td>
<td align="center">Number of epochs to run</td>
</tr>
<tr>
<td align="center">batch_size</td>
<td align="center">50</td>
<td align="center">Batch size</td>
</tr>
<tr>
<td align="center">warmup_epochs</td>
<td align="center">2</td>
<td align="center">Epochs for linear LR warmup</td>
</tr>
<tr>
<td align="center">init_lr</td>
<td align="center">1.00E-04</td>
<td align="center">Initial learning rate</td>
</tr>
<tr>
<td align="center">max_lr</td>
<td align="center">1.00E-03</td>
<td align="center">Maximum learning rate</td>
</tr>
<tr>
<td align="center">final_lr</td>
<td align="center">1.00E-04</td>
<td align="center">Final learning rate</td>
</tr>
<tr>
<td align="center">hidden_size</td>
<td align="center">300</td>
<td align="center">Dimensionality of hidden layers in MPN</td>
</tr>
<tr>
<td align="center">bias</td>
<td align="center">FALSE</td>
<td align="center">Whether to add bias to linear layers</td>
</tr>
<tr>
<td align="center">depth</td>
<td align="center">3</td>
<td align="center">Number of message passing steps</td>
</tr>
<tr>
<td align="center">activation</td>
<td align="center">ReLU</td>
<td align="center">Activation function</td>
</tr>
<tr>
<td align="center">undirected</td>
<td align="center">FALSE</td>
<td align="center">Use undirected edges</td>
</tr>
<tr>
<td align="center">ffn_hidden_size</td>
<td align="center">None</td>
<td align="center">Hidden dim for FFN</td>
</tr>
<tr>
<td align="center">ffn_num_layers</td>
<td align="center">2</td>
<td align="center">Number of layers in FFN</td>
</tr>
<tr>
<td align="center">atom_messages</td>
<td align="center">FALSE</td>
<td align="center">Use atom-to-atom messages</td>
</tr>
<tr>
<td align="center">ensemble_size</td>
<td align="center">1</td>
<td align="center">Number of models in the ensemble</td>
</tr>
<tr>
<td align="center">num_folds</td>
<td align="center">5</td>
<td align="center">Number of folds in cross-validation</td>
</tr>
<tr>
<td align="center">split_sizes</td>
<td align="center">0.7:0.1:0.2</td>
<td align="center">Dataset split ratio</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Learning rate increases linearly from 1e-4 to 1e-3 in the warm-up stage and then decreases exponentially to 1e-4. This method reduces instability during early training, accelerates convergence in the middle stage, and allows for fine-tuning in later stages.</p>
<p>In ReproTox-CMPNN&#x2019;s architecture, we set the hidden layer dimension to 300 to sufficiently capture complex structural information in molecular graphs while avoiding over-parameterization; in molecular graph neural networks, this hidden layer dimension determines the richness of node (atom) and edge (bond) representations, an important part of the model. ReproTox-CMPNN&#x2019;s linear layers default to not using bias terms, reducing the number of model parameters and lowering overfitting risk. For molecular graph representations, relative relationships are typically more important than absolute offsets, and this bias-free design enables the model to focus more on learning the relative feature importance in molecular structures. The message passing depth is set to 3, allowing each atom to perceive neighboring atoms up to three hops away. For most drug molecules, this depth adequately captures key structural information and chemical environments while avoiding over-smoothing and overfitting problems that deeper networks might introduce.</p>
<p>We obtained results presented in <xref ref-type="fig" rid="F4">Figure 4A</xref> which show the ROC curves of the ReproTox-CMPNN model under five-fold cross-validation. The AUC values for folds 0&#x2013;4 are 0.929, 0.942, 0.962, 0.956, and 0.939, respectively, resulting in a mean AUC of 0.946 with a standard deviation of 0.013. These results demonstrate excellent discriminative power and low variability across different folds, indicating robust stability and generalization performance. <xref ref-type="fig" rid="F4">Figure 4B</xref>, showing the evolution of AUC during training epochs, illustrates the model&#x2019;s rapid improvement within the first 10 epochs, with AUC rising from approximately 0.85 to over 0.92, after which the curve gradually flattens and stabilizes around 0.93. This behavior suggests fast convergence and no significant overfitting during later stages of training coupled with a consistent predictive accuracy throughout.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>
<bold>(A)</bold> ROC Curve, <bold>(B)</bold> AUC change during model training.</p>
</caption>
<graphic xlink:href="ftox-07-1640612-g004.tif">
<alt-text content-type="machine-generated">Chart A shows an ROC curve with a mean AUC of 0.946 across five folds, with individual AUCs ranging from 0.929 to 0.962. Chart B plots AUC over epochs, showing an increase from 0.84 to 0.92.</alt-text>
</graphic>
</fig>
<p>Overall, the CMPNN-based model achieves high and consistent AUC scores in cross-validation and demonstrates quick convergence and resilience during the training process, thus providing a solid foundation for reliable reproductive toxicity prediction of new compounds.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<sec id="s4-1">
<title>4.1 Comparison of model performance</title>
<p>To provide a comprehensive benchmark of predictive methods on our reproductive toxicity dataset, we first evaluated eleven classical machine learning algorithms alongside the ReproTox-CMPNN model. As summarized in <xref ref-type="table" rid="T2">Table 2</xref>, ReproTox-CMPNN achieved an AUC of 0.946, substantially outperforming the next-best methods Random Forest (0.825) and Linear SVM (0.823). This clear margin underscores ReproTox-CMPNN&#x2019;s superior ability to capture intricate molecular topology and chemical context, leading to markedly improved discrimination between toxic and non-toxic compounds.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Model performance based on machine learning and the ReproTox-CMPNN algorithm.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">AUC mean (std&#x2a;)</th>
<th align="center">ACC mean (std)</th>
<th align="center">BA mean (std)</th>
<th align="center">Sensitivity mean (std)</th>
<th align="center">Specificity mean (std)</th>
<th align="center">Kappa mean (std)</th>
<th align="center">MCC mean (std)</th>
<th align="center">F1 score mean (std)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Decision Tree</td>
<td align="center">0.730 (0.018)</td>
<td align="center">0.725 (0.020)</td>
<td align="center">0.730 (0.018)</td>
<td align="center">0.683 (0.036)</td>
<td align="center">0.778 (0.016)</td>
<td align="center">0.454 (0.039)</td>
<td align="center">0.459 (0.037)</td>
<td align="center">0.7632 (0.019)</td>
</tr>
<tr>
<td align="center">Nearest Neighbors</td>
<td align="center">0.703 (0.008)</td>
<td align="center">0.684 (0.015)</td>
<td align="center">0.703 (0.008)</td>
<td align="center">0.516 (0.018)</td>
<td align="center">0.889 (0.003)</td>
<td align="center">0.388 (0.022)</td>
<td align="center">0.428 (0.016)</td>
<td align="center">0.642 (0.012)</td>
</tr>
<tr>
<td align="center">Linear SVM</td>
<td align="center">0.825 (0.001)</td>
<td align="center">0.813 (0.003)</td>
<td align="center">0.825 (0.001)</td>
<td align="center">0.710 (0.005)</td>
<td align="center">0.939 (0.07)</td>
<td align="center">0.633 (0.004)</td>
<td align="center">0.655 (0.001)</td>
<td align="center">0.807 (0.001)</td>
</tr>
<tr>
<td align="center">Naive Bayes</td>
<td align="center">0.775 (0.005)</td>
<td align="center">0.767 (0.007)</td>
<td align="center">0.775 (0.005)</td>
<td align="center">0.692 (&#x3c;0.001)</td>
<td align="center">0.858 (0.009)</td>
<td align="center">0.538 (0.013)</td>
<td align="center">0.550 (0.013)</td>
<td align="center">0.765 (&#x3c;0.001)</td>
</tr>
<tr>
<td align="center">Logistic Regression</td>
<td align="center">0.797 (0.021)</td>
<td align="center">0.792 (0.019)</td>
<td align="center">0.797 (0.021)</td>
<td align="center">0.751 (0.008)</td>
<td align="center">0.843 (0.034)</td>
<td align="center">0.586 (0.038)</td>
<td align="center">0.591 (0.040)</td>
<td align="center">0.799 (0.021)</td>
</tr>
<tr>
<td align="center">RandomForest</td>
<td align="center">0.813 (0.007)</td>
<td align="center">0.808 (0.009)</td>
<td align="center">0.813 (0.007)</td>
<td align="center">0.765 (0.015)</td>
<td align="center">0.861 (0.005)</td>
<td align="center">0.617 (0.018)</td>
<td align="center">0.623 (0.016)</td>
<td align="center">0.814 (0.006)</td>
</tr>
<tr>
<td align="center">AdaBoost</td>
<td align="center">0.773 (0.007)</td>
<td align="center">0.766 (0.004)</td>
<td align="center">0.773 (0.007)</td>
<td align="center">0.706 (0.012)</td>
<td align="center">0.841 (0.027)</td>
<td align="center">0.536 (0.010)</td>
<td align="center">0.546 (0.014)</td>
<td align="center">0.768 (0.006)</td>
</tr>
<tr>
<td align="center">GradientBoosting</td>
<td align="center">0.811 (0.004)</td>
<td align="center">0.804 (0.003)</td>
<td align="center">0.811 (0.004)</td>
<td align="center">0.748 (0.002)</td>
<td align="center">0.874 (0.009)</td>
<td align="center">0.611 (0.006)</td>
<td align="center">0.619 (0.008)</td>
<td align="center">0.807 (0.006)</td>
</tr>
<tr>
<td align="center">Extra Trees</td>
<td align="center">0.809 (0.006)</td>
<td align="center">0.804 (0.005)</td>
<td align="center">0.809 (0.006)</td>
<td align="center">0.763 (0.019)</td>
<td align="center">0.855 (0.022)</td>
<td align="center">0.610 (0.010)</td>
<td align="center">0.616 (0.011)</td>
<td align="center">0.811 (0.005)</td>
</tr>
<tr>
<td align="center">Lightboost</td>
<td align="center">0.828 (0.014)</td>
<td align="center">0.821 (0.010)</td>
<td align="center">0.828 (0.014)</td>
<td align="center">0.768 (0.012)</td>
<td align="center">0.888 (0.040)</td>
<td align="center">0.645 (0.022)</td>
<td align="center">0.654 (0.027)</td>
<td align="center">0.825 (0.011)</td>
</tr>
<tr>
<td align="center">XGBoost</td>
<td align="center">0.825 (0.009)</td>
<td align="center">0.820 (0.007)</td>
<td align="center">0.825 (0.009)</td>
<td align="center">0.778 (0.003)</td>
<td align="center">0.873 (0.015)</td>
<td align="center">0.642 (0.015)</td>
<td align="center">0.647 (0.016)</td>
<td align="center">0.826 (0.011)</td>
</tr>
<tr>
<td align="center">ReproTox-CMPNN</td>
<td align="center">0.946 (0.013)</td>
<td align="center">0.857 (0.019)</td>
<td align="center">0.856 (0.018)</td>
<td align="center">0.823 (0.076)</td>
<td align="center">0.890 (0.085)</td>
<td align="center">0.713 (0.037)</td>
<td align="center">0.721 (0.033)</td>
<td align="center">0.846 (0.018)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>&#x2a;standard deviation.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>In terms of overall classification metrics, ReproTox-CMPNN attained a mean accuracy of 0.857 with a standard deviation of 0.019 and a balanced accuracy (BA) of 0.856 with a standard deviation of 0.018, significantly higher than those of other models (e.g., Random Forest 0.825).</p>
<p>These results indicate that ReproTox-CMPNN delivers not only excellent overall predictive power but also maintains high balance between positive (toxic) and negative (non-toxic) classes, effectively mitigating the biases arising from class imbalance. Furthermore, its sensitivity (mean 0.823 and standard deviation 0.076) and specificity (mean 0.890 and standard deviation 0.085) both exceed 0.80, demonstrating reliable recall of toxicants and exclusion of non-toxicants.</p>
<p>Examining more stringent agreement measures, ReproTox-CMPNN&#x2019;s Cohen&#x2019;s Kappa (mean 0.713 and standard deviation 0.037) and Matthews Correlation Coefficient (mean 0.721 and standard deviation 0.033) are well above those of traditional models (most of which fall between 0.50 and 0.65), highlighting strong consistency and correlation with true labels. The F1 score of mean of 0.846 and standard deviation of 0.018, further reflects a balanced trade-off between precision and recall, particularly excelling in identifying toxic compounds. In contrast, simpler approaches such as K-Nearest Neighbors (AUC &#x3d; 0.717) or Naive Bayes (AUC &#x3d; 0.783) exhibit inferior discrimination and stability.</p>
<p>In summary, the ReproTox-CMPNN model, through its advanced representation of molecular structures, significantly surpasses various conventional machine learning algorithms, offering a powerful and robust framework for reproductive toxicity prediction with promising applicability in risk assessment pipelines.</p>
</sec>
<sec id="s4-2">
<title>4.2 Comparison with recent prediction models on reproductive toxicity</title>
<p>As shown in <xref ref-type="table" rid="T3">Table 3</xref>, in recent work regarding reproductive toxicity prediction, <xref ref-type="bibr" rid="B20">Jiang et al. (2019)</xref> used a much more extensive dataset (1,823 chemicals) compared to previous publications, multiple endpoints such as sperm reduction and infertility, and six machine learning methods to develop more reliable models. Their study recommended the SVM model using Molecular Access System Keys Fingerprints (MACCSFP), which generated an AUC of 0.900 and an accuracy score of 0.836. As previous work mainly focused on frequentist methods, <xref ref-type="bibr" rid="B50">Zhang et al. (2020)</xref> investigated Naive Bayes (NB) model together with six molecular descriptors and ten types of fingerprints. Their best model resulted in an AUC of 0.888 and an accuracy score of 0.830. <xref ref-type="bibr" rid="B15">Feng et al. (2021)</xref> used three machine learning methods with nine molecular fingerprints to build more ensemble models. The model they recommended had an AUC of 0.920 and an accuracy score of 0.844. <xref ref-type="bibr" rid="B33">Ren et al. (2024)</xref> developed a deep learning fragment-based graph transformer network (FGTN) model to predict reproductive toxicity, taking pre-generated fragments as nodes and bonds between fragments as edges with a super-molecule-level node to connect all fragment nodes. The FGTN model showed an AUC of 0.914 and an accuracy score of 0.861.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Comparison to the previous models reported in the literature.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model name</th>
<th align="center">AUC</th>
<th align="center">ACC</th>
<th align="center">MCC</th>
<th align="center">No. of compounds</th>
<th align="center">References</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">MACCSFP-SVM<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</td>
<td align="center">0.900</td>
<td align="center">0.836</td>
<td align="center">0.679</td>
<td align="center">1823</td>
<td align="center">
<xref ref-type="bibr" rid="B20">Jiang et al. (2019)</xref>
</td>
</tr>
<tr>
<td align="center">NB-1<xref ref-type="table-fn" rid="Tfn2">
<sup>b</sup>
</xref>
</td>
<td align="center">0.888</td>
<td align="center">0.830</td>
<td align="center">0.663</td>
<td align="center">1685</td>
<td align="center">
<xref ref-type="bibr" rid="B50">Zhang et al. (2020)</xref>
</td>
</tr>
<tr>
<td align="center">Ensemble-Top12<xref ref-type="table-fn" rid="Tfn3">
<sup>c</sup>
</xref>
</td>
<td align="center">0.920</td>
<td align="center">0.844</td>
<td align="center">-<xref ref-type="table-fn" rid="Tfn4">
<sup>d</sup>
</xref>
</td>
<td align="center">1823</td>
<td align="center">
<xref ref-type="bibr" rid="B15">Feng et al. (2021)</xref>
</td>
</tr>
<tr>
<td align="center">FGTN<xref ref-type="table-fn" rid="Tfn5">
<sup>e</sup>
</xref>
</td>
<td align="center">0.914</td>
<td align="center">0.861</td>
<td align="center">0.723</td>
<td align="center">2053</td>
<td align="center">
<xref ref-type="bibr" rid="B33">Ren et al. (2024)</xref>
</td>
</tr>
<tr>
<td align="center">ReproTox-CMPNN<xref ref-type="table-fn" rid="Tfn6">
<sup>f</sup>
</xref>
</td>
<td align="center">0.946</td>
<td align="center">0.857</td>
<td align="center">0.721</td>
<td align="center">2154</td>
<td align="center">Current study</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn1">
<label>
<sup>a</sup>
</label>
<p>MACCSFP-SVM, support vector machine model based on MACCS, fingerprints.</p>
</fn>
<fn id="Tfn2">
<label>
<sup>b</sup>
</label>
<p>NB-1, na&#xef;ve bayes-classifier model based on six molecular descriptors and LCFC_20 fingerprints.</p>
</fn>
<fn id="Tfn3">
<label>
<sup>c</sup>
</label>
<p>Ensemble-Top12, ensemble model based on support vector machine, random forest, and extreme gradient boosting methods and 9 molecular fingerprints.</p>
</fn>
<fn id="Tfn4">
<label>
<sup>d</sup>
</label>
<p>Not reported.</p>
</fn>
<fn id="Tfn5">
<label>
<sup>e</sup>
</label>
<p>Fragment-based graph transformer network.</p>
</fn>
<fn id="Tfn6">
<label>
<sup>f</sup>
</label>
<p>results based on a repeated nested cross-validation procedure.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Due to differences in datasets and data splitting, a direct comparison between the aforementioned models and the ReproTox-CMPNN model is not possible. However, a simple comparison can still provide some valuable information. ReproTox-CMPNN remarkably outperformed other models in AUC and was very close to the FGTN model in ACC and MCC, exceeding the remaining models. These results suggest that the ReproTox-CMPNN model is both reliable and robust (<xref ref-type="table" rid="T3">Table 3</xref>).</p>
<p>CMPNN&#x2019;s strong performance across multiple molecular property benchmarks is attributed to its enhanced messaging framework: unlike traditional MPNNs that focus solely on node-to-node communication, CMPNN utilizes a communicative kernel to reinforce interactions between both node and edge features, complemented by a message booster, generating richer molecular representations. As it is inherently based on 2D molecular graphs, CMPNN may miss stereochemistry and spatial relationships crucial for predicting properties like enantiomer-specific activity. Furthermore, because our training set mainly comprises organic environmental pollutants, our model&#x2019;s generalizability to other chemical domains such as metal complexes or inorganic salts remains to be demonstrated. Finally, since our current evaluation focuses on specific endpoints, validating CMPNN across a broader spectrum of toxicity readouts such as DNA damage, endocrine disruption will be needed to further illuminate its applicability.</p>
</sec>
<sec id="s4-3">
<title>4.3 Conclusion and future work</title>
<p>In this study, we built prediction models on a comprehensive reproductive toxicity dataset with 2,154 chemicals and compared the performance of the deep learning ReproTox-CMPNN model with other methods. Our results show that ReproTox-CMPNN learning results exceed the current best ML models in both embedding quality and predictive accuracy, making it a new state-of-the-art method in this field. ReproTox-CMPNN&#x2019;s deep capture of multi-level molecular relationships offers an efficient and reliable computational tool for rapid chemical safety screening and risk assessment.</p>
<p>Future work includes training ReproTox-CMPNN to predict multiple toxicity endpoints or reproductive toxicity of chemical mixtures as well as combining ReproTox-CMPNN with mask language model (MLM) for protein embedding to further improve predictions.</p>
</sec>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s11">Supplementary Material</xref>, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>OH: Conceptualization, Data curation, Formal Analysis, Methodology, Software, Visualization, Writing &#x2013; original draft, Writing &#x2013; review and editing. DC: Conceptualization, Methodology, Software, Supervision, Writing &#x2013; original draft. YL: Conceptualization, Methodology, Supervision, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research and/or publication of this article.</p>
</sec>
<ack>
<p>The authors would like to thank the reviewers for their thoughtful and valuable comments that improved the clarity, rigor, and overall quality of our manuscript.</p>
</ack>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/ftox.2025.1640612/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/ftox.2025.1640612/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Supplementaryfile1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Archibong</surname>
<given-names>A. E.</given-names>
</name>
<name>
<surname>Meredith</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Rideout</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Harris</surname>
<given-names>K. J.</given-names>
</name>
<name>
<surname>Ramesh</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Oxidative stress in reproductive toxicology</article-title>. <source>Curr. Opin. Toxicol.</source> <volume>7</volume>, <fpage>95</fpage>&#x2013;<lpage>101</lpage>. <pub-id pub-id-type="doi">10.1016/j.cotox.2017.10.004</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Attina</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Hauser</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sathyanarayana</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hunt</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Bourguignon</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Myers</surname>
<given-names>J. P.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Exposure to endocrine-disrupting chemicals in the USA: a population-based disease burden and cost analysis</article-title>. <source>Lancet Diabetes Endocrinol.</source> <volume>4</volume>, <fpage>996</fpage>&#x2013;<lpage>1003</lpage>. <pub-id pub-id-type="doi">10.1016/S2213-8587(16)30275-3</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bahia</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Kaspi</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Touitou</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Binayev</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Dhail</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Spiegel</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>A comparison between 2D and 3D descriptors in QSAR modeling based on bio&#x2010;active conformations</article-title>. <source>Mol. Inf.</source> <volume>42</volume>, <fpage>e2200186</fpage>. <pub-id pub-id-type="doi">10.1002/minf.202200186</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ballester</surname>
<given-names>P. J.</given-names>
</name>
<name>
<surname>Mitchell</surname>
<given-names>J. B. O.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>A machine learning approach to predicting protein&#x2013;ligand binding affinity with applications to molecular docking</article-title>. <source>Bioinformatics</source> <volume>26</volume>, <fpage>1169</fpage>&#x2013;<lpage>1175</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btq112</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Basant</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>K. P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>QSAR modeling for predicting reproductive toxicity of chemicals in rats for regulatory purposes</article-title>. <source>Toxico. Res.</source> <volume>5</volume>, <fpage>1029</fpage>&#x2013;<lpage>1038</lpage>. <pub-id pub-id-type="doi">10.1039/c6tx00083e</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brehm</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Flaws</surname>
<given-names>J. A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Transgenerational effects of endocrine-disrupting chemicals on male and female reproduction</article-title>. <source>Endocrinology</source> <volume>160</volume> (<issue>6</issue>), <fpage>1421</fpage>&#x2013;<lpage>1435</lpage>. <pub-id pub-id-type="doi">10.1210/en.2019-00034</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Mach. Learn.</source> <volume>45</volume>, <fpage>5</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/a:1010933404324</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cortes</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Vapnik</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>1995</year>). <article-title>Support-vector networks</article-title>. <source>Mach. Learn.</source> <volume>20</volume>, <fpage>273</fpage>&#x2013;<lpage>297</lpage>. <pub-id pub-id-type="doi">10.1007/bf00994018</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dai</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Discriminative embeddings of latent variable models for structured data</article-title>. <source>Int. Conf. Mach. Learn.</source> <volume>48</volume>, <fpage>2702</fpage>&#x2013;<lpage>2711</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1603.05629</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Di Renzo</surname>
<given-names>G. C.</given-names>
</name>
<name>
<surname>Conry</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Blake</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>DeFrancesco</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>DeNicola</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Martin</surname>
<given-names>J. N.</given-names>
<suffix>Jr.</suffix>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>International Federation of Gynecology and Obstetrics opinion on reproductive health impacts of exposure to toxic environmental chemicals</article-title>. <source>Int. J. Gynecol. Obstet.</source> <volume>131</volume>, <fpage>219</fpage>&#x2013;<lpage>225</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijgo.2015.09.002</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Duvenaud</surname>
<given-names>D. K.</given-names>
</name>
<name>
<surname>Maclaurin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Iparraguirre</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bombarell</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hirzel</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Aspuru-Guzik</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Convolutional networks on graphs for learning molecular fingerprints</article-title>. <source>Adv. Neur</source>, <fpage>2224</fpage>&#x2013;<lpage>2232</lpage>. <pub-id pub-id-type="doi">10.5555/2969442.2969488</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<collab>European Chemicals Agency</collab> (<year>2017</year>). <article-title>Reproductive toxicity &#x2013; classification under CLP (regulation (EC) no 1272/2008 on classification, labelling and packaging of chemicals)</article-title>
</citation>
</ref>
<ref id="B13">
<citation citation-type="book">
<collab>European Commission</collab> (<year>2022</year>). &#x201c;<article-title>REACH regulation</article-title>,&#x201d; in <source>REACH regulation - European commission</source>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Faizullabhoy</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kamthe</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Reproductive toxicity testing market&#x2014;by product (consumables, assays, equipment), by method (cellular assays, biochemical assays, in-silico models, <italic>ex-vivo</italic> models), by technology (cell culture technology, high-throughput technology, toxicogenomics) by end-use&#x2014;global forecast, 2023&#x2013;2032</article-title>. <source>Glob. Mark. Insights</source>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.gminsights.com/industry-analysis/reproductive-toxicity-testing-market">https://www.gminsights.com/industry-analysis/reproductive-toxicity-testing-market</ext-link>.</comment>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Predicting the reproductive toxicity of chemicals using ensemble learning methods and molecular fingerprints</article-title>. <source>Toxicol. Lett.</source> <volume>340</volume>, <fpage>4</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1016/j.toxlet.2021.01.002</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gholamy</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kreinovich</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Kosheleva</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Why 70/30 or 80/20 relation between training and testing sets: a pedagogical explanation</article-title>. <source>Dep. Tech. Rep. (CS)</source> <volume>1209</volume>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://scholarworks.utep.edu/cs_techrep/1209">https://scholarworks.utep.edu/cs_techrep/1209</ext-link>.</comment>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gilmer</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Schoenholz</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Riley</surname>
<given-names>P. F.</given-names>
</name>
<name>
<surname>Vinyals</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Dahl</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Neural message passing for quantum chemistry</article-title>. <source>Proc. 34th Int. Conf. Mach. Learn.</source> <volume>70</volume>, <fpage>1263</fpage>&#x2013;<lpage>1272</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1704.01212</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Directed message passing neural network (D-MPNN) with graph edge attention (GEA) for property prediction of biofuel-relevant species</article-title>. <source>Energy AI</source> <volume>10</volume>, <fpage>100201</fpage>. <pub-id pub-id-type="doi">10.1016/j.egyai.2022.100201</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Iavarone</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Dasmahapatra</surname>
<given-names>A. K.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Editorial: genotoxic pathways of reproductive outcomes</article-title>. <source>Front. Cell Dev. Biol.</source> <volume>13</volume>, <fpage>1614251</fpage>. <pub-id pub-id-type="doi">10.3389/fcell.2025.1614251</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Di</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>
<italic>In silico</italic> prediction of chemical reproductive toxicity using machine learning</article-title>. <source>J. Appl. Toxicol.</source> <volume>39</volume>, <fpage>844</fpage>&#x2013;<lpage>854</lpage>. <pub-id pub-id-type="doi">10.1002/jat.3772</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kearnes</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>McCloskey</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Berndl</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pande</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Riley</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Molecular graph convolutions: moving beyond fingerprints</article-title>. <source>J. Comput.-Aided Mol. Des.</source> <volume>30</volume>, <fpage>595</fpage>&#x2013;<lpage>608</lpage>. <pub-id pub-id-type="doi">10.1007/s10822-016-9938-8</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>A review of quantitative structure-activity relationship: the development and current status of data sets, molecular descriptors and mathematical models</article-title>. <source>Chemom. Intell. Lab. Syst.</source> <volume>256</volume>, <fpage>105278</fpage>. <pub-id pub-id-type="doi">10.1016/j.chemolab.2024.105278</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ning</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Global, regional, and national burden of female infertility and trends from 1990 to 2021 with projections to 2050 based on the GBD 2021 analysis</article-title>. <source>Sci. Rep.</source> <volume>15</volume>, <fpage>17559</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-025-01498-x</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>AbdulHameed</surname>
<given-names>M. D. M.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Clancy</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Desai</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Wallqvist</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Rapid screening of chemicals for their potential to cause specific toxidromes</article-title>. <source>Front. Drug Discov.</source> <volume>4</volume>. <pub-id pub-id-type="doi">10.3389/fddsv.2024.1324564</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Miller</surname>
<given-names>M. T.</given-names>
</name>
</person-group> (<year>1991</year>). <article-title>Thalidomide embryopathy: a model for the study of congenital incomitant horizontal strabismus</article-title>. <source>Trans. Am. Ophthalmol. Soc.</source> <volume>89</volume>, <fpage>623</fpage>&#x2013;<lpage>674</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>M&#xed;nguez-Alarc&#xf3;n</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gaskins</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Meeker</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Braun</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Chavarro</surname>
<given-names>J. E.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Endocrine-disrupting chemicals and male reproductive health</article-title>. <source>Fertil. Steril.</source> <volume>120</volume> (<issue>6</issue>), <fpage>1138</fpage>&#x2013;<lpage>1149</lpage>. <pub-id pub-id-type="doi">10.1016/j.fertnstert.2023.10.008</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Njagi</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Groot</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Arsenijevic</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Dyer</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mburu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kiarie</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Financial costs of assisted reproductive technology for patients in low- and middle-income countries: a systematic review</article-title>. <source>Hum. Reprod. Open</source> <volume>2023</volume> (<issue>2</issue>), <fpage>hoad007</fpage>. <pub-id pub-id-type="doi">10.1093/hropen/hoad007</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<collab>OECD</collab> (<year>2008</year>). <article-title>Guidance document on mammalian reproductive toxicity testing and assessment (EN)</article-title>
</citation>
</ref>
<ref id="B29">
<citation citation-type="book">
<collab>OECD</collab> (<year>2023</year>). <source>eChemPortal guidance for new participants, Environment, Health and Safety, Environment Directorate</source>. <publisher-name>Paris, France: Organisation for Economic Co-operation and Development</publisher-name>.</citation>
</ref>
<ref id="B30">
<citation citation-type="book">
<collab>Office of the Commissioner</collab> (<year>2025</year>). <source>FDA announces plan to phase out animal testing requirement for monoclonal antibodies and other drugs</source>. <publisher-name>Washington, DC: U.S. Food and Drug Administration</publisher-name>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.fda.gov/news-events/press-announcements/fda-announces-plan-phase-out-animal-testing-requirement-monoclonal-antibodies-and-other-drugs">https://www.fda.gov/news-events/press-announcements/fda-announces-plan-phase-out-animal-testing-requirement-monoclonal-antibodies-and-other-drugs</ext-link>.</comment>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>The adverse role of endocrine disrupting chemicals in the reproductive system</article-title>. <source>Front. Endocrinol.</source> <volume>14</volume>, <fpage>1324993</fpage>. <pub-id pub-id-type="doi">10.3389/fendo.2023.1324993</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Quantitative evaluation of explainable graph neural networks for molecular property prediction</article-title>. <source>Patterns</source> <volume>3</volume> (<issue>12</issue>), <fpage>100628</fpage>. <pub-id pub-id-type="doi">10.1016/j.patter.2022.100628</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>H. Y.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Y. M.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J. R.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>FGTN: fragment-based graph transformer network for predicting reproductive toxicity</article-title>. <source>Arch. Toxicol.</source> <volume>98</volume>, <fpage>4077</fpage>&#x2013;<lpage>4092</lpage>. <pub-id pub-id-type="doi">10.1007/s00204-024-03866-4</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rim</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Reproductive toxic chemicals at work and efforts to protect workers&#x27; health: a literature review</article-title>. <source>Saf. Health Work</source> <volume>8</volume>, <fpage>143</fpage>&#x2013;<lpage>150</lpage>. <pub-id pub-id-type="doi">10.1016/j.shaw.2017.04.003</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sheridan</surname>
<given-names>R. P.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W. M.</given-names>
</name>
<name>
<surname>Liaw</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gifford</surname>
<given-names>E. M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Extreme gradient boosting as a method for quantitative structure-activity relationships</article-title>. <source>J. Chem. Inf. Model.</source> <volume>56</volume>, <fpage>2353</fpage>&#x2013;<lpage>2360</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.6b00591</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>The role of epigenetics in the reproductive toxicity of environmental endocrine disruptors</article-title>. <source>Environ. Mol. Mutagen</source> <volume>62</volume> (<issue>1</issue>), <fpage>78</fpage>&#x2013;<lpage>88</lpage>. <pub-id pub-id-type="doi">10.1002/em.22414</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Niu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Communicative representation learning on attributed molecular graphs</article-title>,&#x201d; in <source>Proceedings of the 29th international joint conference on artificial intelligence</source>, <fpage>2831</fpage>&#x2013;<lpage>2838</lpage>.</citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<collab>United States Environmental Protection Agency</collab> (<year>2025</year>). &#x201c;<article-title>Predictive models and tools for assessing chemicals under the toxic substances control act (TSCA)</article-title>,&#x201d; in <source>Using predictive methods to assess exposure and fate under TSCA</source>. <publisher-name>Washington, DC: U.S. Environmental Protection Agency</publisher-name>.</citation>
</ref>
<ref id="B39">
<citation citation-type="book">
<collab>U.S. Environmental Protection Agency</collab> (<year>1996</year>). <source>Guidelines for reproductive toxicity risk assessment (risk assessment forum, EPA document No. 630/R-96/009A; federal register 61 FR 56274&#x2013;56322)</source>. <publisher-loc>Washington, DC</publisher-loc>: <publisher-name>S. Environmental Protection Agency</publisher-name>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<collab>U.S. Food and Drug Administration</collab> (<year>2011</year>). <article-title>Guidance document: reproductive and developmental toxicities -integrating study results to assess concerns</article-title>. <source>Reproductive Dev. Toxicities -- Integrating Study Results Assess Concerns &#x7c; FDA</source>.</citation>
</ref>
<ref id="B41">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>QSAR modeling based on graph neural networks</article-title>,&#x201d; in <source>QSAR in safety evaluation and risk assessment</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Hong</surname>
<given-names>H.</given-names>
</name>
</person-group> (<publisher-name>Academic Press</publisher-name>), <fpage>139</fpage>&#x2013;<lpage>151</lpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qian</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Phthalates and their impacts on human health</article-title>. <source>Healthc. (Basel). 18</source> <volume>9</volume> (<issue>5</issue>), <fpage>603</fpage>. <pub-id pub-id-type="doi">10.3390/healthcare9050603</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Machine learning based toxicity prediction: from chemical structural description to transcriptome analysis</article-title>. <source>Int. J. Mol. Sci.</source> <volume>19</volume>, <fpage>2358</fpage>. <pub-id pub-id-type="doi">10.3390/ijms19082358</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ramsundar</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Feinberg</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Gomes</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Geniesse</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Pappu</surname>
<given-names>A. S.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>MoleculeNet: a benchmark for molecular machine learning</article-title>. <source>Chem. Sci.</source> <volume>9</volume>, <fpage>513</fpage>&#x2013;<lpage>530</lpage>. <pub-id pub-id-type="doi">10.1039/c7sc02664a</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xia</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Niu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Drug-target binding affinity prediction using message passing neural network and self supervised learning</article-title>. <source>BMC Genomics</source> <volume>24</volume>, <fpage>557</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-023-09664-z</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>
<italic>In silico</italic> prediction of chemical Ames mutagenicity</article-title>. <source>J. Chem. Inf. Model.</source> <volume>52</volume>, <fpage>2840</fpage>&#x2013;<lpage>2847</lpage>. <pub-id pub-id-type="doi">10.1021/ci300400a</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jiao</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Toxic effects of zearalenone on gametogenesis and embryonic development: a molecular point of review</article-title>. <source>Food Chem. Toxicol.</source> <volume>119</volume>, <fpage>24</fpage>-30.<lpage>30</lpage>. <pub-id pub-id-type="doi">10.1016/j.fct.2018.06.003</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Swanson</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Coley</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Eiden</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Analyzing learned molecular representations for property prediction</article-title>. <source>J. Chem. Inf. Model.</source> <volume>59</volume>, <fpage>3370</fpage>&#x2013;<lpage>3388</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.9b00237</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yesildemir</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Celik</surname>
<given-names>M. N.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>The effect of various environmental pollutants on the reproductive health in children: a brief review of the literature</article-title>. <source>Curr. Nutr. Rep.</source> <volume>13</volume>, <fpage>382</fpage>&#x2013;<lpage>392</lpage>. <pub-id pub-id-type="doi">10.1007/s13668-024-00557-5</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mu</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Developing novel <italic>in silico</italic> prediction models for assessing chemical reproductive toxicity using the na&#xef;ve Bayes classifier method</article-title>. <source>J. Appl. Toxicol.</source> <volume>40</volume>, <fpage>1198</fpage>&#x2013;<lpage>1209</lpage>. <pub-id pub-id-type="doi">10.1002/jat.3975</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>