<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Pharmacol.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Pharmacology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Pharmacol.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">1663-9812</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1608843</article-id>
<article-id pub-id-type="doi">10.3389/fphar.2025.1608843</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Drug shortage in South Korea: machine learning-based prediction models and analysis of duration and causal factors</article-title>
<alt-title alt-title-type="left-running-head">Roe et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphar.2025.1608843">10.3389/fphar.2025.1608843</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Roe</surname>
<given-names>Hyun Soo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing - original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal Analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ko</surname>
<given-names>Da Eun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing - original draft</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Nam</surname>
<given-names>Seo Yun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing - original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Oh</surname>
<given-names>Jae Eun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing - original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Park</surname>
<given-names>Su Min</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Lee</surname>
<given-names>Se Hee</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3176911"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing - original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Lee</surname>
<given-names>Jong Hyuk</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2117157"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing - original draft</role>
</contrib>
</contrib-group>
<aff id="aff1">
<label>1</label>
<institution>College of Pharmacy, Chung-Ang University</institution>, <city>Seoul</city>, <country country="KR">Republic of Korea</country>
</aff>
<aff id="aff2">
<label>2</label>
<institution>College of Pharmacy, Sookmyung Women&#x2019;s University</institution>, <city>Seoul</city>, <country country="KR">Republic of Korea</country>
</aff>
<author-notes>
<corresp id="c001">
<label>&#x2a;</label>Correspondence: Jong Hyuk Lee, <email xlink:href="mailto:assajh@cau.ac.kr">assajh@cau.ac.kr</email>; Se Hee Lee, <email xlink:href="mailto:sellysh@naver.com">sellysh@naver.com</email>
</corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2026-01-09">
<day>09</day>
<month>01</month>
<year>2026</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1608843</elocation-id>
<history>
<date date-type="received">
<day>14</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>02</day>
<month>12</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>12</day>
<month>12</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2026 Roe, Ko, Nam, Oh, Park, Lee and Lee.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>Roe, Ko, Nam, Oh, Park, Lee and Lee</copyright-holder>
<license>
<ali:license_ref start_date="2026-01-09">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Drug shortages remain a critical challenge for healthcare systems in South Korea. This study aimed to develop predictive models to forecast drug shortage duration and identify their underlying causes.</p>
</sec>
<sec>
<title>Methods</title>
<p>Using 1,054 regulatory-reported drug shortage cases from 2018 to 2024 obtained from the Korean Ministry of Food and Drug Safety (KMFDS), we developed two machine learning models: (1) Model 1 to estimate shortage duration ranges, and (2) Model 2 to classify shortage causes into seven categories. Eighteen features related to drug shortages were included based on relevance and data availability. Key predictors were identified using Random Forest feature importance.</p>
</sec>
<sec>
<title>Results</title>
<p>Model 1 achieved an accuracy of 62%, with Shortage Incidence Frequency being the most influential variable (importance &#x003D; 0.152). In Model 2, weighted precision, recall, and F1-score all exceeded 70%, indicating robust performance despite imbalanced class distributions. The most important predictors for cause classification included Shortage Incidence Frequency, Existence of alternative drugs with the same ingredient, and Business size of the Marketing Authorization Holder.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>The KMFDS should continuously monitor drugs with repeated shortage episodes through regular reporting, early-warning systems, and supply-risk assessments. Incorporating supply-side indicators&#x2014;particularly those related to economic feasibility&#x2014;into national surveillance programs may help prevent shortages and mitigate their duration. By identifying key predictors associated with shortage causes, this study provides evidence to guide policy prioritization and targeted interventions.</p>
</sec>
</abstract>
<kwd-group>
<kwd>drug shortage</kwd>
<kwd>feature importance</kwd>
<kwd>public health</kwd>
<kwd>random Forest model</kwd>
<kwd>shortage cause</kwd>
<kwd>shortage duration</kwd>
<kwd>supply chain</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declared that financial support was received for this work and/or its publication. This work was supported by a grant (22183MFDS366) from Ministry of Food and Drug Safety of South Korea in 2022-2025.</funding-statement>
</funding-group>
<counts>
<fig-count count="3"/>
<table-count count="4"/>
<equation-count count="3"/>
<ref-count count="27"/>
<page-count count="10"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Drugs Outcomes Research and Policies</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<label>1</label>
<title>Introduction</title>
<p>The term &#x201c;drug shortage&#x201d; lacks a globally standardized definition, with significant variations or no definitions at all across countries (<xref ref-type="bibr" rid="B27">World Health Organization, 2015</xref>). From a supply perspective, drug shortages refer to the inadequate availability of medicines, health-related products, and vaccines required to meet public health and patient needs within the healthcare system. On the demand side, shortages arise when demand exceeds supply at any point in the supply chain, potentially leading to stockouts and an inability to meet clinical needs if not addressed promptly (<xref ref-type="bibr" rid="B1">Acosta et al., 2019</xref>; <xref ref-type="bibr" rid="B9">Health Canada, 2024</xref>).</p>
<p>Drug shortages impose significant societal costs on the healthcare sector, including reduced patient access to medicines, increased healthcare expenses, disruptions in the pharmaceutical markets, heightened anxiety among healthcare providers, and delays in developing innovative drugs (<xref ref-type="bibr" rid="B5">Fox et al., 2014</xref>; <xref ref-type="bibr" rid="B18">Pauwels et al., 2015</xref>; <xref ref-type="bibr" rid="B22">Shukar et al., 2021</xref>). During the COVID-19 pandemic, the global value chain for biopharmaceuticals faced unprecedented threats, intensifying global concerns about drug shortages. One study reported findings from a five-year study that emphasized the growing challenge of drug shortages and pointed the importance of proactive strategies, including investing in technology, strengthening supplier relationships, and advocating for policy reforms (<xref ref-type="bibr" rid="B2">Ayati et al., 2020</xref>; <xref ref-type="bibr" rid="B20">Sallam et al., 2024</xref>). Addressing this challenge is a critical issue for national healthcare systems (<xref ref-type="bibr" rid="B11">Ivanov and Das, 2020</xref>; <xref ref-type="bibr" rid="B2">Ayati et al., 2020</xref>; <xref ref-type="bibr" rid="B19">Piatek et al., 2020</xref>).</p>
<p>The prediction of drug shortages, time to shortage, and recovery period following a shortage, depending on the characteristics of the drug, are key factors in managing drug shortages (<xref ref-type="bibr" rid="B24">Tucker and Daskin, 2021</xref>). In Canada, a study was conducted to develop a machine-learning model to predict drug shortages using data from 22 pharmacy sales and historical drug shortage records (<xref ref-type="bibr" rid="B17">Pall et al., 2023</xref>). The model was able to predict shortage classes (none, low, medium, and high) with 69% accuracy 1&#xa0;month in advance. Despite the lack of inventory data from drug manufacturers and suppliers, this demonstrates meaningful performance given limited data.</p>
<p>In South Korea, the pharmaceutical industry predominantly produces generic drugs and relies heavily on imported active pharmaceutical ingredients (APIs) as well as new drugs, making the supply chain sensitive and vulnerable to shortages. To address these challenges, in 2019, the government developed a government-led drug supply disruption prediction model in collaboration with the Korea Orphan and Essential Drug Center, using data from reports on drug supply interruptions and shortages to address drug shortages (<xref ref-type="bibr" rid="B16">National Institute of Food and Drug Safety Evaluation, 2019</xref>). In addition, to proactively identify supply and demand imbalances, they implemented a supply and demand forecasting project utilizing artificial intelligence. To address these imbalances, they expanded the list of nationally essential drugs and reinforced support for production (<xref ref-type="bibr" rid="B21">Seoul Pharmaceutical Association, 2023</xref>). However, the limitations of this model-designed only to provide real-time alerts for potential shortages-prevented it from addressing the underlying problems, leading to a significant increase in the number of drugs with unstable supply despite the expansion of national essential drugs (<xref ref-type="bibr" rid="B16">National Institute of Food and Drug Safety Evaluation, 2019</xref>; <xref ref-type="bibr" rid="B21">Seoul Pharmaceutical Association, 2023</xref>).</p>
<p>Thus, a retrospective analysis of 1,054 regulatory-reported drug shortages predictive system is urgently required to effectively address drug shortages. While the previous model aimed to provide real-time alerts for potential shortages, we analyze drugs that have already undergone shortages, examining their shortage duration range and cause to identify key features relevant for prediction. Also, unlike the previous model, our dataset includes information not only on the causes of shortage but also on their duration ranges. Therefore, this study aims to develop two types of models: one for predicting the causes of drug shortages and another for estimating their duration range. Identifying key features with high importance in each model could help to determine the factors crucial for predicting the occurrence of shortages based on their causes and estimating their duration in South Korea. By providing not only early indications of whether a shortage may occur but also insights into why it happens and how long it is likely to persist, our models allow decision-makers to implement more strategic and timely interventions. These actionable insights support more targeted policy responses and early risk mitigation, ultimately contributing to a more resilient drug supply system in South Korea.</p>
</sec>
<sec sec-type="methods" id="s2">
<label>2</label>
<title>Methods</title>
<sec id="s2-1">
<label>2.1</label>
<title>Data source</title>
<p>Data on 1,054 cases of drug shortages reported by pharmaceutical companies to the Korean Ministry of Food and Drug Safety (KMFDS) between 2018 and 2024 were collected from the Drug Safety System on the KMFDS website (<ext-link ext-link-type="uri" xlink:href="https://www.mfds.go.kr/">https://www.mfds.go.kr</ext-link>). All 1054 cases reported during this period were included in the analysis.</p>
</sec>
<sec id="s2-2">
<label>2.2</label>
<title>Features</title>
<p>Eighteen variables of Model 1 and 2 related to drug shortages were selected (<xref ref-type="sec" rid="s14">Supplementary Table S1</xref>). These variables were chosen based on drug shortage factors (referred to as &#x201c;features&#x201d;) identified in Canada, the United States, and South Korea, as well as variables selected from the drug supply disruption prediction model in South Korea (<xref ref-type="bibr" rid="B17">Pall et al., 2023</xref>; <xref ref-type="bibr" rid="B16">National Institute of Food and Drug Safety Evaluation, 2019</xref>; <xref ref-type="bibr" rid="B25">United States Government Accountability Office, 2016</xref>; <xref ref-type="bibr" rid="B26">U.S. Food and Drug Administration FDA, 2020</xref>).</p>
<p>Features were selected based on the following criteria: (1) data available from open sources, such as the Drug Safety system in the KMFD, and (2) continuous or nominal variables with convertible numerical values. For the Year of approval, the corresponding decade was considered more meaningful than the specific year. Thus, data were categorized into three intervals (1960&#x2013;1970s, 1980&#x2013;1990s, and after the 2000s) to ensure an even distribution (<xref ref-type="bibr" rid="B22">Shukar et al., 2021</xref>; <xref ref-type="bibr" rid="B17">Pall et al., 2023</xref>; <xref ref-type="bibr" rid="B10">Health Insurance Review and Assessment Service, 2023</xref>; <xref ref-type="bibr" rid="B12">Liu et al., 2021</xref>). Decade-based grouping was used for data balance, but we acknowledge that this approach may obscure more detailed temporal variation. Drug shortage causes were systematically classified into seven major categories in this study as shown in <xref ref-type="sec" rid="s14">Supplementary Table S2</xref>.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Modeling</title>
<sec id="s3-1">
<label>3.1</label>
<title>Data processing</title>
<p>The raw dataset contained 1,054 records reported between 2018 and 2024. Ambiguous entries such as &#x201c;?&#x201d; or &#x201c;&#x2013;&#x201d; and remaining missing values were recoded as 0. For Model 1, shortages with missing duration (Class 0, 0.1%) were removed, leaving 1,053 records. The remaining five duration classes were used as ordered target categories: Class 0 (missing duration), Class 1 (&#x3c;1&#xa0;month), Class 2 (1&#x2013;6&#xa0;months), Class 3 (6&#x2013;12&#xa0;months), Class 4 (&#x2265;12&#xa0;months), and Class 5 (permanent discontinuation/no recovery). For Model 2, all 1,054 records were retained, and each shortage cause was modeled as a separate binary one-vs-rest classifier. The original multi-category &#x201c;Cause of shortage&#x201d; variable was therefore converted into six binary targets. All features were converted to numeric format, and the route of administration variable was one-hot encoded using get_dummies, as required for non-ordinal categorical inputs in Random Forest models. No normalization was applied to continuous variables, as tree-based models such as Random Forest are inherently scale-invariant.</p>
</sec>
<sec id="s3-2">
<label>3.2</label>
<title>Training and testing method</title>
<p>The RandomForestClassifier from sklearn. ensemble was used to ensemble multiple decision trees for Model 1 and Model 2, each splitting the data into two groups based on thresholds to reduce the Gini Index (<xref ref-type="bibr" rid="B16">National Institute of Food and Drug Safety Evaluation, 2019</xref>; <xref ref-type="bibr" rid="B13">Machado, Mendoza, and Corbellini, 2015</xref>). Random Forest was selected because it can handle nonlinear relationships and mixed feature types without requiring feature scaling. The dataset was divided into training and testing sets using a stratified 70:30 split to preserve class distributions. Each tree was constructed using random bootstrap samples from the training data (Out-of-Bag, OOB). Node splits were determined during tree growth by randomly selecting the split that minimized impurity from a random subset of candidate variables.</p>
<p>Instead of using GridSearchCV, which explores all hyperparameter combinations, the Bayesian Optimization from the bayes_opt library was used for hyperparameter tuning. This approach quantifies surrogate model uncertainty and identifies the next sampling location using an acquisition function (<xref ref-type="bibr" rid="B6">Frazier, 2018</xref>). In this study, the Bayesian Optimization (acquisition function) was employed to generate hyperparameter combinations (max_feature, max_samples, and n_depth) at each iteration. Bayesian Optimization was performed within predefined parameter ranges for tree depth, number of trees, and sampling proportions. Each hyperparameter combination was evaluated using stratified five-fold cross-validation, and the final model was selected based on the highest mean validation accuracy. Parameter search ranges were defined as follows: max_samples: 0.5&#x2013;1.0, max_features: 0.5&#x2013;1.0, n_estimators: 100&#x2013;300, max_depth: 3&#x2013;8 (<xref ref-type="fig" rid="F1">Figure 1</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Hyperparameter Tuning: number of parameters (n_estimator, max_features, n_depth)<xref ref-type="fn" rid="fn1">
<sup>1</sup>
</xref>.</p>
</caption>
<graphic xlink:href="fphar-16-1608843-g001.tif">
<alt-text content-type="machine-generated">Two line graphs side by side. The left graph shows accuracy versus depth, with a solid blue line for train score and a dashed orange line for test score. The train score increases with depth, while the test score stabilizes. The right graph shows different parameter scores. A light blue dashed line represents n_estimators train score, and an orange dashed line shows max_features test score, with slight variations around 0.5 to 0.9.</alt-text>
</graphic>
</fig>
<p>The performance of the Shortage Duration Prediction Model (Model 1) and six Shortage Occurrence Prediction Model (Model 2) was evaluated using classification evaluation metrics (precision, recall, and accuracy). Accuracy measures the proportion of correct predictions, precision reflects the proportion of actual positives among the predicted positive samples, and F1_score is the harmonic mean of precision and recall. These metrics were calculated using the macro-average, considering that both models are multiclass classifications (<xref ref-type="bibr" rid="B4">Chicco and Jurman, 2020</xref>; <xref ref-type="bibr" rid="B3">Caelen, 2017</xref>). Although duration classes were imbalanced, we reported macro-averaged precision, recall, and F1-scores to provide a balanced assessment across classes. The test set was kept fully isolated and was not used during model training or cross-validation to prevent information leakage. The training and testing data were generated using the train_test_split function, with the test_size set to 0.3.<disp-formula id="equ1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>_</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>&#xb7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
</sec>
</sec>
<sec sec-type="results" id="s4">
<label>4</label>
<title>Results</title>
<p>
<xref ref-type="table" rid="T1">Table 1</xref> shows a summary of 1,054 cases of drug shortages reported by pharmaceutical companies to the KMFDS between 2018 and 2024. The data were categorized based on key attributes, including those related to drug shortage events, drug supply monitoring system, drug manufacturing, and drug characteristics.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Summary of drug shortage cases in South Korea.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">a. Related with drug shortage events</th>
<th align="left">Count</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td colspan="2" align="left">Shortage incidence frequency (n &#x3d; 888<xref ref-type="fn" rid="fn2">
<sup>2</sup>
</xref>)</td>
</tr>
<tr>
<td align="left">1 time</td>
<td align="left">777</td>
</tr>
<tr>
<td align="left">2 times</td>
<td align="left">80</td>
</tr>
<tr>
<td align="left">3 times</td>
<td align="left">18</td>
</tr>
<tr>
<td align="left">4 times</td>
<td align="left">7</td>
</tr>
<tr>
<td align="left">5 times</td>
<td align="left">4</td>
</tr>
<tr>
<td align="left">6 times</td>
<td align="left">1</td>
</tr>
<tr>
<td align="left">9 times</td>
<td align="left">1</td>
</tr>
<tr>
<td colspan="2" align="left">Shortage cause (n &#x3d; 1116<xref ref-type="fn" rid="fn3">
<sup>3</sup>
</xref>)</td>
</tr>
<tr>
<td align="left">a. Increased demand</td>
<td align="left">83</td>
</tr>
<tr>
<td align="left">b. Decreased demand</td>
<td align="left">6</td>
</tr>
<tr>
<td align="left">c. Troubles in raw material supply</td>
<td align="left">184</td>
</tr>
<tr>
<td align="left">d. Regulatory issues</td>
<td align="left">113</td>
</tr>
<tr>
<td align="left">e. Supply chain management issues</td>
<td align="left">272</td>
</tr>
<tr>
<td align="left">f. Business decision</td>
<td align="left">429</td>
</tr>
<tr>
<td align="left">g. Others/Unknown</td>
<td align="left">29</td>
</tr>
<tr>
<td colspan="2" align="left">Shortage duration range (n &#x3d; 1053<xref ref-type="fn" rid="fn4">
<sup>4</sup>
</xref>)</td>
</tr>
<tr>
<td align="left">0&#x2013;30&#xa0;days (&#x3c;1&#xa0;month)</td>
<td align="left">100</td>
</tr>
<tr>
<td align="left">31&#x2013;180&#xa0;days (1 month to &#x3c;6&#xa0;months)</td>
<td align="left">196</td>
</tr>
<tr>
<td align="left">181&#x2013;360&#xa0;days (6 months to &#x3c;12&#xa0;months)</td>
<td align="left">34</td>
</tr>
<tr>
<td align="left">361&#xa0;days or more (&#x2265;12&#xa0;months)</td>
<td align="left">331</td>
</tr>
<tr>
<td align="left">Suspension</td>
<td align="left">392</td>
</tr>
<tr>
<td colspan="2" align="left">Shortage incidence timing as of COVID pandemic declaration (n &#x3d; 1054)</td>
</tr>
<tr>
<td align="left">Before-pandemic</td>
<td align="left">669</td>
</tr>
<tr>
<td align="left">After-pandemic</td>
<td align="left">385</td>
</tr>
</tbody>
</table>
<table>
<thead valign="top">
<tr>
<th align="left">b. Related with drug supply monitoring system (n &#x3d; 1054)</th>
<th align="left">Count (Yes/No/Unknown)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Drugs required mandatory supply maintenance</td>
<td align="left">(112/ 940/ 2)</td>
</tr>
<tr>
<td align="left">Essential drugs designated by South Korea</td>
<td align="left">(828/ 226/ 0)</td>
</tr>
<tr>
<td align="left">Essential drugs designated by WHO</td>
<td align="left">(731/ 323/ 0)</td>
</tr>
<tr>
<td align="left">Drugs required supply discontinuation reporting</td>
<td align="left">(244/ 810/ 0)</td>
</tr>
</tbody>
</table>
<table>
<thead valign="top">
<tr>
<th align="left">c. Related with drug manufacturing</th>
<th align="left">Count</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td colspan="2" align="left">Imported/Domestic</td>
</tr>
<tr>
<td align="left">Imported products</td>
<td align="left">523</td>
</tr>
<tr>
<td align="left">Domestic products</td>
<td align="left">463</td>
</tr>
<tr>
<td align="left">Unknown</td>
<td align="left">68</td>
</tr>
<tr>
<td colspan="2" align="left">Type of manufacturing site</td>
</tr>
<tr>
<td align="left">Contract manufacturing organization (CMO)</td>
<td align="left">761</td>
</tr>
<tr>
<td align="left">In-house manufacture</td>
<td align="left">293</td>
</tr>
<tr>
<td colspan="2" align="left">Business size of the marketing authorized company</td>
</tr>
<tr>
<td align="left">Small (&#x3c;100 billion KRW)</td>
<td align="left">253</td>
</tr>
<tr>
<td align="left">Medium (100 billion to &#x3c;1 trillion KRW)</td>
<td align="left">449</td>
</tr>
<tr>
<td align="left">Large (&#x2265;1 trillion KRW)</td>
<td align="left">82</td>
</tr>
<tr>
<td align="left">Unknown</td>
<td align="left">270</td>
</tr>
</tbody>
</table>
<table>
<thead valign="top">
<tr>
<th align="left">d. Related with drug characteristics</th>
<th align="left">Count</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td colspan="2" align="left">Drug classification: OTC vs. ETC</td>
</tr>
<tr>
<td align="left">ETC</td>
<td align="left">123</td>
</tr>
<tr>
<td align="left">OTC</td>
<td align="left">931</td>
</tr>
<tr>
<td colspan="2" align="left">Drug classification: Type of ingredients</td>
</tr>
<tr>
<td align="left">Chemical</td>
<td align="left">897</td>
</tr>
<tr>
<td align="left">Bio</td>
<td align="left">143</td>
</tr>
<tr>
<td align="left">Oriental medicine</td>
<td align="left">14</td>
</tr>
<tr>
<td colspan="2" align="left">Single-agent drug vs. combination drug</td>
</tr>
<tr>
<td align="left">Single-agent drug</td>
<td align="left">915</td>
</tr>
<tr>
<td align="left">Combination drug</td>
<td align="left">139</td>
</tr>
<tr>
<td colspan="2" align="left">Routes of drug administration</td>
</tr>
<tr>
<td align="left">Oral</td>
<td align="left">508</td>
</tr>
<tr>
<td align="left">Injectable</td>
<td align="left">385</td>
</tr>
<tr>
<td align="left">Topical</td>
<td align="left">153</td>
</tr>
<tr>
<td align="left">Other</td>
<td align="left">1</td>
</tr>
<tr>
<td align="left">Unknown</td>
<td align="left">7</td>
</tr>
<tr>
<td colspan="2" align="left">Year of approval</td>
</tr>
<tr>
<td align="left">1960s&#x2013;1970s</td>
<td align="left">44</td>
</tr>
<tr>
<td align="left">1980s&#x2013;1990s</td>
<td align="left">293</td>
</tr>
<tr>
<td align="left">&#x2265;2000s</td>
<td align="left">717</td>
</tr>
<tr>
<td colspan="2" align="left">Existence of alternative drugs with same ingredients</td>
</tr>
<tr>
<td align="left">Yes</td>
<td align="left">323</td>
</tr>
<tr>
<td align="left">No</td>
<td align="left">634</td>
</tr>
<tr>
<td align="left">Unknown</td>
<td align="left">972</td>
</tr>
<tr>
<td colspan="2" align="left">National health insurance reimbursement</td>
</tr>
<tr>
<td align="left">Reimbursable</td>
<td align="left">753</td>
</tr>
<tr>
<td align="left">Non-reimbursable</td>
<td align="left">257</td>
</tr>
<tr>
<td align="left">Delisted<xref ref-type="fn" rid="fn5">
<sup>5</sup>
</xref>
</td>
<td align="left">44</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s4-1">
<label>4.1</label>
<title>Model 1 (Shortage duration range prediction model)</title>
<p>A baseline RandomForestClassifier using default hyperparameters achieved 0.43 accuracy under the same train&#x2013;test split. After Bayesian Optimization, the optimized Model 1 reached 0.62 accuracy, demonstrating clear improvement over the baseline. <xref ref-type="fig" rid="F2">Figure 2</xref> and <xref ref-type="table" rid="T2">Table 2</xref> shows that the optimized model reduced misclassification across several duration categories compared with the baseline, supporting the observed performance gain (<xref ref-type="table" rid="T2">Table 2</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Confusion matrix per class (Model 1).</p>
</caption>
<graphic xlink:href="fphar-16-1608843-g002.tif">
<alt-text content-type="machine-generated">Confusion matrix illustrating the performance of a classification model. True labels range from one to five and predicted labels range from one to five. The highest counts are 72 at true label five and predicted label five, and 55 at true label four and predicted label four. A color gradient from purple to yellow indicates frequency, with yellow representing higher values.</alt-text>
</graphic>
</fig>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Classification report and confusion matrix of shortage duration range prediction model (Model 1).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Class (duration)</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1-score</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">0&#x2013;30 days (&#x3c;1&#xa0;month)</td>
<td align="left">1.00</td>
<td align="left">0.03</td>
<td align="left">0.05</td>
</tr>
<tr>
<td align="left">31&#x2013;180&#xa0;days (1&#xa0;month to &#x3c;6&#xa0;months)</td>
<td align="left">0.54</td>
<td align="left">0.67</td>
<td align="left">0.60</td>
</tr>
<tr>
<td align="left">181&#x2013;360&#xa0;days (6&#xa0;months to &#x3c;12&#xa0;months)</td>
<td align="left">1.00</td>
<td align="left">0.17</td>
<td align="left">0.29</td>
</tr>
<tr>
<td align="left">361&#xa0;days or more (&#x2265;12&#xa0;months)</td>
<td align="left">0.61</td>
<td align="left">0.54</td>
<td align="left">0.58</td>
</tr>
<tr>
<td align="left">Suspension</td>
<td align="left">0.66</td>
<td align="left">0.88</td>
<td align="left">0.75</td>
</tr>
<tr>
<td align="center">Accuracy</td>
<td align="left">&#x200b;</td>
<td align="left">&#x200b;</td>
<td align="left">0.62</td>
</tr>
<tr>
<td align="center">Macro average</td>
<td align="left">0.76</td>
<td align="left">0.46</td>
<td align="left">0.45</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The three features with the highest importance for predicting the Shortage Duration Range were the Shortage Incidence frequency (0.152), Imported/Domestic (0.106), and Existence of alternative drugs with same ingredients (0.093). Business size of the Marketing Authorized Company (0.081) and National Health Insurance reimbursement (0.073) also showed high importance (<xref ref-type="fig" rid="F3">Figure 3</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Feature importance of shortage duration range prediction model (model 1).</p>
</caption>
<graphic xlink:href="fphar-16-1608843-g003.tif">
<alt-text content-type="machine-generated">A table and bar graph showing the importance of various features in drug shortage incidence. Key features include &#x22;Number of drug shortage incidence&#x22; with the highest importance of 0.152002, followed by &#x22;Imported/Manufactured&#x22; at 0.105971. Other features like &#x22;Drugs with the same active ingredients,&#x22; &#x22;Company size,&#x22; and &#x22;Insurance benefit&#x22; are also listed with descending importance values. The bar graph visually represents these importance values, aligning with the table data.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s4-2">
<label>4.2</label>
<title>Model 2 (Shortage occurrence prediction model by each shortage cause)</title>
<p>Across all six one-vs-rest models predicting individual shortage causes, accuracy exceeded 0.70 after Bayesian Optimization, indicating stable performance across categories. Weighted precision, recall, and F1-scores also remained above 0.70, demonstrating that each model reliably identified its corresponding shortage cause (<xref ref-type="table" rid="T3">Table 3</xref>). <xref ref-type="table" rid="T4">Table 4</xref> shows the feature importance values of the top five features in each model. <xref ref-type="sec" rid="s14">Supplementary Table S2</xref> shows the feature importance values for all features. <xref ref-type="sec" rid="s14">Supplementary Figure S1</xref> visualizes feature importances for all features.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Classification report of shortage occurrence prediction model by each shortage cause (Model 2).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Shortage cause</th>
<th align="center">Accuracy</th>
<th align="center">Precision (macro/Weighted)</th>
<th align="center">Recall (macro/Weighted)</th>
<th align="center">F1-score (macro/Weighted)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Increased demand</td>
<td align="left">0.93</td>
<td align="left">0.97/0.94</td>
<td align="left">0.53/0.93</td>
<td align="left">0.55/0.91</td>
</tr>
<tr>
<td align="left">Decreased demand</td>
<td align="left">0.99</td>
<td align="left">0.50/0.98</td>
<td align="left">0.50/0.99</td>
<td align="left">0.50/0.99</td>
</tr>
<tr>
<td align="left">Troubles in raw material supply</td>
<td align="left">0.80</td>
<td align="left">0.90/0.84</td>
<td align="left">0.51/0.80</td>
<td align="left">0.47/0.71</td>
</tr>
<tr>
<td align="left">Regulatory issues</td>
<td align="left">0.95</td>
<td align="left">0.47/0.90</td>
<td align="left">0.50/0.95</td>
<td align="left">0.49/0.92</td>
</tr>
<tr>
<td align="left">Supply chain management issues</td>
<td align="left">0.77</td>
<td align="left">0.77/0.77</td>
<td align="left">0.60/0.77</td>
<td align="left">0.61/0.72</td>
</tr>
<tr>
<td align="left">Business decision</td>
<td align="left">0.74</td>
<td align="left">0.73/0.74</td>
<td align="left">0.73/0.74</td>
<td align="left">0.73/0.74</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Top five feature importances of shortage occurrence prediction model by each shortage cause (Model 2).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Features</th>
<th align="center">Feature importance</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td colspan="2" align="left">Cause a. Increased demand (due to the COVID-19 pandemic and other reasons)</td>
</tr>
<tr>
<td align="left">Shortage incidence frequency</td>
<td align="left">0.159860</td>
</tr>
<tr>
<td align="left">Year of approval</td>
<td align="left">0.077193</td>
</tr>
<tr>
<td align="left">Business size of the marketing authorized company</td>
<td align="left">0.076793</td>
</tr>
<tr>
<td align="left">Existence of alternative drugs with same ingredients</td>
<td align="left">0.072283</td>
</tr>
<tr>
<td align="left">Shortage incidence timing as of COVID pandemic declaration</td>
<td align="left">0.069857</td>
</tr>
<tr>
<td colspan="2" align="left">Cause b. Decreased demand (patent expiration, new competitors, and developments of substitutes)</td>
</tr>
<tr>
<td align="left">Existence of alternative drugs with same ingredients</td>
<td align="left">0.212930</td>
</tr>
<tr>
<td align="left">Essential drugs designated by WHO</td>
<td align="left">0.140258</td>
</tr>
<tr>
<td align="left">Drugs required supply discontinuation reporting</td>
<td align="left">0.135990</td>
</tr>
<tr>
<td align="left">Imported/Domestic</td>
<td align="left">0.102378</td>
</tr>
<tr>
<td align="left">Type of manufacturing site</td>
<td align="left">0.062515</td>
</tr>
<tr>
<td colspan="2" align="left">Cause c. Troubles in raw material supply (shortage, quality degradation, delivery delay, price increase, and contract expiration raw materials)</td>
</tr>
<tr>
<td align="left">Shortage incidence frequency</td>
<td align="left">0.123844</td>
</tr>
<tr>
<td align="left">Business size of the marketing authorized company</td>
<td align="left">0.108332</td>
</tr>
<tr>
<td align="left">Existence of alternative drugs with same ingredients</td>
<td align="left">0.093989</td>
</tr>
<tr>
<td align="left">Year of approval</td>
<td align="left">0.071775</td>
</tr>
<tr>
<td align="left">National health insurance reimbursement</td>
<td align="left">0.063433</td>
</tr>
<tr>
<td colspan="2" align="left">Cause d. Regulatory issues (license revocation, suspension of production/import/sales, recall/destruction, import/customs issues)</td>
</tr>
<tr>
<td align="left">Business size of the marketing authorized company</td>
<td align="left">0.104231</td>
</tr>
<tr>
<td align="left">Existence of alternative drugs with same ingredients</td>
<td align="left">0.082905</td>
</tr>
<tr>
<td align="left">Imported/Domestic</td>
<td align="left">0.081812</td>
</tr>
<tr>
<td align="left">Year of approval</td>
<td align="left">0.074065</td>
</tr>
<tr>
<td align="left">National health insurance reimbursement</td>
<td align="left">0.072294</td>
</tr>
<tr>
<td colspan="2" align="left">Cause e. Supply chain management issues (capacity and production line limitations, manufacturing site loss and change, facility obsolescence and failure, and product defects)</td>
</tr>
<tr>
<td align="left">Shortage incidence frequency</td>
<td align="left">0.110314</td>
</tr>
<tr>
<td align="left">Business size of the marketing authorized company</td>
<td align="left">0.104795</td>
</tr>
<tr>
<td align="left">Imported/Domestic</td>
<td align="left">0.078509</td>
</tr>
<tr>
<td align="left">Year of approval</td>
<td align="left">0.076567</td>
</tr>
<tr>
<td align="left">Existence of alternative drugs with same ingredients</td>
<td align="left">0.072971</td>
</tr>
<tr>
<td colspan="2" align="left">Cause f. Business decision (mergers and acquisitions, new product for substitution launches, low profitability, management difficulties, and deteriorating management)</td>
</tr>
<tr>
<td align="left">Business size of the marketing authorized company</td>
<td align="left">0.115123</td>
</tr>
<tr>
<td align="left">Shortage incidence frequency</td>
<td align="left">0.096659</td>
</tr>
<tr>
<td align="left">Existence of alternative drugs with same ingredients</td>
<td align="left">0.077964</td>
</tr>
<tr>
<td align="left">Imported/Domestic</td>
<td align="left">0.076190</td>
</tr>
<tr>
<td align="left">Year of approval</td>
<td align="left">0.075517</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The top five features with the highest importance in predicting shortages caused by &#x201c;Shortage Cause a&#x201d; (Increased demand) were Shortage Incidence frequency (0.160), Year of Approval (0.077), Business size of the Marketing Authorized Company (0.077), Existence of alternative drugs with same ingredients (0.0723), and Shortage incidence timing as of COVID pandemic declaration (0.070).</p>
<p>The top four features with the highest importance in predicting shortages caused by &#x201c;Shortage Cause b&#x201d; (Decreased demand) were Existence of alternative drugs with same ingredients (0.213), Essential drugs designated by WHO (0.140), Drugs required supply discontinuation reporting (0.136), and Imported/Domestic (0.102).</p>
<p>The top three features with the highest importance in predicting shortages caused by &#x201c;Shortage Cause c&#x201d; (Troubles in raw material supply) were Shortage Incidence frequency (0.124), Business size of the Marketing Authorized Company (0.108), and Existence of alternative drugs with same ingredients (0.094).</p>
<p>The most important feature in predicting shortage caused by &#x201c;Shortage Cause d&#x201d; (Regulatory issues) was Business size of the Marketing Authorized Company (0.104). Compared to the prediction models for other shortage causes, the feature importance values were relatively uniform. The difference in importance among features, excluding Business size of the Marketing Authorized Company (the most important), Routes of drug administration_0 and Routes of drug administration_4 (the least important), was &#x3c;0.01.</p>
<p>The most important feature in predicting shortage caused by &#x201c;Shortage Cause e&#x201d; (Supply chain management issues) was Shortage Incidence frequency (0.110), followed by Business size of the Marketing Authorized Company (0.105).</p>
<p>The most important feature in predicting shortage caused by &#x201c;Shortage Cause f&#x201d; (Business decision) was Business size of the Marketing Authorized Company (0.115), followed by the Shortage Incidence frequency (0.097).</p>
<p>Taken together, the results show that Shortage incidence frequency and Business size of the Marketing Authorization Holder consistently ranked among the top predictors across several shortage causes, indicating their broad influence. In contrast, decreased-demand shortages showed high importance for substitutability-related variables, reflecting the different dynamics underlying this category. Overall, the feature-importance patterns suggest that each shortage cause is associated with a distinct set of determinants.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s5">
<label>5</label>
<title>Discussion</title>
<p>The study found that the primary factor for predicting the shortage duration range in South Korea is the Shortage Incidence frequency. A repeated history of shortages reflects persistent supply-side vulnerabilities&#x2014;such as limited manufacturing capacity, unstable access to APIs, insufficient redundancy across suppliers, or recurring quality-control issues&#x2014;which remain unresolved over time and increase the likelihood that the same product will experience future disruptions.</p>
<p>The results indicated that the Shortage Incidence frequency, Business size of the Marketing Authorized Company, and Existence of alternative drugs with same ingredients were highly important in most shortage causes. These findings suggest that products marketed by larger Marketing Authorized Companies tend to have more stable production capability, whereas drugs that do not have therapeutically equivalent alternatives face greater risk during supply interruptions. Because these features are all supply-side characteristics, the results indicate that production and supply-chain constraints exert a stronger influence on drug shortages in South Korea than demand-side factors.</p>
<p>The study period (2018&#x2013;2024) encompassed both pre-pandemic and post-pandemic environments, enabling evaluation of whether COVID-19&#x2013;related supply-chain disruptions altered shortage determinants. Although national shortage counts fluctuated during the pandemic, the ranking of key predictors remained stable, indicating that the determinants identified in this study represent underlying structural characteristics rather than temporary pandemic-specific effects.</p>
<p>Drug shortages arise from a complex interplay of factors, encompassing supply and demand issues, regulatory challenges, and quality concerns, with various risks and frequencies depending on the drug characteristics (<xref ref-type="bibr" rid="B14">Mazer-Amirshahi et al., 2014</xref>). These factors vary depending on national healthcare systems, pharmaceutical market structures, and government regulations (<xref ref-type="bibr" rid="B8">Gray and Manasse, 2012</xref>; <xref ref-type="bibr" rid="B1">Acosta et al., 2019</xref>). The risk and occurrence patterns of drug shortages also differ according to manufacturer characteristics, domestic production versus importation status, therapeutic class, route of administration, and whether a drug is branded or a generic product (<xref ref-type="bibr" rid="B5">Fox et al., 2014</xref>; <xref ref-type="bibr" rid="B7">Giammona et al., 2020</xref>). Generic drugs, in particular, are more susceptible to shortages due to low profitability and limited production capacity, and this susceptibility is further exacerbated by quality-control issues and the fragile supply chain of APIs (<xref ref-type="bibr" rid="B5">Fox et al., 2014</xref>; <xref ref-type="bibr" rid="B25">United States Government Accountability Office, 2016</xref>).</p>
<p>The U.S. pharmaceutical market faces challenges (factors) that drive drug shortages, including limited incentives for producing low-margin medicines, inadequate evaluation and compensation for quality control in manufacturing, and logistical and regulatory challenges in ensuring a stable drug supply (<xref ref-type="bibr" rid="B26">U.S. Food and Drug Administration, 2020</xref>). In contrast, most drug supply disruptions in Canada arise from supply-related issues, including pharmaceutical quality control problems, production delays, recalls, regulatory actions, product suspensions, and the unavailability of raw materials (<xref ref-type="bibr" rid="B23">The Multi-Stakeholder Steering Committee on Drug Shortages, 2017</xref>). In China, drug supply disruptions are primarily attributed to unprofitable pricing. Price competition in the drug market and government-imposed price reduction policies, which lower prices to unprofitable levels, are identified as key contributors to drug shortages (<xref ref-type="bibr" rid="B28">Yang et al., 2016</xref>). Previous research also reports that middle-income countries experience additional challenges such as licensing delays and limited raw materials, while low-income countries face more severe shortages due to insufficient research capacity and weak policy frameworks (<xref ref-type="bibr" rid="B8">Gray and Manasse, 2012</xref>). These findings show that drug shortages in many countries are driven predominantly by supply-side factors, which is consistent with the pattern observed in South Korea.</p>
<p>In high-income nations, manufacturing issues, business decisions, raw material shortages, and regulatory challenges are the primary factors contributing to drug shortages. In middle-income countries, similar issues prevail, but additional factors, such as licensing delays and insufficient raw materials for local manufacturers, also play a significant role. In contrast, low-income countries face exacerbated shortages due to limited research, insufficient data, and weak policy frameworks (<xref ref-type="bibr" rid="B8">Gray and Manasse, 2012</xref>). These international patterns demonstrate that drug shortages are largely driven by supply-side vulnerabilities, which aligns with the situation in South Korea, where supply-side factors were also found to be highly influential in this study.</p>
<p>Research on drug shortages has been more extensive in developed nations than in South Korea, and many of the insights identified in these countries are relevant to the Korean context. Therefore, addressing drug shortages in South Korea may benefit from applying and adapting strategies that have been effective in other high-income settings.</p>
<p>Complex machine-learning models often show strong internal fit but are vulnerable to overfitting, particularly when class distributions are imbalanced or when the number of predictors is large relative to the sample size. The risk of overfitting is especially important, as specific patterns learned from past data could hinder early recognition of new or atypical shortage signals. To mitigate this risk, the study incorporated multiple safeguards. First, a stratified 70/30 train&#x2013;test split was used to preserve class proportions, and the test set was kept entirely independent during model development to prevent information leakage. Second, Random Forest&#x2013;specific techniques such as bootstrap aggregation and out-of-bag (OOB) validation helped assess model performance from quasi-independent samples. Third, Bayesian optimization was applied to tune key hyperparameters including tree depth, sample size, and feature sampling rate, thereby preventing overly complex trees and improving generalizability. Collectively, these techniques reduced the likelihood of overfitting and contributed to the robustness of the final prediction models.</p>
<p>Since drug shortages are closely related to both supply and demand factors, the feature analysis of the study would be enhanced by integrating data from the supply side (i.e., pharmaceutical companies) for modeling. In addition, when the distribution of categorical features is skewed toward &#x201c;&#x27;0,&#x201d; indicating &#x201c;unknown,&#x201d; using out-of-bag bootstrap sampling may lead to the construction of a random forest model where many trees exclude data other than &#x201c;0&#x201d; from their samples. Among the 1,054 products analyzed in this study, only seven (0.66%) had an &#x201c;unknown&#x201d; route-of-administration label, a level of missingness too small to meaningfully influence model performance. Therefore, the low feature importance of Routes of drug administration in the Shortage Occurrence Prediction Models is likely attributable to the inherently limited relevance of the route-of-administration variable to shortage mechanisms rather than distortions caused by missing data.</p>
<p>This study showed features with high importance for predicting shortage duration causes based on drug shortage data in South Korea. Identifying the importance of each factor for predicting shortage duration helps to identify the duration when a shortage occurs and facilitates the rapid normalization of the drug supply. Analyzing the key features with high importance for each shortage cause would help establish strategies to identify the root cause of future shortages. With additional integration of supply-chain and production data from manufacturers, the proposed models could be interfaced with South Korea&#x2019;s national drug shortage monitoring system to support real-time risk identification and regulatory response. Such applications would contribute to healthcare system resilience by shifting shortage management from reactive to proactive approaches.</p>
<p>Conducting shortage pattern research using data from various organizations within the drug market, such as pharmaceutical companies, community pharmacies, and hospital pharmacies, along with publicly available data, may contribute to the stability of drug supply in South Korea. Future refinement and practical implementation of the models will require collaboration among government regulators, marketing authorization holders and manufacturers, wholesalers, hospital and community pharmacies, and academic researchers. Integrating data across these stakeholders will help establish a more coordinated national framework for drug shortage prevention and management.</p>
</sec>
<sec id="s6">
<label>6</label>
<title>Limitation</title>
<p>This study was conducted to analyze drug shortages in South Korea in depth but has some limitations. First, this study selected 18 features based on data accessibility and relevance, but potentially important factors such as changes in patient behavior (e.g., changes in purchasing patterns or drug compliance), global supply-chain disruptions, and policy changes were not included due to data collection constraints. These factors were difficult to measure consistently and were not available in standardized, detailed datasets. Second, this study utilized data from 2018 to 2024, and data collection within a limited period may not fully reflect long-term trends in drug shortages. Third, the prediction model developed in this study mainly focused on supply-side factors, so interactions with demand-side factors could not be fully considered. Fourth, the distributions of shortage duration and shortage-cause categories were uneven, which may have biased the prediction models toward majority classes. We acknowledge that the current model does not incorporate specific imbalance-handling techniques (e.g., class-weight adjustments or resampling), which may limit predictive performance for rare but meaningful shortage types. Future work should explore tailored imbalance mitigation strategies to improve minority-class sensitivity. Fifth, external validation using independent datasets&#x2014;such as those from hospitals, wholesalers, or community pharmacies&#x2014;was not performed, limiting the ability to assess model performance in real-world or cross-institutional settings. Sixth, because the model was trained solely on South Korean regulatory data, its generalizability to other health systems with different pharmaceutical structures may be limited. Lastly, the predictive model proposed in this study requires additional empirical studies to verify real-world applicability and effectiveness. These limitations are expected to be supplemented in follow-up studies, thereby enhancing the accuracy and practicality of drug shortage prediction models and contributing to the development of more effective policies.</p>
</sec>
<sec sec-type="conclusion" id="s7">
<label>7</label>
<title>Conclusion</title>
<p>This study developed two machine-learning models&#x2014;one predicting shortage duration ranges and another predicting shortage occurrence by cause&#x2014;and identified shortage incidence frequency as the most important predictor of future shortage duration. South Korea must implement an integrated data-sharing system among stakeholders to enable a swift response to future shortages and ensure their rapid resolution. Data should be shared transparently and consistently to complement the current incomplete reporting system. Drugs that have experienced several shortages should have their supply chain continuously monitored. Furthermore, the shortage prediction program in South Korea, which focuses on supply-side features such as economic viability, may help prevent shortages or reduce their duration and contribute to strengthening healthcare system resilience. Effective nationwide implementation will require collaboration among government regulators, marketing authorization holders, manufacturers, wholesalers, hospital and community pharmacies, and academic researchers.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s8">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s9">
<title>Author contributions</title>
<p>HR: Conceptualization, Writing &#x2013; review and editing, Investigation, Writing &#x2013; original draft, Validation, Formal Analysis, Data curation. DK: Writing &#x2013; review and editing, Data curation, Investigation, Writing &#x2013; original draft. SN: Writing &#x2013; original draft, Data curation, Investigation, Writing &#x2013; review and editing. JO: Data curation, Writing &#x2013; review and editing, Writing &#x2013; original draft, Investigation. SP: Writing &#x2013; review and editing. SL: Writing &#x2013; review and editing, Validation, Writing &#x2013; original draft, Project administration. JL: Investigation, Methodology, Supervision, Writing &#x2013; review and editing, Conceptualization, Writing &#x2013; original draft.</p>
</sec>
<sec sec-type="COI-statement" id="s11">
<title>Conflict of interest</title>
<p>The author(s) declared that this work was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s12">
<title>Generative AI statement</title>
<p>The author(s) declared that generative AI was not used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s13">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s14">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fphar.2025.1608843/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fphar.2025.1608843/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Supplementaryfile1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn fn-type="abbr" id="abbrev1">
<label>Abbreviations:</label>
<p>APIs, Active pharmaceutical ingredients; CMO, Contract Manufacturing Organization; OTC, Over the counter; ETC, Ethical drug; KMFDS, Korean Ministry of Food and Drug Safety; Out of Bag, OOB.</p>
</fn>
</fn-group>
<fn-group>
<fn id="fn1">
<label>1</label>
<p>The left graph illustrates how the stratified_k_fold score changes with variations in n_estimators and max_features in the untuned raw model. In contrast, the right graph depicts how the stratified_k_fold score changes with variations in n_depth in the raw mode.</p>
</fn>
<fn id="fn2">
<label>2</label>
<p>The number of &#x2018;drugs&#x2019; corresponding to each frequency rather than the number of &#x2018;shortage cases&#x2019; was counted.</p>
</fn>
<fn id="fn3">
<label>3</label>
<p>Among 1,054 cases of drug shortages reported in total, some shortage incidents have two or more causes.</p>
</fn>
<fn id="fn4">
<label>4</label>
<p>Shortage duration range of one product could not be verified, so it was excluded.</p>
</fn>
<fn id="fn5">
<label>5</label>
<p>Drugs that were formerly listed as reimbursable but have been delisted.</p>
</fn>
</fn-group>
<fn-group>
<fn fn-type="custom" custom-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2813297/overview">Vincent Okungu</ext-link>, University of Nairobi, Kenya</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1019134/overview">Ivonne Marisol Torres-Atencio</ext-link>, University of Panama, Panama</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2781847/overview">Mohammed Sallam</ext-link>, Mediclinic Parkview Hospital, United Arab Emirates</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2994728/overview">Dinesh Meena</ext-link>, St. John&#x2019;s Research Institute, India</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Acosta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Vanegas</surname>
<given-names>E. P.</given-names>
</name>
<name>
<surname>Rovira</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Godman</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bochenek</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Medicine shortages: gaps between countries and global perspectives</article-title>. <source>Front. Pharmacol.</source> <volume>10</volume>, <fpage>763</fpage>. <pub-id pub-id-type="doi">10.3389/fphar.2019.00763</pub-id>
<pub-id pub-id-type="pmid">31379565</pub-id>
</mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ayati</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Saiyarsarai</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Nikfar</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Short and long term impacts of COVID-19 on the pharmaceutical sector</article-title>. <source>Daru</source> <volume>28</volume> (<issue>2</issue>), <fpage>799</fpage>&#x2013;<lpage>805</lpage>. <pub-id pub-id-type="doi">10.1007/s40199-020-00358-5</pub-id>
<pub-id pub-id-type="pmid">32617864</pub-id>
</mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Caelen</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A Bayesian interpretation of the confusion matrix</article-title>. <source>Ann. Math. Artif. Intell.</source> <volume>81</volume>, <fpage>429</fpage>&#x2013;<lpage>450</lpage>. <pub-id pub-id-type="doi">10.1007/s10472-017-9564-8</pub-id>
</mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chicco</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Jurman</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>The advantages of the matthews correlation coefficient (MCC) over F1 score and accuracy in binary classification evaluation</article-title>. <source>BMC Genomics</source> <volume>21</volume> (<issue>1</issue>), <fpage>6</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-019-6413-7</pub-id>
<pub-id pub-id-type="pmid">31898477</pub-id>
</mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fox</surname>
<given-names>E. R.</given-names>
</name>
<name>
<surname>Sweet</surname>
<given-names>B. V.</given-names>
</name>
<name>
<surname>Jensen</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Drug shortages: a complex health care crisis</article-title>. <source>Mayo Clin. Proc.</source> <volume>89</volume> (<issue>3</issue>), <fpage>361</fpage>&#x2013;<lpage>373</lpage>. <pub-id pub-id-type="doi">10.1016/j.mayocp.2013.11.014</pub-id>
<pub-id pub-id-type="pmid">24582195</pub-id>
</mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Frazier</surname>
<given-names>P. I.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A tutorial on Bayesian optimization</article-title>. <comment>arXiv preprint, arXiv:1807.02811</comment>
</mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Giammona</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Martino</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Leonardi Vinci</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Provenzani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Polidori</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>5PSQ-096 hazard vulnerability analysis to evaluate the risk of drug shortages according to therapeutic class</article-title>. <source>Value Health</source> <volume>27</volume>, <fpage>A194</fpage>. <pub-id pub-id-type="doi">10.1136/ejhpharm-2020-eahpconf.413</pub-id>
</mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gray</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Manasse</surname>
<given-names>H. R.</given-names>
<suffix>Jr</suffix>
</name>
</person-group> (<year>2012</year>). <article-title>Shortages of medicines: a complex global challenge</article-title>. <source>Bull. World Health Organ</source> <volume>90</volume> (<issue>3</issue>), <fpage>158</fpage>&#x2013;<lpage>158a</lpage>. <pub-id pub-id-type="doi">10.2471/blt.11.101303</pub-id>
<pub-id pub-id-type="pmid">22461706</pub-id>
</mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="book">
<collab>Health Canada</collab> (<year>2024</year>). <source>Health Canada: drug shortages</source>. <publisher-name>Ottawa: Government of Canada</publisher-name>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.canada.ca/en/health-canada/services/drugs-health-products/drug-products/drug-shortages.html">https://www.canada.ca/en/health-canada/services/drugs-health-products/drug-products/drug-shortages.html</ext-link> (Accessed December 27, 2024)</comment>.</mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="web">
<collab>Health Insurance Review and Assessment Service</collab> (<year>2023</year>). <article-title>Study on improving the distribution system for pharmaceuticals: development of a risk detection model for pharmaceutical supply instability</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.hira.or.kr/cms/08/05/02/01/hira_eng.pdf">https://www.hira.or.kr/cms/08/05/02/01/hira_eng.pdf</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ivanov</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Das</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Coronavirus (COVID-19/SARS-CoV-2) and supply chain resilience: a research note</article-title>. <source>Int. J. Integr. Supply Manag.</source> <volume>13</volume>, <fpage>90</fpage>. <pub-id pub-id-type="doi">10.1504/IJISM.2020.107780</pub-id>
</mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Colmenares</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Tak</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Vest</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Clark</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Oertel</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Development and validation of a predictive model to predict and manage drug shortages</article-title>. <source>Am. J. Health Syst. Pharm.</source> <volume>78</volume> (<issue>14</issue>), <fpage>1309</fpage>&#x2013;<lpage>1316</lpage>. <pub-id pub-id-type="doi">10.1093/ajhp/zxab152</pub-id>
<pub-id pub-id-type="pmid">33821926</pub-id>
</mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Machado</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Mendoza</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Corbellini</surname>
<given-names>L. G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>What variables are important in predicting bovine viral diarrhea virus? A random forest approach</article-title>. <source>Vet. Res.</source> <volume>46</volume> (<issue>1</issue>), <fpage>85</fpage>. <pub-id pub-id-type="doi">10.1186/s13567-015-0219-7</pub-id>
<pub-id pub-id-type="pmid">26208851</pub-id>
</mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mazer-Amirshahi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pourmand</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Singer</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pines</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>van den Anker</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Critical drug shortages: implications for emergency medicine</article-title>. <source>Acad. Emerg. Med.</source> <volume>21</volume> (<issue>6</issue>), <fpage>704</fpage>&#x2013;<lpage>711</lpage>. <pub-id pub-id-type="doi">10.1111/acem.12389</pub-id>
<pub-id pub-id-type="pmid">25039558</pub-id>
</mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="book">
<collab>National Institute of Food and Drug Safety Evaluation</collab> (<year>2019</year>). <source>Development of prediction model for drug shortage based on big data</source>. <publisher-name>Seoul: Government of Korea</publisher-name>.</mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pall</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gauthier</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Auer</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mowaswes</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Predicting drug shortages using pharmacy data and machine learning</article-title>. <source>Health Care Manag. Sci.</source> <volume>26</volume>, <fpage>1</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1007/s10729-022-09627-y</pub-id>
<pub-id pub-id-type="pmid">36913071</pub-id>
</mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pauwels</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Simoens</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Casteels</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Huys</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Insights into European drug shortages: a survey of hospital pharmacists</article-title>. <source>PLoS One</source> <volume>10</volume> (<issue>3</issue>), <fpage>e0119322</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0119322</pub-id>
<pub-id pub-id-type="pmid">25775406</pub-id>
</mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Piatek</surname>
<given-names>O. I.</given-names>
</name>
<name>
<surname>Ning</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Touchette</surname>
<given-names>D. R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>National drug shortages worsen during COVID-19 crisis: proposal for a comprehensive model to monitor and address critical drug shortages</article-title>. <source>Am. J. Health Syst. Pharm.</source> <volume>77</volume> (<issue>21</issue>), <fpage>1778</fpage>&#x2013;<lpage>1785</lpage>. <pub-id pub-id-type="doi">10.1093/ajhp/zxaa228</pub-id>
<pub-id pub-id-type="pmid">32716030</pub-id>
</mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sallam</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Oliver</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Allam</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Kassem</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Damani</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Addressing drug shortages at mediclinic parkview hospital: a Five-Year Studysof Challenges, Impact, ani Strategies</article-title>s <source>Cureus</source> <volume>16</volume>, <fpage>e76377</fpage>. <pub-id pub-id-type="doi">10.7759/cureus.76377</pub-id>
<pub-id pub-id-type="pmid">39867009</pub-id>
</mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="book">
<collab>Seoul Pharmaceutical Association</collab> (<year>2023</year>). <source>Call for active measures on the supply instability of medicines to the Korean pharmaceutical association and</source>. <publisher-name>Seoul: Seoul Pharmaceutical Association</publisher-name>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.spa.or.kr/board/PRESS/board.do?SQ=277858#">https://www.spa.or.kr/board/PRESS/board.do?SQ&#x3d;277858&#x23;</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shukar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zahoor</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Hayat</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Saeed</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gillani</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Omer</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Drug shortage: causes, impact, and mitigation strategies</article-title>. <source>Front. Pharmacol.</source> <volume>12</volume>, <fpage>693426</fpage>. <pub-id pub-id-type="doi">10.3389/fphar.2021.693426</pub-id>
<pub-id pub-id-type="pmid">34305603</pub-id>
</mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="journal">
<collab>The Multi-Stakeholder Steering Committee on Drug Shortages (MSSC)</collab> (<year>2017</year>). <article-title>Preventing drug shortages: identifying risks and strategies to address manufacturing-related drug shortages in Canada. Canada: the multi-stakeholder steering committee on drug shortages (MSSC)</article-title>
</mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tucker</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Daskin</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Pharmaceutical supply chain reliability and effects on drug shortages</article-title>
</mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="journal">
<collab>United States Government Accountability Office</collab> (<year>2016</year>). <article-title>Drug shortages: certain factors are strongly associated with this persistent public health challenge</article-title>.</mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="web">
<collab>U.S. Food and Drug Administration (FDA)</collab> (<year>2020</year>). <article-title>
<italic>Report</italic> on drug shortages for calendar year 2019</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.fda.gov/media/139613/download">https://www.fda.gov/media/139613/download</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="book">
<collab>World Health Organization (WHO)</collab> (<year>2015</year>). <source>Technical consultation on preventing and managing shortages of medicines and vaccines</source>. <publisher-loc>Geneva</publisher-loc>: <publisher-name>WHO</publisher-name>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://cdn.who.int/media/docs/default-source/medicines/medicines_shortages.pdf">https://cdn.who.int/media/docs/default-source/medicines/medicines_shortages.pdf</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Current situation, determinants, and solutions to drug shortages in Shaanxi Province, China: a qualitative study</article-title>. <source>PLoS One</source> <volume>11</volume> (<issue>10</issue>), <fpage>e0165183</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0165183</pub-id>
<pub-id pub-id-type="pmid">27780218</pub-id>
</mixed-citation>
</ref>
</ref-list>
</back>
</article>