<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Pharmacol.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Pharmacology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Pharmacol.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">1663-9812</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1514445</article-id>
<article-id pub-id-type="doi">10.3389/fphar.2025.1514445</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Large language models management of complex medication regimens: a case-based evaluation</article-title>
<alt-title alt-title-type="left-running-head">Chase et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphar.2025.1514445">10.3389/fphar.2025.1514445</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Chase</surname>
<given-names>Aaron</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Most</surname>
<given-names>Amoreena</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3225533"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xu</surname>
<given-names>Shaochen</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Barreto</surname>
<given-names>Erin</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Murray</surname>
<given-names>Brian</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2929142"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Henry</surname>
<given-names>Kelli</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Smith</surname>
<given-names>Susan</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hedrick</surname>
<given-names>Tanner</given-names>
</name>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Xianyan</given-names>
</name>
<xref ref-type="aff" rid="aff8">
<sup>8</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Sheng</given-names>
</name>
<xref ref-type="aff" rid="aff9">
<sup>9</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Tianming</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Sikora</surname>
<given-names>Andrea</given-names>
</name>
<xref ref-type="aff" rid="aff10">
<sup>10</sup>
</xref>
<xref ref-type="aff" rid="aff11">
<sup>11</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1394948"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
</contrib-group>
<aff id="aff1">
<label>1</label>
<institution>Department of Pharmacy, Wellstar MCG Health</institution>, <city>Augusta</city>, <state>GA</state>, <country country="US">United States</country>
</aff>
<aff id="aff2">
<label>2</label>
<institution>Department of Pharmacy, UNM Health System</institution>, <city>Albuquerque</city>, <state>NM</state>, <country country="US">United States</country>
</aff>
<aff id="aff3">
<label>3</label>
<institution>Department of Computer Science, University of Georgia</institution>, <city>Athens</city>, <state>GA</state>, <country country="US">United States</country>
</aff>
<aff id="aff4">
<label>4</label>
<institution>Department of Pharmacy, Mayo Clinic</institution>, <city>Rochester</city>, <state>MN</state>, <country country="US">United States</country>
</aff>
<aff id="aff5">
<label>5</label>
<institution>Department of Clinical Pharmacy, University of Colorado Skaggs School of Pharmacy</institution>, <city>Aurora</city>, <state>CO</state>, <country country="US">United States</country>
</aff>
<aff id="aff6">
<label>6</label>
<institution>Department of Clinical and Administrative Pharmacy, University of Georgia College of Pharmacy</institution>, <city>Athens</city>, <state>GA</state>, <country country="US">United States</country>
</aff>
<aff id="aff7">
<label>7</label>
<institution>Department of Pharmacy, University of North Carolina Medical Center</institution>, <city>Chapel Hill</city>, <state>NC</state>, <country country="US">United States</country>
</aff>
<aff id="aff8">
<label>8</label>
<institution>Department of Epidemiology &#x26; Biostatistics, University of Georgia College of Public Health</institution>, <city>Athens</city>, <state>GA</state>, <country country="US">United States</country>
</aff>
<aff id="aff9">
<label>9</label>
<institution>School of Data Science, University of Virginia</institution>, <city>Charlottesville</city>, <state>VA</state>, <country country="US">United States</country>
</aff>
<aff id="aff10">
<label>10</label>
<institution>Department of Biomedical Informatics, University of Colorado School of Medicine</institution>, <city>Aurora</city>, <state>CO</state>, <country country="US">United States</country>
</aff>
<aff id="aff11">
<label>11</label>
<institution>Department of Clinical and Administrative Pharmacy, University of Georgia College of Pharmacy</institution>, <city>Augusta</city>, <state>GA</state>, <country country="US">United States</country>
</aff>
<author-notes>
<corresp id="c001">
<label>&#x2a;</label>Correspondence: Andrea Sikora, <email xlink:href="andrea.sikora@cuanschutz.edu">andrea.sikora@cuanschutz.edu</email>
</corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-11-24">
<day>24</day>
<month>11</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1514445</elocation-id>
<history>
<date date-type="received">
<day>13</day>
<month>03</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>31</day>
<month>10</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>05</day>
<month>11</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Chase, Most, Xu, Barreto, Murray, Henry, Smith, Hedrick, Chen, Li, Liu and Sikora.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Chase, Most, Xu, Barreto, Murray, Henry, Smith, Hedrick, Chen, Li, Liu and Sikora</copyright-holder>
<license>
<ali:license_ref start_date="2025-11-24">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Large language models (LLMs) have shown the ability to diagnose complex medical cases, but only limited studies have evaluated the performance of LLMs in the development of evidence-based treatment plans. The purpose of this evaluation was to test four LLMs on their ability to develop safe and efficacious treatment plans on complex patients managed in the intensive care unit (ICU).</p>
</sec>
<sec>
<title>Methods</title>
<p>Eight high-fidelity patient cases focusing on medication management were developed by critical care clinicians including history of present illness, laboratory values, vital signs, home medications, and current medications. Four LLMs [ChatGPT (GPT-3.5), ChatGPT (GPT-4), Claude-2, and Llama-2&#x2013;70b] were prompted to develop an optimized medication regimen for each case. LLM generated medication regimens were then reviewed by a panel of seven critical care clinicians to assess safety and efficacy, as defined by medication errors identified and appropriate treatment for the clinical conditions. Appropriate treatment was measured by the average rate of clinician agreement to continue each medication in the regimen and compared using analysis of variance (ANOVA).</p>
</sec>
<sec>
<title>Results</title>
<p>Clinicians identified a median of 4.1&#x2013;6.9 medication errors per recommended regimen, and life-threatening medication recommendations were present in 16.3%&#x2013;57.1% of the regimens, depending on LLM. Clinicians continued LLM-recommended medications at a rate of 54.6%&#x2013;67.3%, with GPT-4 having the highest rate of medication continuation among all LLMs tested (p &#x3c; 0.001) and the lowest rate of life-threatening medication errors (p &#x3c; 0.001).</p>
</sec>
<sec>
<title>Conclusion</title>
<p>Caution is warranted using present LLMs for medication regimens given the number of medication errors that were identified in this pilot study. However, LLMs did demonstrate potential to serve as clinical decision support for the management of complex medication regimens given the need for domain specific prompting and testing.</p>
</sec>
</abstract>
<kwd-group>
<kwd>large language model</kwd>
<kwd>artificial intelligence</kwd>
<kwd>pharmacy</kwd>
<kwd>medication regimen complexity</kwd>
<kwd>natural language processing (NLP)</kwd>
</kwd-group>
<funding-group>
<award-group id="gs1">
<funding-source id="sp1">
<institution-wrap>
<institution>Agency for Healthcare Research and Quality</institution>
<institution-id institution-id-type="doi" vocab="open-funder-registry" vocab-identifier="10.13039/open_funder_registry">10.13039/100000133</institution-id>
</institution-wrap>
</funding-source>
</award-group>
<funding-statement>The authors declare that financial support was received for the research and/or publication of this article. Funding through Agency of Healthcare Research and Quality for Drs. Sikora, Smith, Li, and Liu was provided through R21HS028485 and R01HS029009.</funding-statement>
</funding-group>
<counts>
<fig-count count="1"/>
<table-count count="4"/>
<equation-count count="0"/>
<ref-count count="29"/>
<page-count count="9"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Drugs Outcomes Research and Policies</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>Large language models (LLMs) have demonstrated proficiency across a wide spectrum of natural language processing (NLP) tasks, including notable achievements like passing medical licensing exams and making correct diagnoses of complex patient cases (<xref ref-type="bibr" rid="B13">Kanjee et al., 2023</xref>; <xref ref-type="bibr" rid="B9">Gilson et al., 2023</xref>). However, these tasks have largely focused on highly structured problems of disease diagnosis, and LLMs have undergone limited evaluations for the more unstructured task of choosing the correct treatment course for the diagnosed disease (<xref ref-type="bibr" rid="B5">Bu&#x17e;an&#x10d;i&#x107; et al., 2024</xref>; <xref ref-type="bibr" rid="B12">Hsu et al., 2023</xref>; <xref ref-type="bibr" rid="B14">Kunitsu, 2023</xref>).</p>
<p>Comprehensive medication management (CMM) refers to &#x201c;the standard of care that ensures each patient&#x2019;s medications are appropriate, effective for the medical condition, safe given the comorbidities and other medications being taken, and able to be taken as intended.&#x201d; (<xref ref-type="bibr" rid="B1">ASHP, 2025</xref>) Each year, there are approximately 1.8 million adverse drug events (ADEs) in hospitalized patients with estimates that 9,000 patients die as a direct result of a medication error (<xref ref-type="bibr" rid="B16">Leape et al., 1999</xref>; <xref ref-type="bibr" rid="B18">Nuckols et al., 2014</xref>; <xref ref-type="bibr" rid="B20">Slight et al., 2018</xref>). Costs related to medication errors exceed $40 billion (<xref ref-type="bibr" rid="B22">Tariq et al., 2024</xref>). Given the morbidity and cost to the healthcare system associated with ADEs, evaluating novel tools such as LLMs for the potential to facilitate CMM activities and improve medication safety is essential (<xref ref-type="bibr" rid="B15">Kwan et al., 2025</xref>). LLMs process text and understand human language in large quantities and at rapid speeds, which can be helpful in fields such as healthcare and medication management, which include large amount of information processing (<xref ref-type="bibr" rid="B15">Kwan et al., 2025</xref>). Thus far, LLMs have been tested specifically in the realm of medication management for deprescribing benzodiazepines, identifying drug-herb interactions, and performance on a national pharmacist examination (<xref ref-type="bibr" rid="B5">Bu&#x17e;an&#x10d;i&#x107; et al., 2024</xref>; <xref ref-type="bibr" rid="B12">Hsu et al., 2023</xref>; <xref ref-type="bibr" rid="B14">Kunitsu, 2023</xref>). However, there have been no investigations for the potential for LLMs to aid in delivery of CMM.</p>
<p>The purpose of this pilot study was to compare performance of four LLMs [ChatGPT (GPT-3.5), ChatGPT (GPT-4), Claude-2, and Llama-2&#x2013;70b] in conducting CMM for complex medication regimens for critically ill patients.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>Methods</title>
<sec id="s2-3">
<title>Study design</title>
<p>The primary objective was to evaluate the capabilities of LLMs in generating safe and efficacious treatment plans for complex patient cases. This involved a carefully structured prompting process, intended to elicit the most accurate and clinically relevant responses from the LLMs. Our study used a comparative analysis approach, testing four advanced LLMs: GPT-3.5, GPT-4, Llama-2&#x2013;70b, and Claude-2. These LLMs were chosen to parallel other exploratory analyses by our team and were thought to be representative of LLM capability and functionality (<xref ref-type="bibr" rid="B6">Chase et al., 2025</xref>; <xref ref-type="bibr" rid="B26">Yang et al., 2024</xref>). Seven distinct patient cases were used in the fall of 2023, with one that served as an initial example for single-shot prompting, and the subsequent seven cases utilized as actual test scenarios. All test scenarios were entered in separate chats. ChatGPT was accessed via the chatbot interface using the standard settings of temperature &#x3d; 0.7 and Top P &#x3d; 1.0. Llama-2&#x2013;70b was also used with the standard settings. The primary outcomes were based on the safety and efficacy of the recommended scenarios, as assessed by a panel of seven critical care clinicians. Safety was measured by the rate of clinician-identified medication errors and life-threatening medication errors recommended by the LLMs. Efficacy was measured by the average rate of clinician continuation of medications recommended by the LLMs. Other outcomes included the overall agreement of clinicians with the recommended regimen based on a five-point Likert scale and characterization of reasons for discontinuation of medications recommended by LLM.</p>
</sec>
<sec id="s2-4">
<title>LLM testing</title>
<p>A total of eight patient cases were developed by critical care clinicians, with one used as an example in the prompting process. These patient cases included traditional critical care disease states, including sepsis, pneumonia, shock, diabetes, etc. Medication-related problems were intended to reflect critically ill patients cared for in the intensive care unit (ICU), and included evaluations for gastrointestinal ulcer prophylaxis, venous thromboembolism prophylaxis, antibiotic selection, sepsis management, etc. Cases incorporated a history of present illness, relevant laboratory and vital sign data, home medications, and current medications. The patient cases included a &#x201c;ground truth&#x201d; which was a list of appropriate medications determined to be the most correct approach to their management by the panel of clinicians, which was agreed upon via majority vote prior to LLM testing. The ground truth was provided to the LLM in the initial prompting process but then was asked to be generated by the LLM in the new patient scenario process. The approach employed a one-shot prompting with in-context learning designed to guide the LLMs through a structured evaluation of the patient cases to generate an optimized medication regimen (<xref ref-type="bibr" rid="B11">Holmes et al., 2023</xref>). This approach is especially beneficial in complex decision-making tasks, such as medical treatment planning, where contextual understanding and synthesis of information are crucial.</p>
</sec>
<sec id="s2-1">
<title>One-shot prompting with in-context learning process</title>
<p>
<list list-type="order">
<list-item>
<p>Initial Example Prompting: &#x201c;Please review the case below and pay close attention to how the ground truth section at the end is structured.&#x201d; This step involved providing the LLMs with a comprehensive patient case, including detailed medical history, current treatment plans, and the ground truth medication plan. The LLMs were instructed to closely analyze the structure and formatting of the ground truth section, which outlined the updated medication plan. This initial example served as a form of single-shot prompting, aiming to familiarize the LLMs with the expected output format and clinical reasoning required for generating appropriate medication plans.</p>
</list-item>
<list-item>
<p>New Patient Scenario Prompting: &#x201c;Now, I will give you a separate case, please review all the information given and based on it provide a new updated prescribed medication list exactly like how the ground truth section is structured and formatted in the example given before.&#x201d; Following the initial example, the LLMs were presented with new patient scenarios, each featuring unique conditions, clinical scenarios, and medications challenges. The LLMs were tasked with synthesizing this information to propose an updated medication plan, mirroring the structure and format of the ground truth example provided earlier.</p>
</list-item>
</list>
</p>
<p>A panel of seven critical care board-certified and critical care residency trained pharmacists was then asked to review the medication regimen generated by each of the four LLMs for the 7 test patient cases. Individuals were blinded to model identity and to each other. Each individual was asked to review the generated medication regimen and provide the following information: (1) itemized &#x201c;continue&#x201d; or &#x201c;discontinue&#x201d; recommendations for each medication in the recommended regimen with brief rationale, (2) reasons for discontinuation including overt error, therapy optimization, lack of indication, or other, (3) binary evaluation of the presence of at least one life-threatening recommendation made by the LLM, (4) perceived agreement with the overall medication regimen recommended by the LLM on a 1-5 Likert Scale with 1 being strongly disagree and 5 being strongly agree, and (5) any qualitative comments on perception of the medication regimens. The decision to &#x201c;continue&#x201d; or &#x201c;discontinue&#x201d; was based on the ground truth which was approved by a majority vote prior to the testing. The presence of a potential life-threatening medication regimen was at the clinician&#x2019;s discretion. The methods are summarized in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Methodology for LLM-assessment of comprehensive medication management Created with <ext-link ext-link-type="uri" xlink:href="http://biorender.com">biorender.com</ext-link>. O2, oxygen; Glc, serum glucose; gm, Gram; HR, heart rate; ICU, intensive care unit; K, serum potassium; L/min, liters per minute; LLM, large language model; MAP, mean arterial pressure; mcg/kg/min, microgram per kilogram per minute; mg, milligram; MRSA PCR, methicillin-resistant <italic>staphylococcus aureus</italic> nasal polymerase chain reaction; Na, serum sodium; NLF, non-lactose fermenting; PRN, as needed; Q12H, every 12&#xa0;h; Q24H, every 24&#xa0;h; Q6H, every 6&#xa0;h; Q8H, every 8&#xa0;h; RR, respiratory rate; SBP, systolic blood pressure; SCr, serum creatinine; TID, three times daily; Tmax, maximum temperature; unit/hr, unit per hour; WBC, white blood cell.</p>
</caption>
<graphic xlink:href="fphar-16-1514445-g001.tif">
<alt-text content-type="machine-generated">Flowchart showing a case study of an 85-year-old male with respiratory distress and septic shock, requiring oxygen. Home and inpatient medications are detailed. Vital signs and lab results are listed. Ground truth medications are shown with LLM output and clinician feedback. LLM suggestions include cefepime and hydrocortisone, with some drugs suggested for discontinuation due to harm. Clinician results indicate a 57% continuation rate, 28% error rate, and 60% agreement rate.</alt-text>
</graphic>
</fig>
<p>Data Analysis: All statistical analyses were conducted in R version 4.3.1 (2023&#x2013;06&#x2013;16). (<xref ref-type="bibr" rid="B23">Team, 2025</xref>) The rate of continuation of medications was compared between each LLM using analysis of variance (ANOVA) with a Tukey&#x2019;s post-hoc test for pairwise comparisons. Identification of life-threatening errors was compared with Chi-squared test for overall comparison. Chi-squared test with Bonferroni adjustment was used for pairwise comparisons. The median rate of agreement of pharmacists with medication regimen on the Likert Scale was assessed with the Kruskal-Wallis test with a post-hoc Dunn&#x2019;s test with Bonferroni correction for pairwise comparisons. Descriptive analyses were conducted on all variables. Data are reported as mean and standard deviation or median and interquartile range based on parametricity of data.</p>
<p> Data availability: De-identified case prompts are provided in the Appendix. LLM outputs, clinician item-level ratings and analysis code available upon request.</p>
<p>Use of Generative AI: Generative AI was used as a study instrument but was not used for preparation of this manuscript.</p>
<p>Institutional Review Board: The University of Colorado Institutional Review Board determined this study to be exempt (COMIRB 24&#x2013;2328).</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<p>The panel consisted of 7 critical care clinicians with board certification in critical care pharmacotherapy. Demographic characteristics are provided in Supplemental Content&#x2013;<xref ref-type="table" rid="T1">Table 1</xref>. Patient-case prompts are located in the Supplemental Content&#x2013;<xref ref-type="sec" rid="s12">Supplementary Appendix 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Pooled rate of medication continuation per LLM.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">LLM</th>
<th align="left"/>
<th align="center">Case 1</th>
<th align="center">Case 2</th>
<th align="center">Case 3</th>
<th align="center">Case 4</th>
<th align="center">Case 5</th>
<th align="center">Case 6</th>
<th align="center">Case 7</th>
<th align="center">All cases<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</th>
<th align="center">p-value</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="center">GPT-3.5</td>
<td align="center">Continuation rate, median (IQR)</td>
<td align="center">54.6 (43.2&#x2013;54.5)</td>
<td align="center">66.7 (45.8&#x2013;70.8)</td>
<td align="center">66.7 (50&#x2013;77.8)</td>
<td align="center">55.6 (52.8&#x2013;66.7)</td>
<td align="center">57.1 (42.9&#x2013;57.1)</td>
<td align="center">85.7 (64.3&#x2013;89.3)</td>
<td align="center">61.1 (55.9&#x2013;70.6)</td>
<td align="center">59.7 (&#xb1;17.5)</td>
<td rowspan="7" align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="center">Total medications, n</td>
<td align="center">11</td>
<td align="center">12</td>
<td align="center">18</td>
<td align="center">10</td>
<td align="center">7</td>
<td align="center">14</td>
<td align="center">17</td>
<td align="center">89</td>
</tr>
<tr>
<td rowspan="2" align="center">GPT-4</td>
<td align="center">Continuation rate, median (IQR)</td>
<td align="center">64.3 (57.7&#x2013;67.9)</td>
<td align="center">84.2 (78.9&#x2013;86.5)</td>
<td align="center">66.7 (55.6&#x2013;77.8)</td>
<td align="center">57.1 (57.1&#x2013;60.7)</td>
<td align="center">44.4 (27.8&#x2013;55.6)</td>
<td align="center">86.7 (80&#x2013;90)</td>
<td align="center">76.5 (66.7&#x2013;88.2)</td>
<td align="center">67.3 (&#xb1;18.1)<sup>a,b</sup>
</td>
</tr>
<tr>
<td align="center">Total medications, n</td>
<td align="center">14</td>
<td align="center">19</td>
<td align="center">18</td>
<td align="center">14</td>
<td align="center">9</td>
<td align="center">15</td>
<td align="center">17</td>
<td align="center">106</td>
</tr>
<tr>
<td rowspan="2" align="center">Llama-2-70b</td>
<td align="center">Continuation rate, median (IQR)</td>
<td align="center">60 (50&#x2013;70)</td>
<td align="center">59 (50&#x2013;76.2)</td>
<td align="center">72.3 (67.3&#x2013;79)</td>
<td align="center">33.3 (30&#x2013;59)</td>
<td align="center">55.6 (44.4&#x2013;55.6)</td>
<td align="center">63.2 (52.6&#x2013;68.4)</td>
<td align="center">40 (30&#x2013;45)</td>
<td align="center">55 (&#xb1;17.7)<sup>a</sup>
</td>
</tr>
<tr>
<td align="center">Total medications, n</td>
<td align="center">15</td>
<td align="center">21</td>
<td align="center">22</td>
<td align="center">15</td>
<td align="center">9</td>
<td align="center">19</td>
<td align="center">10</td>
<td align="center">111</td>
</tr>
<tr>
<td rowspan="2" align="center">Claude-2</td>
<td align="center">Continuation rate, median (IQR)</td>
<td align="center">54.6 (50&#x2013;59.1)</td>
<td align="center">46.7 (46.7&#x2013;53.3)</td>
<td align="center">60 (55.2&#x2013;70)</td>
<td align="center">55.6 (44.4&#x2013;77.8)</td>
<td align="center">62.5 (37.5&#x2013;62.5)</td>
<td align="center">41.7 (37.5&#x2013;62.5)</td>
<td align="center">62.5 (50&#x2013;62.5)</td>
<td align="center">54.6 (&#xb1;15.4)<sup>b</sup>
</td>
</tr>
<tr>
<td align="center">Total medications, n</td>
<td align="center">11</td>
<td align="center">15</td>
<td align="center">15</td>
<td align="center">9</td>
<td align="center">8</td>
<td align="center">12</td>
<td align="center">8</td>
<td align="center">78</td>
<td align="left"/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>LLM: large language model, IQR: interquartile range.</p>
</fn>
<fn>
<p>Median percentage of medications that were deemed appropriate for continuation by clinician panel after reviewing LLM-generated medication list.</p>
</fn>
<fn>
<p>a, b: rows with matching superscripts are significantly different from each other upon pairwise comparison using Tukey&#x2019;s test for multiple comparisons (ex. GPT-4, is significantly different compared to both Llama-2&#x2013;70b and Claude-2). Adjusted p-values for pairwise comparisons using Tukey&#x2019;s test: GPT-3.5 vs. GPT-4, p &#x3d; 0.131; GPT-3.5 vs. Llama-2&#x2013;70b, p &#x3d; 0.593; GPT-3.5 vs. Claude-2, p &#x3d; 0.446; <sup>a</sup>GPT-4, vs. Llama-2&#x2013;70b, p &#x3d; 0.003; <sup>b</sup>GPT-4, vs. Claude-2, p &#x3d; 0.002; Llama-2&#x2013;70b vs. Claude-2, p &#x3d; 0.999.</p>
</fn>
<fn id="Tfn1">
<label>
<sup>a</sup>
</label>
<p>All cases reports the mean (&#xb1;standard deviation) for all clinician reviews of all cases for that LLM (n &#x3d; 49 [7 cases multiplied by 7 clinician responses]).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>As a measure of efficacy, when clinicians evaluated the LLM-generated medication regimens the median percent of medications continued by each clinician was highest for GPT-4 (67.3% &#xb1; 18.1%) followed by GPT-3.5 (59.7% &#xb1; 17.5%), Llama-2&#x2013;70b (55% &#xb1; 17.7%), and Claude-2 (54.6% &#xb1; 15.4%). Upon post-hoc pairwise analysis, GPT-4 had a significantly higher rate of continuation compared to Llama-2&#x2013;70b (p &#x3d; 0.003) or Claude-2 (p &#x3d; 0.002). These results are summarized in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<p>For overall agreement with the LLM-generated regimen, the Likert scores were significantly different among LLMs (&#x3c7;<sup>2</sup> &#x3d; 15.93, p &#x3d; 0.001). Post-hoc pairwise comparison showed that GPT-4 had a significantly higher rate of agreement compared to Llama-2&#x2013;70b or Claude-2 but other comparisons were not different (see <xref ref-type="table" rid="T2">Table 2</xref>). <xref ref-type="table" rid="T3">Table 3</xref> summarizes rationales for clinician discontinuation of medications in the LLM-generated pharmacotherapy regimen. The median number of medication errors identified by the clinician panel in the pharmacotherapy regimens generated by each LLM were 32, 29, 48, and 34 for GPT-3.5, GPT-4, Llama-2&#x2013;70b, and Claude-2, respectively, with a total of 224, 222, 325, and 246 errors identified in total for each LLM. Therapy optimization was recommended by the clinician panel for 140 medications in the pharmacotherapy regimens generated by GPT-3.5 and GPT-4, 180 medications in Llama-2&#x2013;70b, and 138 medications in the pharmacotherapy regimen generated by Claude-2. And Claude-2, while optimization was recommended for 147 medications in the pharmacotherapy regimen generated by Llama-2&#x2013;70b. Lack of indication was identified by the clinician panel for 58 medication recommendations in GPT-3.5, 68 medication recommendations for GPT-4, 104 medication recommendations for Llama-2&#x2013;70b, and 64 medication recommendations for Claude-2.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Pooled median Likert scores expressing clinician agreement with each LLM-generated medication regimen.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">LLM</th>
<th rowspan="2" align="center">Case 1</th>
<th rowspan="2" align="center">Case 2</th>
<th rowspan="2" align="center">Case 3</th>
<th rowspan="2" align="center">Case 4</th>
<th rowspan="2" align="center">Case 5</th>
<th rowspan="2" align="center">Case 6</th>
<th rowspan="2" align="center">Case 7</th>
<th rowspan="2" align="center">Overall score, median (IQR)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">GPT-3.5</td>
<td align="center">3 (2&#x2013;3)</td>
<td align="center">1 (1&#x2013;1.5)</td>
<td align="center">3 (2.5&#x2013;3.5)</td>
<td align="center">2 (2&#x2013;3)</td>
<td align="center">1 (1&#x2013;2)</td>
<td align="center">3 (2.5&#x2013;4)</td>
<td align="center">1 (1&#x2013;2)</td>
<td align="center">2 (1&#x2013;3)</td>
</tr>
<tr>
<td align="left">GPT-4</td>
<td align="center">3 (2.5&#x2013;3.5)</td>
<td align="center">3 (2&#x2013;3.5)</td>
<td align="center">3 (2.5&#x2013;3.5)</td>
<td align="center">2 (2&#x2013;3)</td>
<td align="center">2 (1&#x2013;2)</td>
<td align="center">3 (2.5&#x2013;4.5)</td>
<td align="center">2 (1&#x2013;3)</td>
<td align="center">3 (2&#x2013;3)<sup>ab</sup>
</td>
</tr>
<tr>
<td align="left">Llama-2&#x2013;70b</td>
<td align="center">2 (2&#x2013;3)</td>
<td align="center">2 (1&#x2013;2.5)</td>
<td align="center">3 (2&#x2013;3)</td>
<td align="center">2 (2&#x2013;3)</td>
<td align="center">1 (1&#x2013;2)</td>
<td align="center">1 (1&#x2013;2.5)</td>
<td align="center">1 (1&#x2013;1)</td>
<td align="center">2 (1&#x2013;3)<xref ref-type="table-fn" rid="Tfn2">
<sup>a</sup>
</xref>
</td>
</tr>
<tr>
<td align="left">Claude-2</td>
<td align="center">2 (2&#x2013;2.5)</td>
<td align="center">1 (1&#x2013;2)</td>
<td align="center">2 (2&#x2013;2.5)</td>
<td align="center">2 (1&#x2013;2)</td>
<td align="center">1 (1&#x2013;2)</td>
<td align="center">1 (1&#x2013;2.5)</td>
<td align="center">1 (1&#x2013;1)</td>
<td align="center">2 (1&#x2013;2)<xref ref-type="table-fn" rid="Tfn3">
<sup>b</sup>
</xref>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>LLM: large language model, IQR: interquartile range.</p>
</fn>
<fn>
<p>a, b: rows with matching superscripts are significantly different from each other upon pairwise comparison using Dunn&#x2019;s test with Bonferroni correction for multiple comparisons (ex. GPT-4, is significantly different compared to both Llama-2&#x2013;70b and Claude-2).</p>
</fn>
<fn>
<p>Adjusted p-values for pairwise comparisons.</p>
</fn>
<fn id="Tfn2">
<label>
<sup>a</sup>
</label>
<p>GPT-4, vs. Llama-2&#x2013;70b, p &#x3d; 0.0014.</p>
</fn>
<fn id="Tfn3">
<label>
<sup>b</sup>
</label>
<p>GPT-4, vs. Claude-2, p &#x3c; 0.001; All other pairwise comparisons, non-significant.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Reason for discontinuation of medications by the clinician panel.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Error type</th>
<th align="left">GPT3.5, median (IQR)</th>
<th align="left">GPT4, median (IQR)</th>
<th align="left">Llama-2&#x2013;70b, median (IQR)</th>
<th align="left">Claude-2, median (IQR)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Overt error</td>
<td align="left">2 (1.5&#x2013;3)</td>
<td align="left">1 (0&#x2013;2)</td>
<td align="left">3 (0.5&#x2013;7)</td>
<td align="left">4 (2&#x2013;6.5)</td>
</tr>
<tr>
<td align="left">Therapy optimization<xref ref-type="table-fn" rid="Tfn4">
<sup>a</sup>
</xref>
</td>
<td align="left">19 (16.5&#x2013;19.5)</td>
<td align="left">19 (17&#x2013;19.5)</td>
<td align="left">22 (21&#x2013;30.5)</td>
<td align="left">21 (16&#x2013;24)</td>
</tr>
<tr>
<td align="left">Lack of indication</td>
<td align="left">10 (5.5&#x2013;11)</td>
<td align="left">7 (6.5&#x2013;13.5)</td>
<td align="left">12 (10&#x2013;20)</td>
<td align="left">9 (6&#x2013;12)</td>
</tr>
<tr>
<td align="left">Other</td>
<td align="left">1 (1&#x2013;2)</td>
<td align="left">0 (0&#x2013;0.5)</td>
<td align="left">1 (0&#x2013;1.5)</td>
<td align="left">1 (0&#x2013;3)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>For each model the reported median represents the median number of errors reported per clinician across all cases.</p>
</fn>
<fn>
<p>LLM, large language model; IQR, interquartile range.</p>
</fn>
<fn id="Tfn4">
<label>
<sup>a</sup>
</label>
<p>Therapy optimization would include anything that was deemed not optimal by the clinician panel but not necessarily harmful to the patient (ex. If the LLM, selected a twice daily blood pressure medication as opposed to a simpler once daily regimen, or if it selected an antibiotic that more commonly causes side effects as opposed to a better-tolerated regimen).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>As an assessment of safety, the presence of potentially life-threatening recommendations was assessed by clinicians in 57.1% in Claude-2 recommendations followed by 38.8% GPT-3.5 recommendations, 28.6% of Llama-2&#x2013;70b recommendations, and 16.3% of GPT-4 recommendations. Upon pairwise analysis, GPT-4 had significantly fewer potentially life-threatening errors than GPT-3.5 (p &#x3d; 0.013) or Claude-2 (p &#x3c; 0.001) and Llama-2&#x2013;70b had significantly fewer potentially life-threatening errors than Claude-2 (p &#x3d; 0.0043) (see <xref ref-type="table" rid="T4">Table 4</xref>). All other comparisons were non-significantly different. Life-threatening errors per case and a description of those errors are reported in the Supplementary Content- <xref ref-type="table" rid="T2">Tables 2</xref>, <xref ref-type="table" rid="T3">3</xref>.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Medication errors.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">LLM</th>
<th align="center">Total errors (across all cases), median (IQR)</th>
<th align="center">Cases with at least 1 clinician reporting a life-threatening error, n (%)<break/>N &#x3d; 7</th>
<th align="center">Rate of life-threatening errors<xref ref-type="table-fn" rid="Tfn5">
<sup>a</sup>
</xref>, n (%)<break/>N &#x3d; 49</th>
<th align="center">Chi-square p-value</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">GPT-3.5</td>
<td align="center">32 (27&#x2013;34)</td>
<td align="center">7 (100)</td>
<td align="center">19 (38.8)<sup>a</sup>
</td>
<td rowspan="4" align="center">&#x3c;0.001</td>
</tr>
<tr>
<td align="left">GPT-4</td>
<td align="center">29 (28&#x2013;34.5)</td>
<td align="center">3 (43.9)</td>
<td align="center">8 (16.3)<xref ref-type="table-fn" rid="Tfn6">
<sup>b</sup>
</xref>
<sup>,</sup>
<xref ref-type="table-fn" rid="Tfn7">
<sup>c</sup>
</xref>
</td>
</tr>
<tr>
<td align="left">Llama-2&#x2013;70b</td>
<td align="center">48 (44&#x2013;49)</td>
<td align="center">6 (85.7)</td>
<td align="center">14 (28.6)<xref ref-type="table-fn" rid="Tfn8">
<sup>d</sup>
</xref>
</td>
</tr>
<tr>
<td align="left">Claude-2</td>
<td align="center">34 (33&#x2013;35)</td>
<td align="center">7 (100)</td>
<td align="center">28 (57.1)<xref ref-type="table-fn" rid="Tfn7">
<sup>c</sup>
</xref>
<sup>,</sup>
<xref ref-type="table-fn" rid="Tfn8">
<sup>d</sup>
</xref>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>a, b, c: rows with matching superscripts are significantly different from each other upon pairwise comparison using Chi-squared test with Bonferroni correction for multiple comparisons (ex. GPT-4, is significantly different compared to both GPT-3.5 and Claude-2).</p>
</fn>
<fn id="Tfn5">
<label>
<sup>a</sup>
</label>
<p>percentage is calculated using cases that were assessed as having a potential life threatening error divided by total cases (n &#x3d; 49).</p>
</fn>
<fn>
<p>Adjusted p-values for pairwise comparisons.</p>
</fn>
<fn id="Tfn6">
<label>
<sup>b</sup>
</label>
<p>GPT-3.5 vs. GPT-4, p &#x3d; 0.013, GPT-3.5 vs. Llama-2&#x2013;70b, p &#x3d; 0.29, GPT-3.5 vs. Claude-2, p &#x3d; 0.069, GPT-4, vs. Llama-2&#x2013;70b, p &#x3d; 0.15.</p>
</fn>
<fn id="Tfn7">
<label>
<sup>c</sup>
</label>
<p>GPT-4, vs. Claude-2, p &#x3c; 0.001.</p>
</fn>
<fn id="Tfn8">
<label>
<sup>d</sup>
</label>
<p>Llama-2&#x2013;70b vs. Claude-2, p &#x3d; 0.0043.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>In an early evaluation of the ability of LLMs to provide CMM for complex, critically ill patients, a high rate of life-threatening medication recommendations were provided. Of the four LLMs tested GPT-4 had the best performance, demonstrating the highest rates of clinician agreement and lowest rates of life-threatening medical errors. Although the outputs demonstrated contextual grasp of domain-specific content (e.g., correctly matching drugs with doses and routes and matching certain therapies with diseases), LLMs did not consistently evaluate patient specific cases. This study patently supports a stepwise prompting and implementation approach for LLMs in the CMM space.</p>
<p>Using LLMs for medication management has untapped potential given the prolific use of prescription medications and risk for ADEs (<xref ref-type="bibr" rid="B19">Sikora, 2023</xref>). However, there are significant challenges that must be overcome. Most LLMs are trained on a widely available corpus (e.g., the Internet), which creates the potential for problems in domains marked by highly technical language or rarely occurring scenarios, as is a hallmark of medical and pharmacy domains (<xref ref-type="bibr" rid="B7">Clusmann et al., 2023</xref>; <xref ref-type="bibr" rid="B21">Soroush et al., 2024</xref>). Medication use is fraught with errors, so identifying &#x2018;ground truth&#x2019; remains a perennial challenge. Additionally, high-quality CMM requires a combination of both recall-based knowledge and application-oriented skills to understand how the individual drug, dose, and formulation interact with the patient, disease, and other medications in a given context to ascertain risk and benefit profiles (<xref ref-type="bibr" rid="B3">Bainum et al., 2024</xref>; <xref ref-type="bibr" rid="B4">Branan et al., 2024</xref>). Practice-based expertise that encompasses a wide array of relatively rare scenarios is also hard to replicate in datasets. Owing to the challenges as well as potential dangers associated with poor performance, there have been calls for thoughtful evaluation of LLMs prior to use in the healthcare setting (<xref ref-type="bibr" rid="B2">Ayers et al., 2024</xref>).</p>
<p>As a key finding of this study, in holistic evaluation clinicians ranked the highest performing LLM as a median 2 out of 5 on level of agreement (i.e., disagree). It is worth noting that given the complexity of the cases and the nuance of clinical practice, there can be differences between a reasonable choice and the best choice. Similarly, &#x201c;medication error&#x201d; is a broad term, inclusive of minor oversights with little potential to cause patient harm as well as critical mistakes that can result in significant adverse outcomes). However, our study categorized the reasons why clinical experts discontinued medications recommended by the LLMs and found a high rate of life-threatening pharmacotherapy recommendations, pointing to a concerning knowledge gap for LLMs. For example, in one case with a patient experiencing elevated intracranial pressure, one LLM recommended administering a 250&#xa0;mL bolus of 23.4% hypertonic saline, a medication that is typically administered as a 30&#xa0;mL bolus when treating neurologic emergencies: if this had occurred in practice, it would likely have led to significant morbidity and mortality for the patient and notable quality improvement and root cause analysis processes.</p>
<p>There was also a lack of consistency in LLM recommendations across cases with similar features. For example, GPT-3.5 recommended vancomycin in two cases, but different dosing strategies. In one case, it simply recommended vancomycin 1,250&#xa0;mg x1 with no mention of target trough concentrations, but in another case it recommended 1,250&#xa0;mg every 12&#xa0;h with a target trough of 15&#x2013;20&#xa0;mg/L. Similarly, GPT-4 had inconsistent recommendations with regards to stress-dose steroids in septic shock. In one case it recommended the addition of steroids for a patient on norepinephrine alone, but in a second case it did not add steroids for a patient on norepinephrine plus vasopressin. This inconsistency in recommendations raises concerns about the background logic being applied by LLMs.</p>
<p>Another observed pattern was a predisposition to continuing medications in the &#x201c;current medications&#x201d; content of the case presented to the LLM. This could include continuing a medication without a clear indication for prior-to-admission use (e.g., baclofen in a patient without spasticity) or continuing medications exactly as written in the &#x201c;current medications&#x201d; (e.g., &#x201c;norepinephrine 0.09 mcg/kg/min&#x201d; rather than norepinephrine titrated to a MAP goal). These patterns give the sense that LLMs are simply transcribing data rather than evaluating the medications on their merits. Other observations included that the LLMs struggled to provide appropriate renal dose adjustments based on patient conditions and committed frequent opioid-related errors (e.g., administering an oral medication intravenously or intravenous opioids to non-intubated patients).</p>
<p>There were some positive observations with regard to data synthesis, particularly with GPT-3.5 and GPT-4. In case 5, GPT-3.5 picked up on &#x201c;sepsis&#x201d; in the case and recommended crystalloid 30&#xa0;mL/kg for the patient in line with best practice guidelines for sepsis management (<xref ref-type="bibr" rid="B8">Evans et al., 2021</xref>). Unfortunately, the patient had already received resuscitation, so repeating 30&#xa0;mL/kg would likely not be indicated. Nonetheless, this observation suggests a stronger ability to collect information from the history of present illness compared to Llama-2&#x2013;70b or Claude-2. Similarly, in case 4, GPT-4 picked up on &#x201c;reduced oral intake&#x201d; in the history of present illness and recommended a fluid bolus &#x201c;to address dehydration from reduced oral intake&#x201d;. This represents an impressive ability to collect and synthesize data before making recommendations.</p>
<p>Our methodology was structured to maximize LLM understanding and application of clinical knowledge in the formulation of medication plans (<xref ref-type="bibr" rid="B28">Zhao et al., 2023</xref>). By employing reasoning engines (i.e., chain of thought) and one-shot prompting via emphasizing the importance of the in-context demonstration for formatting, we aimed to enhance the models&#x2019; ability to process and apply complex medical information. This was further supported by the comparative analysis of the responses across different LLMs, providing insights into their respective capabilities and limitations in medical decision-making tasks. Throughout the study, the effectiveness of the one-shot prompting with in-context learning and the chain-of-thought method was assessed based on the accuracy and clinical relevance of the medication plans generated by the LLMs. The structured approach and comparative analysis offer valuable contributions to the ongoing exploration of the potential of LLMs in healthcare applications, particularly in the context of medication management and treatment planning. The refinement of chain-of-thought (or related concepts like tree-of-thought and graph-of-thought) in combination with zero or few shot learning are rapidly implementable methods even as new medication knowledge and LLM technology progress, which are helpful for keeping such technology up to date. Indeed, this strategy is particularly helpful in healthcare where labeled data (i.e., a dataset with annotated &#x2018;correct&#x2019; answers) are scarce and because the prompts support in-context learning, which can strengthen and accelerate the exhaustive fine-tuning process (<xref ref-type="bibr" rid="B17">Ma et al., 2024</xref>; <xref ref-type="bibr" rid="B10">Guan et al., 2023</xref>). Reasoning engines break up problems into steps from which logical inferences can be made. Our team has shown that zero- and few-shot learning can contribute to dealing with unseen scenarios that lack training datasets, including a new abductive reasoning method via natural language processing (<xref ref-type="bibr" rid="B29">Zhong et al., 2025</xref>).</p>
<p>Reasoning engines are useful because they reduce hallucinations and support assessment for logical or training gaps (<xref ref-type="bibr" rid="B11">Holmes et al., 2023</xref>; <xref ref-type="bibr" rid="B25">Wei et al., 2022</xref>). This structured approach to reasoning can be particularly beneficial in capturing the nuances of clinical decision-making. This study used a one-shot prompting approach in which each model was shown an example case that included a complete &#x201c;ground truth&#x201d; medication plan before generating new responses. The exemplar was designed to illustrate how outputs should be structured and reasoned through, not to provide clinical content for reuse. Nevertheless, this setup introduces a potential for in-context leakage: the models could have inferred therapeutic logic or stylistic patterns from the exemplar rather than developing their own reasoning independently. Although the exemplar and test cases involved different patients and clinical details, some overlap in themes (such as sepsis or shock management) may have subtly influenced model outputs. Recognizing this trade-off is important. The exemplar likely improved consistency and formatting across models but may have partially guided their clinical reasoning. Future research could reduce this risk by randomizing or rotating exemplars, using multiple independent examples, or adopting a zero-shot design to isolate genuine model reasoning and generalizability.</p>
<p>This evaluation assesses the ability of LLMs to manage complex medication regimens, with strengths including the establishment of a clinically valid ground truth and inclusion of a diverse clinician panel. However, some limitations exist including that the LLM was not provided all information generally available in the electronic health record and the LLMs were tested on a small number of cases which had similarities throughout and lacked repeating trials to evaluate consistency of model performance. Future analyses would benefit from repeated prompting as well as sensitivity analyses with different model settings. Additionally, the LLMs used were not specifically designed for healthcare-specific assessments, so they likely lacked prior training in these areas. Our analysis was intended to sample LLM capability with different complex cases in critical care, but we recognize that differences in cases (sepsis vs. stroke) may account for some of the variability. However, this proof-of-concept analysis was not designed to explore that component. Additionally, at the time of testing, the LLMs selected were the most up-to-date LLMs available on the market. We recognize that newer models have since been release; however, the latest work suggests that while these models have improved computing capacity, human alignment and domain specific testing remain important.</p>
<p>Ground truth is difficult to establish, as it does still require some aspect of clinical acumen: it is important that future evaluations consider how to account for stylistic variation that is within the confines of evidence-based medicine and not truly reflective of LLM performance. Clinicians may have different opinions on error assessment and adverse event likelihood that may have led to heterogeneity in the &#x201c;ground truth&#x201d; determination: this is particularly true in critical care, which observes practice variation given clinical uncertainty in the treatment of various disease states. While our panel attempted to reference guidelines wherever possible, this is a limitation of the study due to practice variation. In clinical scenarios where the guidelines may not be fully applicable to the patient or where there may be several appropriate courses of action, the &#x201c;ground truth&#x201d; may be difficult to determine. Though out of scope for this exploratory analysis, establishing how LLMs should act in the setting of clinical uncertainty (i.e., when the ground truth is unknown) is an essential step for their clinical use. In this case, our panel expected to the LLMs to make recommendations that do not overtly cause harm (e.g., high doses of potassium in the setting of renal failure leading to life-threatening arrythmias), to make use of available guidelines whenever available, and to treat the conditions stated in the cases (e.g., antibiotics for sepsis). There is more recent work with LLMs teaching them to say &#x201c;I do not know,&#x201d; which may also be a future expectation (<xref ref-type="bibr" rid="B27">Zhang et al., 2024</xref>). Notably, the criteria used for evaluating these LLM-generated treatment plans is not standardized and involved human review (instead of automation). Objective, standardized, and ideally automated means of establishing clinical acceptance criteria and performance benchmarking for clinical NLP is an essential area for future development. Indeed, the FDA&#x2019;s recent viewpoint in JAMA specifically stated that industry and other stakeholders must improve quality assurance and evaluation of artificial intelligence so that there can be consistency and rigor in the critique of artificial intelligence studies (<xref ref-type="bibr" rid="B24">Warraich et al., 2025</xref>).</p>
<p>Despite the limitations of this proof-of-concept analysis, findings suggest that available training and fine-tuning methods may support the use of LLMs for treatment selection. The pipeline necessary to develop LLMs to assist with CMM will likely include a thoughtful integration of domain-specific demonstrations including prompt engineering and real-life human feedback and direct preference optimization combined with infrastructure that allows for continual updates as medication knowledge expands. Though these undertakings are time- and resource-intensive, the potential shown here supports future investigations.</p>
</sec>
<sec sec-type="conclusion" id="s5">
<title>Conclusion</title>
<p>Using present LLMs as a clinical support tool warrants caution, as without thoughtful human interaction, generated recommendations could cause overt harm. However, there is potential for specifically engineered LLMs tailored for medication management given a thoughtful training and fine-tuning paradigm and appropriate clinical benchmarking. Further development is necessary before LLMs can be reliably used as a clinical support tool given their underperformance in this analysis.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>AC: Writing &#x2013; original draft, Writing &#x2013; review and editing, Formal Analysis. AM: Investigation, Writing &#x2013; original draft, Writing &#x2013; review and editing. SX: Investigation, Writing &#x2013; review and editing. EB: Methodology, Writing &#x2013; original draft. BM: Writing &#x2013; review and editing. KH: Methodology, Writing &#x2013; review and editing. SS: Methodology, Writing &#x2013; review and editing. TH: Investigation, Methodology, Writing &#x2013; original draft. XC: Writing &#x2013; review and editing. SL: Conceptualization, Writing &#x2013; review and editing. TL: Conceptualization, Methodology, Writing &#x2013; review and editing. AS: Methodology, Conceptualization, Writing &#x2013; review and editing.</p>
</sec>
<ack>
<title>Acknowledgements</title>
<p>Liana Ha, Garrett Brown, Timothy W. Jones.</p>
</ack>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The authors declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fphar.2025.1514445/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fphar.2025.1514445/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Supplementaryfile1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn fn-type="custom" custom-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/42464/overview">Bernd Rosenkranz</ext-link>, Fundisa African Academy of Medicines Development, South Africa</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1814320/overview">X C</ext-link>, Peking Union Medical College Hospital, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3030113/overview">Kaitlin Alexander</ext-link>, University of Florida, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3124950/overview">Philip Chung</ext-link>, Stanford University, United States</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="web">
<collab>ASHP</collab> <article-title>Comprehensive medication management</article-title> (<year>2025</year>). <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.ashp.org/advocacy-and-issues/key-issues/other-issues/comprehensive-medication-management?loginreturnUrl=SSOCheckOnly#:%7E:text=Definition%20of%20CMM%3A%20The%20standard,effective%20for%20the%20medical%20condition">https://www.ashp.org/advocacy-and-issues/key-issues/other-issues/comprehensive-medication-management?loginreturnUrl&#x3d;SSOCheckOnly&#x23;:&#x223c;:text&#x3d;Definition%20of%20CMM%3A%20The%20standard,effective%20for%20the%20medical%20condition</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ayers</surname>
<given-names>J. W.</given-names>
</name>
<name>
<surname>Desai</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>D. M.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Regulate artificial intelligence in health care by prioritizing patient outcomes</article-title>. <source>Jama</source> <volume>331</volume> (<issue>8</issue>), <fpage>639</fpage>&#x2013;<lpage>640</lpage>. <pub-id pub-id-type="doi">10.1001/jama.2024.0549</pub-id>
<pub-id pub-id-type="pmid">38285467</pub-id>
</mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bainum</surname>
<given-names>T. B.</given-names>
</name>
<name>
<surname>Krueger</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hawkins</surname>
<given-names>W. A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Cultivating expert thinking skills for experiential pharmacy trainees</article-title>. <source>Am. J. Health Syst. Pharm.</source> <volume>82</volume> (<issue>9</issue>), <fpage>e472</fpage>&#x2013;<lpage>e478</lpage>. <pub-id pub-id-type="doi">10.1093/ajhp/zxae366</pub-id>
</mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Branan</surname>
<given-names>T. N.</given-names>
</name>
<name>
<surname>Darley</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hawkins</surname>
<given-names>W. A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>How critical is it? Integrating critical care into the pharmacy didactic curriculum</article-title>. <source>Am. J. Health Syst. Pharm.</source> <volume>9</volume> (<issue>81</issue>), <fpage>871</fpage>&#x2013;<lpage>875</lpage>. <pub-id pub-id-type="doi">10.1093/ajhp/zxae153</pub-id>
</mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bu&#x17e;an&#x10d;i&#x107;</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Belec</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Dr&#x17e;ai&#x107;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kummer</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Brki&#x107;</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fialov&#xe1;</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Clinical decision-making in benzodiazepine deprescribing by healthcare providers vs. AI-assisted approach</article-title>. <source>Br. J. Clin. Pharmacol.</source> <volume>90</volume> (<issue>3</issue>), <fpage>662</fpage>&#x2013;<lpage>674</lpage>. <pub-id pub-id-type="doi">10.1111/bcp.15963</pub-id>
<pub-id pub-id-type="pmid">37949663</pub-id>
</mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chase</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Most</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sikora</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Devlin</surname>
<given-names>J. W.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Evaluation of large language models&#x27; ability to identify clinically relevant drug-drug interactions and generate high-quality clinical pharmacotherapy recommendations</article-title>. <source>Am. J. Health Syst. Pharm.</source> <volume>2025</volume>, <fpage>zxaf168</fpage>. <pub-id pub-id-type="doi">10.1093/ajhp/zxaf168</pub-id>
<pub-id pub-id-type="pmid">40590636</pub-id>
</mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clusmann</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kolbinger</surname>
<given-names>F. R.</given-names>
</name>
<name>
<surname>Muti</surname>
<given-names>H. S.</given-names>
</name>
<name>
<surname>Carrero</surname>
<given-names>Z. I.</given-names>
</name>
<name>
<surname>Eckardt</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Laleh</surname>
<given-names>N. G.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>The future landscape of large language models in medicine</article-title>. <source>Commun. Med. (Lond)</source> <volume>3</volume> (<issue>1</issue>), <fpage>141</fpage>. <pub-id pub-id-type="doi">10.1038/s43856-023-00370-1</pub-id>
<pub-id pub-id-type="pmid">37816837</pub-id>
</mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Evans</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Rhodes</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Alhazzani</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Antonelli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Coopersmith</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>French</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Surviving sepsis campaign: international guidelines for management of sepsis and septic shock 2021</article-title>. <source>Crit. Care Med.</source> <volume>49</volume> (<issue>11</issue>), <fpage>e1063</fpage>&#x2013;<lpage>e1143</lpage>. <pub-id pub-id-type="doi">10.1097/ccm.0000000000005337</pub-id>
<pub-id pub-id-type="pmid">34605781</pub-id>
</mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gilson</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Safranek</surname>
<given-names>C. W.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Socrates</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Chi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Taylor</surname>
<given-names>R. A.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>How does ChatGPT perform on the United States medical licensing examination (USMLE)? The implications of large language models for medical education and knowledge assessment</article-title>. <source>JMIR Med. Educ.</source> <volume>9</volume>, <fpage>e45312</fpage>. <pub-id pub-id-type="doi">10.2196/45312</pub-id>
<pub-id pub-id-type="pmid">36753318</pub-id>
</mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Guan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). &#x201c;<article-title>CohortGPT: an enhanced GPT for participant recruitment in clinical study</article-title>&#x201d;. <publisher-name>arXiv preprint</publisher-name>. <pub-id pub-id-type="doi">10.48550/arXiv:2307.11346</pub-id>
</mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Holmes</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sio</surname>
<given-names>T. T.</given-names>
</name>
<name>
<surname>McGee</surname>
<given-names>L. A.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Evaluating large language models on a highly-specialized topic, radiation oncology physics</article-title>. <source>Front. Oncol</source>. <volume>23</volume>, <lpage>1219326</lpage>. <pub-id pub-id-type="doi">10.3389/fonc.2023.1219326</pub-id>
<pub-id pub-id-type="pmid">37529688</pub-id>
</mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hsu</surname>
<given-names>H. Y.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>K. C.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>S. Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Hsieh</surname>
<given-names>Y. W.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>Y. D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Examining real-world medication consultations and drug-herb interactions: ChatGPT performance evaluation</article-title>. <source>JMIR Med. Educ.</source> <volume>9</volume>, <fpage>e48433</fpage>. <pub-id pub-id-type="doi">10.2196/48433</pub-id>
<pub-id pub-id-type="pmid">37561097</pub-id>
</mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kanjee</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Crowe</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Rodman</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Accuracy of a generative artificial intelligence model in a complex diagnostic challenge</article-title>. <source>JAMA</source> <volume>330</volume> (<issue>1</issue>), <fpage>78</fpage>&#x2013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1001/jama.2023.8288</pub-id>
<pub-id pub-id-type="pmid">37318797</pub-id>
</mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kunitsu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>The potential of GPT-4 as a support tool for pharmacists: analytical study using the Japanese national examination for pharmacists</article-title>. <source>JMIR Med. Educ.</source> <volume>9</volume>, <fpage>e48452</fpage>. <pub-id pub-id-type="doi">10.2196/48452</pub-id>
<pub-id pub-id-type="pmid">37837968</pub-id>
</mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kwan</surname>
<given-names>H. Y.</given-names>
</name>
<name>
<surname>Shell</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fahy</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Integrating large language models into medication management in remote healthcare: current applications, challenges, and future prospects</article-title>. <source>Systems</source> <volume>13</volume> (<issue>4</issue>), <fpage>281</fpage>. <pub-id pub-id-type="doi">10.3390/systems13040281</pub-id>
</mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Leape</surname>
<given-names>L. L.</given-names>
</name>
<name>
<surname>Cullen</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Clapp</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Burdick</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Demonaco</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Erickson</surname>
<given-names>J. I.</given-names>
</name>
<etal/>
</person-group> (<year>1999</year>). <article-title>Pharmacist participation on physician rounds and adverse drug events in the intensive care unit</article-title>. <source>JAMA</source> <volume>282</volume> (<issue>3</issue>), <fpage>267</fpage>&#x2013;<lpage>270</lpage>. <pub-id pub-id-type="doi">10.1001/jama.282.3.267</pub-id>
<pub-id pub-id-type="pmid">10422996</pub-id>
</mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>An interative optimizing framwork for radiology report summarization with ChatGPT</article-title>. <source>IEEE Trans. Artif. Intell.</source> <volume>5</volume> (<issue>8</issue>), <fpage>4163</fpage>&#x2013;<lpage>4175</lpage>. <pub-id pub-id-type="doi">10.1109/TAI.2024.3364586</pub-id>
</mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nuckols</surname>
<given-names>T. K.</given-names>
</name>
<name>
<surname>Smith-Spangler</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Morton</surname>
<given-names>S. C.</given-names>
</name>
<name>
<surname>Asch</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Patel</surname>
<given-names>V. M.</given-names>
</name>
<name>
<surname>Anderson</surname>
<given-names>L. J.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>The effectiveness of computerized order entry at reducing preventable adverse drug events and medication errors in hospital settings: a systematic review and meta-analysis</article-title>. <source>Syst. Rev.</source> <volume>3</volume>, <fpage>56</fpage>. <pub-id pub-id-type="doi">10.1186/2046-4053-3-56</pub-id>
<pub-id pub-id-type="pmid">24894078</pub-id>
</mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sikora</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Critical care pharmacists: a focus on Horizons</article-title>. <source>Crit. Care Clin.</source> <volume>39</volume> (<issue>3</issue>), <fpage>503</fpage>&#x2013;<lpage>527</lpage>. <pub-id pub-id-type="doi">10.1016/j.ccc.2023.01.006</pub-id>
<pub-id pub-id-type="pmid">37230553</pub-id>
</mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Slight</surname>
<given-names>S. P.</given-names>
</name>
<name>
<surname>Seger</surname>
<given-names>D. L.</given-names>
</name>
<name>
<surname>Franz</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bates</surname>
<given-names>D. W.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>The national cost of adverse drug events resulting from inappropriate medication-related alert overrides in the United States</article-title>. <source>J. Am. Med. Inf. Assoc.</source> <volume>25</volume> (<issue>9</issue>), <fpage>1183</fpage>&#x2013;<lpage>1188</lpage>. <pub-id pub-id-type="doi">10.1093/jamia/ocy066</pub-id>
<pub-id pub-id-type="pmid">29939271</pub-id>
</mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Soroush</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Glicksberg</surname>
<given-names>B. S.</given-names>
</name>
<name>
<surname>Zimlichman</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Barash</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Freeman</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Charney</surname>
<given-names>A. W.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Large language models are poor medical coders &#x2014; benchmarking of medical code querying</article-title>. <source>NEJM AI</source> <volume>1</volume> (<issue>5</issue>), <fpage>AIdbp2300040</fpage>. <pub-id pub-id-type="doi">10.1056/AIdbp2300040</pub-id>
</mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tariq</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Vashisht</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sinha</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Scherbak</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Medication dispensing errors and prevention</article-title>. <source>StatPearls</source>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/books/NBK519065/">https://www.ncbi.nlm.nih.gov/books/NBK519065/</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Team</surname>
<given-names>R. C. R.</given-names>
</name>
</person-group> (<year>2025</year>). <source>A language and environment for statistical &#x23;&#x23; computing</source>. <publisher-name>R Foundation for Statistical Computing</publisher-name>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.R-project.org/">https://www.R-project.org/</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Warraich</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Tazbaz</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Califf</surname>
<given-names>R. M.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>FDA perspective on the regulation of artificial intelligence in health care and biomedicine</article-title>. <source>JAMA</source> <volume>333</volume> (<issue>3</issue>), <fpage>241</fpage>&#x2013;<lpage>247</lpage>. <pub-id pub-id-type="doi">10.1001/jama.2024.21451</pub-id>
<pub-id pub-id-type="pmid">39405330</pub-id>
</mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tay</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bommasani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Raffel</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zoph</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Borgeaud</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Emergent abilities of large language models</article-title>. <comment>arXiv preprint</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.2206.07682</pub-id>
</mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Most</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hawkins</surname>
<given-names>W. A.</given-names>
</name>
<name>
<surname>Murray</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>S. E.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Evaluating accuracy and reproducibility of large language model performance on critical care assessments in pharmacy education</article-title>. <source>Front. Artif. Intell.</source> <volume>7</volume>, <fpage>1514896</fpage>. <pub-id pub-id-type="doi">10.3389/frai.2024.1514896</pub-id>
<pub-id pub-id-type="pmid">39850846</pub-id>
</mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Diao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fung</surname>
<given-names>Y. R.</given-names>
</name>
<name>
<surname>Lian</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <source>R-Tuning: instructing large language models to say &#x2018;I don&#x2019;t know&#x2019;</source>. <publisher-loc>Mexico City, Mexico</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>, <fpage>7113</fpage>&#x2013;<lpage>7139</lpage>.</mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>When brain-inspired AI meets AGI</article-title>. <source>Meta-Radiology</source> <volume>1</volume> (<issue>1</issue>), <fpage>100005</fpage>. <pub-id pub-id-type="doi">10.1016/j.metrad.2023.100005</pub-id>
</mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhong</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>ChatABL: Abductive learning via natural language interaction with ChatGPT</article-title>. <source>IEEE Trans. Neural Netw. Learn Syst.</source> <volume>36</volume>, <fpage>17635</fpage>&#x2013;<lpage>17649</lpage>. <pub-id pub-id-type="doi">10.1109/tnnls.2025.3567945</pub-id>
<pub-id pub-id-type="pmid">40526555</pub-id>
</mixed-citation>
</ref>
</ref-list>
</back>
</article>