<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Digit. Health</journal-id>
<journal-title>Frontiers in Digital Health</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Digit. Health</abbrev-journal-title>
<issn pub-type="epub">2673-253X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fdgth.2025.1624786</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Digital Health</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Diagnostic efficacy of large language models in the pediatric emergency department: a pilot study</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes"><name><surname>Del Monte</surname><given-names>Francesco</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="an1"><sup>&#x2020;</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/1269137/overview"/><role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/><role content-type="https://credit.niso.org/contributor-roles/data-curation/"/><role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/><role content-type="https://credit.niso.org/contributor-roles/methodology/"/></contrib>
<contrib contrib-type="author" equal-contrib="yes"><name><surname>Barolo</surname><given-names>Roberta</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="an1"><sup>&#x2020;</sup></xref><role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/><role content-type="https://credit.niso.org/contributor-roles/investigation/"/><role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/></contrib>
<contrib contrib-type="author"><name><surname>Circhetta</surname><given-names>Maria</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref><role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/><role content-type="https://credit.niso.org/contributor-roles/data-curation/"/><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/></contrib>
<contrib contrib-type="author"><name><surname>Delmonaco</surname><given-names>Angelo Giovanni</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref><role content-type="https://credit.niso.org/contributor-roles/validation/"/><role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/><role content-type="https://credit.niso.org/contributor-roles/resources/"/><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/></contrib>
<contrib contrib-type="author" corresp="yes"><name><surname>Castagno</surname><given-names>Emanuele</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="cor1">&#x002A;</xref><uri xlink:href="https://loop.frontiersin.org/people/1402934/overview" /><role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/><role content-type="https://credit.niso.org/contributor-roles/methodology/"/><role content-type="https://credit.niso.org/contributor-roles/visualization/"/><role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/></contrib>
<contrib contrib-type="author"><name><surname>Pivetta</surname><given-names>Emanuele</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/2260062/overview" /><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/><role content-type="https://credit.niso.org/contributor-roles/investigation/"/></contrib>
<contrib contrib-type="author"><name><surname>Bergamasco</surname><given-names>Letizia</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/2417670/overview" /><role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/><role content-type="https://credit.niso.org/contributor-roles/data-curation/"/><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/></contrib>
<contrib contrib-type="author"><name><surname>Franco</surname><given-names>Matteo</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/2999489/overview" /><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/><role content-type="https://credit.niso.org/contributor-roles/visualization/"/><role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/><role content-type="https://credit.niso.org/contributor-roles/data-curation/"/></contrib>
<contrib contrib-type="author"><name><surname>Olmo</surname><given-names>Gabriella</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/1255395/overview" /><role content-type="https://credit.niso.org/contributor-roles/project-administration/"/><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/><role content-type="https://credit.niso.org/contributor-roles/supervision/"/></contrib>
<contrib contrib-type="author"><name><surname>Bondone</surname><given-names>Claudia</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/><role content-type="https://credit.niso.org/contributor-roles/project-administration/"/><role content-type="https://credit.niso.org/contributor-roles/supervision/"/></contrib>
</contrib-group>
<aff id="aff1"><label><sup>1</sup></label><institution>Department of Pediatric Emergency, Regina Margherita Children&#x2019;s Hospital&#x2014;A.O.U. Citt&#x00E0; Della Salute e Della Scienza di Torino</institution>, <addr-line>Turin</addr-line>, <country>Italy</country></aff>
<aff id="aff2"><label><sup>2</sup></label><institution>Department of Public Health and Pediatrics, Postgraduate School of Pediatrics, University of Turin</institution>, <addr-line>Turin</addr-line>, <country>Italy</country></aff>
<aff id="aff3"><label><sup>3</sup></label><institution>Department of Control and Computer Engineering, Politecnico di Torino</institution>, <addr-line>Turin</addr-line>, <country>Italy</country></aff>
<aff id="aff4"><label><sup>4</sup></label><institution>Division of Emergency Medicine and High Dependency Unit, Department of Medical Sciences, Citt&#x00E0; Della Salute e Della Scienza di Torino and University of Turin</institution>, <addr-line>Turin</addr-line>, <country>Italy</country></aff>
<aff id="aff5"><label><sup>5</sup></label><institution>LINKS Foundation</institution>, <addr-line>Turin</addr-line>, <country>Italy</country></aff>
<aff id="aff6"><label><sup>6</sup></label><institution>Department of Clinical and Biological Sciences, University of Turin, Orbassano</institution>, <addr-line>Turin</addr-line>, <country>Italy</country></aff>
<author-notes>
<fn fn-type="edited-by"><p><bold>Edited by:</bold> Xi Long, Eindhoven University of Technology, Netherlands</p></fn>
<fn fn-type="edited-by"><p><bold>Reviewed by:</bold> Zheng Peng, Eindhoven University of Technology, Netherlands</p>
<p>Srinivasan Suresh, University of Pittsburgh, United States</p></fn>
<corresp id="cor1"><label>&#x002A;</label><bold>Correspondence:</bold> Emanuele Castagno <email>ecastagno@cittadellasalute.to.it</email></corresp>
<fn fn-type="equal" id="an1"><label><sup>&#x2020;</sup></label><p>These authors have contributed equally to this work and share first authorship</p></fn>
</author-notes>
<pub-date pub-type="epub"><day>01</day><month>07</month><year>2025</year></pub-date>
<pub-date pub-type="ecorrected"><day>16</day><month>07</month><year>2025</year></pub-date>
<pub-date pub-type="collection"><year>2025</year></pub-date>
<volume>7</volume><elocation-id>1624786</elocation-id>
<history>
<date date-type="received"><day>08</day><month>05</month><year>2025</year></date>
<date date-type="accepted"><day>16</day><month>06</month><year>2025</year></date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2025 Del Monte, Barolo, Circhetta, Delmonaco, Castagno, Pivetta, Bergamasco, Franco, Olmo and Bondone.</copyright-statement>
<copyright-year>2025</copyright-year><copyright-holder>Del Monte, Barolo, Circhetta, Delmonaco, Castagno, Pivetta, Bergamasco, Franco, Olmo and Bondone</copyright-holder><license license-type="open-access" xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract><sec><title>Background</title>
<p>The Pediatric Emergency Department (PED) faces significant challenges, such as high patient volumes, time-sensitive decisions, and complex diagnoses. Large Language Models (LLMs) have the potential to enhance patient care; however, their effectiveness in supporting the diagnostic process remains uncertain, with studies showing mixed results regarding their impact on clinical reasoning. We aimed to assess LLM-based chatbots performance in realistic PED scenarios, and to explore their use as diagnosis-making assistants in pediatric emergency.</p>
</sec><sec><title>Methods</title>
<p>We evaluated the diagnostic effectiveness of 5 LLMs (ChatGPT-4o, Gemini 1.5 Pro, Gemini 1.5 Flash, Llama-3-8B, and ChatGPT-4o mini) compared to 23 physicians (including 10 PED physicians, 6 PED residents, and 7 Emergency Medicine residents). Both LLMs and physicians had to provide one primary diagnosis and two differential diagnoses for 80 real-practice pediatric clinical cases from the PED of a tertiary care Children&#x0027;s Hospital, with three different levels of diagnostic complexity. The responses from both LLMs and physicians were compared to the final diagnoses assigned upon patient discharge; two independent experts evaluated the answers using a five-level accuracy scale. Each physician or LLM received a total score out of 80, based on the sum of all answer points.</p>
</sec><sec><title>Results</title>
<p>The best performing chatbots were ChatGPT-4o (score: 72.5) and Gemini 1.5 Pro (score: 62.75), the first performing better (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.05) than PED physicians (score: 61.88). Emergency Medicine residents performed worse (score: 43.75) than both the other physicians and chatbots (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01). Chatbots&#x0027; performance was inversely proportional to case difficulty, but ChatGPT-4o managed to match the majority of the correct answers even for highly difficult cases.</p>
</sec><sec><title>Discussion</title>
<p>ChatGPT-4o and Gemini 1.5 Pro could be a valid tool for ED physicians, supporting clinical decision-making without replacing the physician&#x0027;s judgment. Shared protocols for effective collaboration between AI chatbots and healthcare professionals are needed.</p>
</sec>
</abstract>
<kwd-group>
<kwd>artificial intelligence</kwd>
<kwd>chatbot</kwd>
<kwd>diagnostic accuracy</kwd>
<kwd>large language model</kwd>
<kwd>pediatric emergency department</kwd>
</kwd-group><counts>
<fig-count count="7"/>
<table-count count="3"/><equation-count count="0"/><ref-count count="27"/><page-count count="9"/><word-count count="0"/></counts><custom-meta-wrap><custom-meta><meta-name>section-at-acceptance</meta-name><meta-value>Connected Health</meta-value></custom-meta></custom-meta-wrap>
</article-meta>
</front>
<body><sec id="s1" sec-type="intro"><label>1</label><title>Introduction</title>
<p>Large Language Models (LLMs) are advanced artificial intelligence (AI) systems that understand and generate natural language (<xref ref-type="bibr" rid="B1">1</xref>). Among the most popular ones, OpenAI&#x0027;s GPT models (<xref ref-type="bibr" rid="B2">2</xref>) such as Chat Generative Pre-trained Transformer (ChatGPT) (<xref ref-type="bibr" rid="B3">3</xref>), Google&#x0027;s Gemini series (<xref ref-type="bibr" rid="B4">4</xref>), and Meta&#x0027;s LLaMA family (<xref ref-type="bibr" rid="B5">5</xref>), gained attention in the open-source community. These models are trained on vast amounts of textual data, and their performance improves as the quantity and quality of training data increase (<xref ref-type="bibr" rid="B1">1</xref>).</p>
<p>LLMs can be applied in clinical decision support, medical record analysis, patient engagement, and dissemination of health information (<xref ref-type="bibr" rid="B6">6</xref>). AI-based tools can support healthcare professionals by offering diagnostic assistance, thereby increasing accuracy, efficiency and enhancing clinical outcomes (<xref ref-type="bibr" rid="B7">7</xref>, <xref ref-type="bibr" rid="B8">8</xref>). However, sometimes their responses could be inaccurate or misleading, underscoring the need for rigorous validation and oversight in clinical settings (<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B10">10</xref>).</p>
<p>In 2023, Kanjee et al. (<xref ref-type="bibr" rid="B11">11</xref>) examined the diagnostic accuracy of ChatGPT-4, showing that AI included the correct diagnosis in differential-diagnosis lists in 64.0&#x0025; of cases, successfully identifying the main diagnosis in 39.0&#x0025;. In the same year, Hirosawa et al. (<xref ref-type="bibr" rid="B12">12</xref>) evaluated ChatGPT-3 on common clinical scenarios, showing that it included the correct diagnosis in 93.3&#x0025; of differential-diagnosis lists, though physicians outperformed the model in ranking accuracy. In a follow-up study (<xref ref-type="bibr" rid="B13">13</xref>), the same team showed that ChatGPT-4 performed better than ChatGPT-3.5 and comparably to physicians, although the differences were not significant. Recently Hirosawa et al. (<xref ref-type="bibr" rid="B14">14</xref>) tested different chatbots on adult cases: ChatGPT-4 achieved the highest accuracy, including correct diagnoses in 86.7&#x0025; of lists and identifying the main diagnosis in 54.6&#x0025; of cases.</p>
<p>To our knowledge, the role of LLMs as a diagnostic support tool in the Pediatric Emergency Department (PED) has not been explored yet. In our pilot study we tested the diagnostic efficacy of some of the most used LLMs on pediatric emergency clinical vignettes and compared their performance to a group of physicians. Our aim was to evaluate whether LLMs can serve as an effective support to ED physicians in formulating accurate diagnoses for pediatric emergency clinical cases.</p>
</sec>
<sec id="s2" sec-type="methods"><label>2</label><title>Materials and methods</title>
<sec id="s2a"><label>2.1</label><title>Study design</title>
<p>This prospective observational diagnostic study was conducted at our PED between March and October 2024. Our tertiary care teaching hospital provides care for critically ill patients younger than 18 years. The study was performed according to the international regulatory guidelines and current codes of Good Epidemiological Practice.</p>
<p>Two experienced pediatricians created a dataset of 80 cases with varying clinical complexity, from different pediatric subspecialties (<xref ref-type="table" rid="T1">Table&#x00A0;1</xref>). We extracted the cases from anonymized records of children admitted to our PED between September 2018 and May 2024. We excluded trauma and cases in which the final diagnosis was reached mainly through laboratory or instrumental tests. Patients and their parents did not provide written or oral informed consent, as all the cases were anonymized before the vignettes were generated and no sensitive data was reported. Since it was not possible to trace the identity of the patients and since this study did not retrospectively influence in any way the clinical management of the cases described, the approval of the Ethics Committee was not necessary.</p>
<table-wrap id="T1" position="float"><label>Table 1</label>
<caption><p>Clinical cases divided by pediatric subspecialties.</p></caption>
<table frame="hsides" rules="groups">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th valign="top" align="left">Pediatric subspecialty</th>
<th valign="top" align="center">Number of cases</th>
<th valign="top" align="center">List of clinical cases</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Respiratory system</td>
<td valign="top" align="center">8 (4)</td>
<td valign="top" align="left"><underline>Bronchiolitis</underline>, Pneumothorax, Foreign body inhalation, <underline>Pneumonia</underline>, Wheezing, <underline>Acute laryngitis</underline>, Pneumomediastinum, <underline>Whooping cough</underline></td>
</tr>
<tr>
<td valign="top" align="left">Infectivology</td>
<td valign="top" align="center">17 (13)</td>
<td valign="top" align="left"><underline>Bronchiolitis, Thyroglossal duct infection, Acute otitis media</underline>, Periorbital cellulitis, <underline>Pneumonia</underline>, Group A beta hemolytic Streptococcus acute pharyngotonsillitis, <underline>Otomastoiditis, Pyelonephritis, Retropharyngeal abscess</underline>, Malaria, Mononucleosis, <underline>Staphylococcal Scalded Skin Syndrome, Osteomyelitis, Meningoencephalitis, Pertussis, Staphylococcal toxic shock syndrome, Acute laryngitis</underline></td>
</tr>
<tr>
<td valign="top" align="left">Orthopedics</td>
<td valign="top" align="center">7 (2)</td>
<td valign="top" align="left">Painful pronation of the elbow, Transient synovitis of the hip, Legg-Calv&#x00E9;-Perthes disease, Epiphysiolysis, Griesel&#x0027;s syndrome, <underline>Osteomyelitis, Osteosarcoma</underline></td>
</tr>
<tr>
<td valign="top" align="left">Ear-Nose-Throat (ENT)</td>
<td valign="top" align="center">5 (4)</td>
<td valign="top" align="left"><underline>Thyroglossal duct infection, Acute otitis media</underline>, Laryngomalacia, <underline>Retropharyngeal abscess, Otomastoiditis</underline></td>
</tr>
<tr>
<td valign="top" align="left">Gastroenterology</td>
<td valign="top" align="center">8 (1)</td>
<td valign="top" align="left">Appendicitis, Intestinal intussusception, Inflammatory bowel disease, Cyclic vomiting syndrome, Biliary tract atresia, Hirschsprung&#x0027;s disease, Functional abdominal pain, <underline>Alagille&#x0027;s syndrome</underline></td>
</tr>
<tr>
<td valign="top" align="left">Oncology</td>
<td valign="top" align="center">4 (2)</td>
<td valign="top" align="left">Osteosarcoma, Leukemia, <underline>Central nervous system tumor</underline> (2 cases, different clinical presentation)</td>
</tr>
<tr>
<td valign="top" align="left">Endocrinology</td>
<td valign="top" align="center">3 (0)</td>
<td valign="top" align="left">Onset of diabetes mellitus type 1, Hypothyroidism, Addison&#x0027;s disease</td>
</tr>
<tr>
<td valign="top" align="left">Haematology</td>
<td valign="top" align="center">7 (2)</td>
<td valign="top" align="left">Immune thrombocytopenia, Post-infectious bone marrow aplasia in patient with spherocytosis, Haemophilia, Post-infectious acute hemolytic anaemia, Acute haemolytic crisis in favism, <underline>Retinal thrombosis in autoimmune disease, Haemolytic-uremic syndrome</underline></td>
</tr>
<tr>
<td valign="top" align="left">Nephrology</td>
<td valign="top" align="center">4 (2)</td>
<td valign="top" align="left">Post-infectious glomerulonephritis, <underline>Pyelonephritis</underline>, Idiopathic nephrotic syndrome, <underline>Haemolytic-uremic syndrome</underline></td>
</tr>
<tr>
<td valign="top" align="left">Immunology and rheumatology</td>
<td valign="top" align="center">7 (4)</td>
<td valign="top" align="left">Kawasaki syndrome, <underline>Sydenham&#x0027;s chorea, Rheumatic disease</underline>, Systemic juvenile idiopathic arthritis, Schoenlein-Henoch purpura, <underline>Ataxia telangiectasia, Retinal thrombosis in autoimmune disease</underline></td>
</tr>
<tr>
<td valign="top" align="left">Neurology</td>
<td valign="top" align="center">17 (6)</td>
<td valign="top" align="left">Guillain-Barr&#x00E9; syndrome, <underline>Charcot-Marie-Tooth disease</underline>, Conversion disorder, Transverse myelitis, Febrile seizures, Trigeminal neuralgia, Central nervous system demyelinating disease, Gastroenteritis-associated seizures, Migraine with aura, Iatrogenic peripheral neuropathy, <underline>Meningoencephalitis, Central nervous system tumor</underline> (2 cases, different clinical presentation), Peripheral paralysis of the VII cranial nerve, <underline>Ataxia telangiectasia</underline>, Narcolepsy, <underline>Sydenham&#x0027;s chorea</underline></td>
</tr>
<tr>
<td valign="top" align="left">Allergology</td>
<td valign="top" align="center">3 (0)</td>
<td valign="top" align="left">Cow&#x0027;s milk protein allergy, Food Protein-Induced Enterocolitis Syndrome, Anaphylactic shock</td>
</tr>
<tr>
<td valign="top" align="left">Cardiology</td>
<td valign="top" align="center">5 (2)</td>
<td valign="top" align="left">Complete atrioventricular block in rare pathology (KSS), Myocarditis/heart failure, Vaso-vagal syncope, <underline>Rheumatic disease, Alagille&#x0027;s syndrome</underline></td>
</tr>
<tr>
<td valign="top" align="left">Dermatology</td>
<td valign="top" align="center">4 (3)</td>
<td valign="top" align="left"><underline>Staphylococcal Scalded Skin Syndrome</underline>, Subgaleal hematoma, <underline>Kwashiorkor, Staphylococcal toxic shock syndrome</underline></td>
</tr>
<tr>
<td valign="top" align="left">Dietetics and Nutrition</td>
<td valign="top" align="center">2 (1)</td>
<td valign="top" align="left">Scurvy, <underline>Kwashiorkor</underline></td>
</tr>
<tr>
<td valign="top" align="left">Genetics</td>
<td valign="top" align="center">2 (2)</td>
<td valign="top" align="left"><underline>Alagille&#x0027;s syndrome, Charcot-Marie-Tooth disease</underline></td>
</tr>
<tr>
<td valign="top" align="left">Toxicology</td>
<td valign="top" align="center">2 (0)</td>
<td valign="top" align="left">Acute accidental intoxication by cannabinoids, Methaemoglobinemia due to local anesthetic</td>
</tr>
<tr>
<td valign="top" align="left">Ophthalmology</td>
<td valign="top" align="center">1 (1)</td>
<td valign="top" align="left"><underline>Retinal thrombosis in autoimmune disease</underline></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="table-fn1a"><p>If the case involved more than one medical subspecialty, it was included in all categories and was underlined in the table. The number of cases for each subspecialty is reported in the second column as the total number (number of cases referring to more than one subspecialty). In the right column, the underlined diagnoses are those referring to more than one subspecialty.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>Each case was used to generate a clinical vignette written in Italian by the two main investigators. The clinical vignettes were prepared both to be input as a prompt to different LLM-based chatbots and to be evaluated by a group of physicians. In each vignette (<xref ref-type="fig" rid="F1">Figure&#x00A0;1</xref>), we presented all the main details as follows: recent and past medical history, relevant family medical history, physical examination and vital signs. Laboratory tests were not reported.</p>
<fig id="F1" position="float"><label>Figure 1</label>
<caption><p>Example of a clinical vignette translated in English, subdivided into its main parts.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fdgth-07-1624786-g001.tif"><alt-text content-type="machine-generated">A structured table outlines a clinical case of a 3-year-old girl admitted with a non-blanching rash, ecchymoses on the limbs, and a recent history of fever and sore throat. The table includes six rows: recent medical history, remote pathological history, family history, physical examination, vital signs, and a concluding question. Notable findings include petechiae on skin and palate, ecchymoses on lower limbs, and normal vital signs. The remote history is unremarkable, and the family history mentions maternal Hashimoto&#x2019;s thyroiditis. The final row poses a clinical question regarding the most likely diagnosis and two differential diagnoses.</alt-text>
</graphic>
</fig>
<p>The vignettes were submitted to a panel of three independent expert pediatricians who validated the cases or recommended a revision. They also independently ranked them according to three levels (lowly difficult, difficult, and highly difficult), based on solving complexity according only to available clinical data. The final level for each case was determined based on the majority agreement among the experts: 20 (25.00&#x0025;) highly difficult, 31 (38.75&#x0025;) difficult, and 29 (36.25&#x0025;) lowly difficult.</p>
<p>The two main investigators evaluated all the answers generated by LLMs and physicians, and statistical analysis was performed.</p>
</sec>
<sec id="s2b"><label>2.2</label><title>LLM-based chatbots answers</title>
<p>We selected four of the highest rated (<xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B16">16</xref>) LLMs publicly available during the period in which this study was conducted: ChatGPT-4o (<xref ref-type="bibr" rid="B17">17</xref>) and ChatGPT-4o mini (<xref ref-type="bibr" rid="B18">18</xref>) (OpenAI); Gemini 1.5 Flash (<xref ref-type="bibr" rid="B19">19</xref>) and Gemini 1.5 Pro (<xref ref-type="bibr" rid="B19">19</xref>) (Google); and Llama-3-8B (<xref ref-type="bibr" rid="B20">20</xref>) Instruct version (Meta), an open-source model satisfying our computational resources constraints. Unlike the other LLMs, which were used through the web interface, Llama-3-8B was deployed in our computing infrastructure and could be used without requiring internet access. The characteristics and access details of the selected LLMs are summarized in <xref ref-type="table" rid="T2">Table&#x00A0;2</xref> (<xref ref-type="bibr" rid="B3">3</xref>, <xref ref-type="bibr" rid="B25">25</xref>&#x2013;<xref ref-type="bibr" rid="B27">27</xref>).</p>
<table-wrap id="T2" position="float"><label>Table 2</label>
<caption><p>LLM-based chatbots selected for the study and their access details.</p></caption>
<table frame="hsides" rules="groups">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th valign="top" align="left">Chatbot access 
details</th>
<th valign="top" align="center">ChatGPT-4o (3)</th>
<th valign="top" align="center">ChatGPT-4o mini (3)</th>
<th valign="top" align="center">Gemini 1.5 Flash (<xref ref-type="bibr" rid="B25">25</xref>)</th>
<th valign="top" align="center">Gemini 1.5 Pro (<xref ref-type="bibr" rid="B26">26</xref>)</th>
<th valign="top" align="center">Llama-3-8B (<xref ref-type="bibr" rid="B27">27</xref>)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Provider</td>
<td valign="top" align="left">OpenAI</td>
<td valign="top" align="left">OpenAI</td>
<td valign="top" align="left">Google</td>
<td valign="top" align="left">Google</td>
<td valign="top" align="left">Meta</td>
</tr>
<tr>
<td valign="top" align="left">Access date</td>
<td valign="top" align="left">August 15, 2024</td>
<td valign="top" align="left">August 26, 2024</td>
<td valign="top" align="left">August 21, 2024</td>
<td valign="top" align="left">August 19&#x2013;20, 2024</td>
<td valign="top" align="left">August 27, 2024</td>
</tr>
<tr>
<td valign="top" align="left">Open-source</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">Yes</td>
</tr>
<tr>
<td valign="top" align="left">Free (at the time of this study)</td>
<td valign="top" align="left">No (free questions available up to a daily limit)</td>
<td valign="top" align="left">Yes, after login</td>
<td valign="top" align="left">Yes, after login</td>
<td valign="top" align="left">No (Free questions available up to a daily limit)</td>
<td valign="top" align="left">Yes</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The chatbots were not provided with any example of the task at hand. Moreover, each vignette was given as a prompt to each chatbot only once in independent chats, to prevent LLMs from applying any learning and inference to subsequent cases. At the end of each vignette, we asked two open-ended questions: &#x201C;What is the most likely diagnosis? Which are the next two more likely differential diagnoses?&#x201D;.</p>
</sec>
<sec id="s2c"><label>2.3</label><title>Physicians&#x0027; answers</title>
<p>Twenty-three physicians were selected to evaluate the clinical cases, including 10 PED physicians with at least 5 years of experience, 6 residents attending their last year of residency in Pediatrics at our PED, and 7 residents attending their last year of residency in Emergency Medicine (EM) at the University of Turin, Italy. These three subgroups were selected to ensure a diverse range of clinical experiences and perspectives. The main investigators were excluded.</p>
<p>Between July and August 2024, the participants were asked to resolve the 80 vignettes through Google Forms. The use of digital resources, textual assistance or consulting colleagues were forbidden, to ensure that the responses were purely the result of the physician&#x0027;s independent clinical reasoning and experience.</p>
<p>The vignettes were presented in random difficulty order and divided in 4 standardized forms with 20 cases each, in order to minimize the risk of fatigue for participants, thus influencing the quality of responses. As for chatbots, we asked the same questions to physicians for each vignette.</p>
</sec>
<sec id="s2d"><label>2.4</label><title>Evaluation method</title>
<p>The answers obtained from LLMs and physicians were independently evaluated by the two main investigators and compared to the final diagnoses established at the time of patients&#x0027; discharge from the PED or following hospitalization. Each answer was evaluated through a 5-point accuracy scale, in order to avoid penalizing incomplete or imprecise diagnoses that still demonstrated adequate clinical reasoning: 1 (correct main diagnosis); 0.75 (if the correct diagnosis was identified within differential diagnoses); 0.5 (if the main diagnosis was correct, but not precise); 0.25 (if the correct diagnosis was identified within differential diagnoses, but not precise); and 0 (both main and differential diagnoses were incorrect).</p>
<p>In case of disagreements between the two main investigators, they reached consensus facing each other. Each physician or LLM received a total score by summing the points obtained from all answers, thus obtaining 80 as the maximum possible score.</p>
</sec>
<sec id="s2e"><label>2.5</label><title>Statistical analysis</title>
<p>Descriptive statistics were presented using mean&#x2009;&#x00B1;&#x2009;standard deviation (SD), and median and Interquartile Range (IQR) to report the performance of LLMs and physicians, as appropriate. Bar charts, stacked charts, and dot plots were used to visualize the total scores obtained by the different groups and comparison. For the statistical analysis, a long-format dataset was created. The distribution of accuracy count was checked using histograms and Q-Q plots. Comparisons between physicians and LLMs were made using the Kruskal&#x2013;Wallis H test. Pairwise comparisons were made using Dunn&#x0027;s procedure with Bonferroni correction for multiple comparisons. Multinomial logistic regression models were used to assess the probability of correct response (accuracy&#x2009;&#x003D;&#x2009;1) of physician groups and chatbots by difficulty of the cases. Adjusted predicted probabilities of scoring one in accuracy and their 95&#x0025; confidence intervals were estimated for each difficulty level and group. The results were reported using a line plot with error bars. Statistical significance was set at <italic>p</italic>&#x2009;&#x003C;&#x2009;0.05. Analyses were conducted using STATA 18.5.</p>
</sec>
</sec>
<sec id="s3" sec-type="results"><label>3</label><title>Results</title>
<p>Overall, we obtained a total of 1,840 responses from the 23 physicians (800 from PED physicians, 480 from PED residents, 560 from EM residents) and 400 responses from the 5 selected chatbots.</p>
<p>The highest and lowest total accuracy scores were obtained respectively by ChatGPT-4o (72.5) and Llama-3-8B (33.75). Gemini 1.5 Flash, ChatGPT-4o mini and Gemini 1.5 Pro scored 56.5, 56.75, and 62.75, respectively. PED physicians (60.88&#x2009;&#x00B1;&#x2009;4.83) and PED residents (63.96&#x2009;&#x00B1;&#x2009;2.3) achieved the highest scores, followed by EM residents (44.25&#x2009;&#x00B1;&#x2009;4.64) (<xref ref-type="fig" rid="F2">Figure&#x00A0;2</xref>; <xref ref-type="sec" rid="s11">Supplementary Material Table 1</xref>).</p>
<fig id="F2" position="float"><label>Figure 2</label>
<caption><p>Total scores for each evaluator, grouped by category. PED, pediatric emergency department; EM, emergency medicine.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fdgth-07-1624786-g002.tif"><alt-text content-type="machine-generated">A dot plot compares total scores among three human groups and five chatbots. The y-axis represents total score, ranging from 30 to 80. Individual scores are shown as colored dots for PED physicians (blue), PED residents (orange), and EM residents (green). Mean chatbot scores are represented as labeled black-outlined circles: ChatGPT-4o scores highest, followed by Gemini 1.5 Pro, ChatGPT-4o mini, and Gemini 1.5 Flash, and lastly Llama-3-8B with the lowest score. A dashed line near 80 suggests a possible upper benchmark. The chart visually contrasts chatbot vs. human diagnostic performance.</alt-text>
</graphic>
</fig>
<p>As regards chatbots, significant difference was found between the total accuracy performance of ChatGPT-4o and ChatGPT-4o mini (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01), and between ChatGPT-4o and Gemini 1.5 Flash (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01). Llama-3-8B performed worse than all the other chatbots (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01). No difference was observed between ChatGPT-4o and Gemini 1.5 Pro (<italic>p</italic>&#x2009;&#x003D;&#x2009;0.26) (<xref ref-type="fig" rid="F3">Figure&#x00A0;3</xref>).</p>
<fig id="F3" position="float"><label>Figure 3</label>
<caption><p>Total scores of chatbots. The &#x002A;&#x002A;&#x002A; above the bar shows the <italic>p</italic>-values of the comparisons of that subject vs. all others. &#x002A;&#x002A;&#x002A;: <italic>p</italic>&#x2009;&#x003C;&#x2009;0.01.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fdgth-07-1624786-g003.tif"><alt-text content-type="machine-generated">A bar graph compares total scores of five chatbots. The y-axis indicates total score, ranging from 0 to 80. From highest to lowest, bars represent: ChatGPT-4o (black), Gemini 1.5 Pro (red), ChatGPT-4o mini (dark gray), Gemini 1.5 Flash (orange), and Llama-3-8B (blue). ChatGPT-4o achieved the highest score, while Llama-3-8B scored the lowest. Asterisks above some comparisons indicate statistically significant differences.</alt-text>
</graphic>
</fig>
<p>Comparing the median total scores of physicians to the single performance of the best performing chatbots (ChatGPT-4o and Gemini 1.5 Pro) (<xref ref-type="fig" rid="F4">Figure&#x00A0;4</xref>), we observed no significant difference between PED residents and chatbots. However, ChatGPT-4o performed better than PED physicians (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.05), while EM residents performed worse than both the other physicians and chatbots (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01).</p>
<fig id="F4" position="float"><label>Figure 4</label>
<caption><p>Total scores of chatbots and physician subgroups. PED, pediatric emergency department; EM, emergency medicine. <sup>a</sup>Median of total score for physicians. &#x002A;: <italic>p</italic>&#x2009;&#x003C;&#x2009;0.05. &#x002A;&#x002A;&#x002A;: <italic>p</italic>&#x2009;&#x003C;&#x2009;0.01.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fdgth-07-1624786-g004.tif"><alt-text content-type="machine-generated">A bar chart compares the performance scores of three human groups and two chatbots. The y-axis represents score from 0 to 80. From left to right, bars represent: PED physicians (blue), PED residents (orange), EM residents (green), ChatGPT-4o (black), and Gemini 1.5 Pro (red). ChatGPT-4o has the highest score, followed by PED residents and Gemini 1.5 Pro. EM residents score the lowest, marked with three asterisks, indicating statistical significance. A single asterisk above the top of the chart shows a significant difference between selected groups.</alt-text>
</graphic>
</fig>
<p>In lowly difficult cases, all chatbots but Llama-3-8B performed well; Llama-3-8B showed a significant difference compared to other chatbots (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01). In difficult cases, ChatGPT-4o performed better than Gemini 1.5 Flash (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.05) and Llama-3-8B (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01). Gemini 1.5 Pro and ChatGPT-4o mini performed better than Llama-3-8B (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01). We did not find any significant between ChatGPT-4o and Gemini 1.5 Pro and between Gemini 1.5 Flash and Llama-3-8B. As regards highly difficult cases, ChatGPT-4o performed significantly better than ChatGPT-4o mini (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01) and Llama-3-8B (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01); also Gemini 1.5 Pro performed significantly better than Llama-3-8B (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01) (<xref ref-type="fig" rid="F5">Figure&#x00A0;5</xref>). ChatGPT-4o showed not only higher performance, but also better accuracy (<xref ref-type="fig" rid="F5">Figure&#x00A0;5C</xref>), providing completely incorrect answers only in 4/80 cases (3 difficult, 1 highly difficult).</p>
<fig id="F5" position="float"><label>Figure 5</label>
<caption><p>Chatbots&#x2019; diagnostic performance by case difficulty. Panels on the left show the total scores for each chatbot; panels on the right show the frequency of accuracy levels achieved. <bold>(A)</bold> Lowly difficult cases; <bold>(B)</bold> difficult cases; <bold>(C)</bold> highly difficult cases. The dashed line shows the maximum obtainable total score for the specific difficulty level. &#x002A;: <italic>p</italic>-value&#x2009;&#x003C;&#x2009;0.05. &#x002A;&#x002A;&#x002A;: <italic>p</italic>-value&#x2009;&#x003C;&#x2009;0.01.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fdgth-07-1624786-g005.tif"><alt-text content-type="machine-generated">Three pairs of bar plots compare chatbot performance across low, moderate, and high difficulty levels. Left panels show total scores for five models: ChatGPT-4o (black), Gemini 1.5 Pro (red), ChatGPT-4o mini (gray), Gemini 1.5 Flash (orange), and Llama-3-8B (blue). Right panels display frequency of correct responses, color-coded from red (0) to green (1). ChatGPT-4o consistently scores highest, especially on highly difficult cases. Llama-3-8B performs worst overall. Statistical significance is marked with asterisks. Performance drops as difficulty increases, with more red/yellow bars in right panels.</alt-text>
</graphic>
</fig>
<p>Last, we compared the two best performing chatbots (ChatGPT-4o and Gemini 1.5 Pro) to the median score obtained from the subgroups of physicians, stratified by difficulty (<xref ref-type="fig" rid="F6">Figure&#x00A0;6</xref>). As regards the lowly and highly difficult cases, PED physicians, PED residents and both chatbots performed significantly better than EM residents (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01). In difficult cases, PED physicians, PED residents and ChatGPT-4o performed significantly better than EM residents (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01), but not Gemini 1.5 Pro (<italic>p</italic>&#x2009;&#x003E;&#x2009;0.05). In highly difficult cases, both ChatGPT-4o and Gemini 1.5 Pro performed better than PED physicians and PED residents; however, statistical significance was reached only in the comparison between ChatGPT-4o and PED physicians (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01).</p>
<fig id="F6" position="float"><label>Figure 6</label>
<caption><p>Score of the best performing chatbots (ChatGPT-4o and Gemini 1.5 Pro) compared to the median score obtained from physician subgroups, stratified by case difficulty. <bold>(A)</bold> Lowly difficult cases; <bold>(B)</bold> difficult cases; <bold>(C)</bold> highly difficult cases. PED, pediatric emergency department; EM, emergency medicine. The dashed line shows the maximum obtainable score for the specific difficulty level. &#x002A;: <italic>p</italic>-value&#x2009;&#x003C;&#x2009;0.05. &#x002A;&#x002A;&#x002A;: <italic>p</italic>-value&#x2009;&#x003C;&#x2009;0.01.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fdgth-07-1624786-g006.tif"><alt-text content-type="machine-generated">Three bar charts compare scores across task difficulty levels (lowly difficult, difficult, highly difficult) for three human groups and two chatbots. Each chart includes five bars: PED physicians (blue), PED residents (orange), EM residents (green), ChatGPT-4o (black), and Gemini 1.5 Pro (red). In the low difficulty chart, all groups perform similarly except EM residents, who score significantly lower. In the difficult and highly difficult charts, ChatGPT-4o outperforms all other groups, followed by Gemini 1.5 Pro. EM residents consistently score lowest. Asterisks indicate statistically significant differences. Dashed purple lines mark maximum possible scores for each level.</alt-text>
</graphic>
</fig>
<p><xref ref-type="fig" rid="F7">Figure&#x00A0;7</xref> illustrates the adjusted predictions and their 95&#x0025; confidence interval for the probability of giving the right answer in the &#x201C;main diagnosis&#x201D;, stratified by vignette difficulty. The adjusted prediction of the probability of obtaining the highest accuracy score (score&#x2009;&#x003D;&#x2009;1) was very close across all levels of difficulty for PED physicians and PED residents, with their respective confidence intervals overlapping. EM residents showed lower probability of obtaining the maximum level of accuracy than both PED physicians and residents groups, and chatbots in all levels of difficulty. ChatGPT-4o showed marginally better probability prediction than Gemini 1.5 Pro and the other groups, particularly for highly difficult cases. However, due to the single imputation, it retained broad confidence intervals.</p>
<fig id="F7" position="float"><label>Figure 7</label>
<caption><p>Adjusted predictions and their 95&#x0025; confidence intervals of the probability of identifying the correct answer of the vignettes in the main diagnosis (accuracy&#x2009;&#x003D;&#x2009;1) for subgroups of physicians, ChatGPT-4o and Gemini 1.5 Pro, stratified by difficulty. PED, pediatric emergency department; EM, emergency medicine.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fdgth-07-1624786-g007.tif"><alt-text content-type="machine-generated">A line chart shows accuracy probability trends across three difficulty levels-lowly difficult, difficult, and highly difficult-for three human groups and two chatbots. The y-axis represents the probability of a correct answer, ranging from 0 to 1. Lines with error bars represent: PED physicians (blue), PED residents (orange), EM residents (green), ChatGPT-4o (black), and Gemini 1.5 Pro (red). Accuracy decreases with increasing difficulty for all groups. ChatGPT-4o maintains the highest accuracy overall, while EM residents show the steepest decline, with the lowest accuracy in the highly difficult category. Error bars illustrate variability.</alt-text>
</graphic>
</fig>
<p>Moreover, <xref ref-type="table" rid="T3">Table&#x00A0;3</xref> shows the obtained estimates of the multinomial logistic regression, i.e., the Relative Risk Ratio (RRR) for the probability of scoring 1 (vs. 0) by evaluators. Given the same difficulty, EM residents showed a 76&#x0025; lower probability of scoring 1 (vs. 0) than PED physicians. Furthermore, ChatGPT-4o had a 376&#x0025; higher probability of scoring 1 (vs. 0) than PED physicians. However, the estimates show a very wide 95&#x0025; confidence interval, due to the comparison between one measurement (ChatGPT-4o) vs. multiple measurements (group of PED physicians).</p>
<table-wrap id="T3" position="float"><label>Table 3</label>
<caption><p>Multinomial regression: relative risk ratio (RRR) for the probability of scoring 1 (vs. 0) by evaluators (physician groups, ChatGPT-4o, and Gemini 1.5 Pro).</p></caption>
<table frame="hsides" rules="groups">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th valign="top" align="left">Accuracy</th>
<th valign="top" align="left">Groups</th>
<th valign="top" align="center">RRR</th>
<th valign="top" align="center">SD</th>
<th valign="top" align="center"><italic>p</italic>-value</th>
<th valign="top" align="center">(95&#x0025; CI)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left" rowspan="5">1 vs. 0</td>
<td valign="top" align="left">PED physicians</td>
<td valign="top" align="center">1</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">PED residents</td>
<td valign="top" align="center">1.39</td>
<td valign="top" align="center">0.25</td>
<td valign="top" align="center">0.07</td>
<td valign="top" align="center">(0.98&#x2013;1.97)</td>
</tr>
<tr>
<td valign="top" align="left">EM residents</td>
<td valign="top" align="center">0.24</td>
<td valign="top" align="center">0.04</td>
<td valign="top" align="center">&#x003C;0.01</td>
<td valign="top" align="center">(0.17&#x2013;0.32)<xref ref-type="table-fn" rid="table-fn2">&#x002A;&#x002A;&#x002A;</xref></td>
</tr>
<tr>
<td valign="top" align="left">ChatGPT-4o</td>
<td valign="top" align="center">4.76</td>
<td valign="top" align="center">2.56</td>
<td valign="top" align="center">&#x003C;0.01</td>
<td valign="top" align="center">(1.66&#x2013;13.67)<xref ref-type="table-fn" rid="table-fn2">&#x002A;&#x002A;&#x002A;</xref></td>
</tr>
<tr>
<td valign="top" align="left">Gemini 1.5 Pro</td>
<td valign="top" align="center">1.04</td>
<td valign="top" align="center">0.37</td>
<td valign="top" align="center">0.90</td>
<td valign="top" align="center">(0.53&#x2013;2.07)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="table-fn1"><p>Adjustment: difficulty.</p></fn>
<fn id="table-fn2"><label>&#x002A;&#x002A;&#x002A;</label>
<p><italic>p</italic>&#x2009;&#x003C;&#x2009;0.01.</p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4" sec-type="discussion"><label>4</label><title>Discussion</title>
<p>To our knowledge, this is the first study exploring the role of LLMs as diagnostic support tools in pediatric emergency cases. Among the tested chatbots, ChatGPT-4o achieved the highest accuracy, with most diagnoses aligning with correct answers for any level of complexity. In fact, ChatGPT provided a completely incorrect answer, scoring 0, in only 4 cases out of 80 (3 classified as difficult, and 1 as highly difficult). Gemini 1.5 Pro performed slightly below ChatGPT-4o, being more affected by case difficulty. Gemini 1.5 Flash and ChatGPT-4o mini achieved similar performance, but were inferior to ChatGPT-4o and Gemini 1.5 Pro: their performance was notably better in simpler cases, while it dropped in difficult and highly difficult cases. In contrast, Llama-3-8B showed significantly lower performance than all the other LLMs considered in this research. This was aligned with expectations, as it had only 8 billion parameters and the lowest scores on benchmarks (<xref ref-type="bibr" rid="B1">1</xref>, <xref ref-type="bibr" rid="B15">15</xref>) and leaderboards (<xref ref-type="bibr" rid="B16">16</xref>). However, during the study period, models like Gemini 1.5 Pro and ChatGPT-4o were paid services, with free questions available up to a daily limit; this may represent a limitation for some users. On the other hand, Llama-3-8B is open-source, free, and offers greater data privacy when used on-premises, though it requires a more complex setup and adequate computational resources compared to web-based chatbots. With more computational available resources, larger models such as Llama-3-70B (<xref ref-type="bibr" rid="B20">20</xref>) could be tested, offering significantly more parameters and potentially better performance.</p>
<p>Ultimately, this study underscores the importance of human oversight in the use of LLMs, as their success in healthcare stands on accurate data collection (e.g., medical history, physical examinations and vital signs) and interpretation, which only qualified practitioners can provide. LLMs are designed to complement physicians (<xref ref-type="bibr" rid="B21">21</xref>), whose role is not replaceable by AI since clinical data must be evaluated by a human and then be presented to AI in the correct way, such as in terms of language, in order to be analyzed effectively and usefully. Establishing specific clinical guidelines and protocols for the use of AI in healthcare is crucial to ensure in the future the safe integration of these tools into clinical practice. Looking ahead, the integration of LLMs into PED workflows such as electronic health records or diagnostic decision support systems is a desirable goal, but remains premature at this stage. Further research is needed to assess their reliability, clinical utility, and safe implementation in real-time diagnostic settings.</p>
<p>Regarding the physician groups, there was no significant difference in diagnostic accuracy between PED physicians and PED residents, while a clear difference emerged between EM residents and the two pediatric physician groups. As expected, all the human subgroups showed a decline in diagnostic accuracy as case complexity increased. ChatGPT-4o and Gemini 1.5 Pro performed like PED physicians and PED residents in lowly difficult and difficult cases, and proved to be effective aids in solving highly difficult cases (e.g., rare, complex diseases).</p>
<p>Interestingly, ChatGPT-4o performed better than both PED residents and PED physicians, but significance was reached only vs. the latter, particularly in highly difficult cases. This observation is difficult to interpret and could be due to different physician&#x0027;s subgroups sample size. We can argue that PED residents performed better than PED physicians in those cases requiring knowledge of rare internal conditions, due to their more recent training. Anyway, our results cannot support this hypothesis and further investigation on a larger sample should be carried out.</p>
<p>All LLMs outperformed EM residents, likely due to their limited experience with pediatrics cases. In situations where a pediatrician is not immediately available, EM physicians could leverage the insights provided by LLMs alongside their own knowledge, allowing for initial diagnostic hypotheses. In the fast-paced ED environment, this could be a valuable advantage, speeding up the diagnostic process. On the other hand, our observation highlights the importance of implementing pediatric skills for EM residents, as in many cases children accessing the EDs are first evaluated by adult EM specialists, and not by specifically trained pediatricians. Pediatric skills should be not only acquired, but also maintained through longitudinal training programs during residency, as recently proposed (<xref ref-type="bibr" rid="B22">22</xref>).</p>
<p>While our study demonstrates the effectiveness of advanced LLMs in pediatric cases, a similar study by Barile et al. (<xref ref-type="bibr" rid="B23">23</xref>) showed significantly poorer outcomes using ChatGPT-3.5. Their investigation on 100 pediatric case challenges found a diagnostic error rate of 83&#x0025;, highlighting limitations of older LLM versions. In contrast, our results indicate that state-of-art models (i.e., ChatGPT-4o and Gemini 1.5 Pro) achieved diagnostic accuracy comparable or even better than emergency pediatricians. This observation underscores the rapid advances in LLM technology and the importance of leveraging the most up-to-date tools to maximize clinical usefulness.</p>
<p>In fact, a general limitation when trying to evaluate LLMs performance in each context is the rapid advancement of these technologies, which can quickly make the results outdated. Moreover, LLMs are limited by the point in time when their training data are updated. If they are not fine-tuned or updated periodically, they may lack awareness of more recent data and information.</p>
<p>Our study has some strengths. First, we evaluated the effectiveness of the latest available versions of LLMs, ranked among the top models on the Chatbot Arena leaderboard (<xref ref-type="bibr" rid="B16">16</xref>) and across various benchmarks (<xref ref-type="bibr" rid="B1">1</xref>, <xref ref-type="bibr" rid="B15">15</xref>). Such chatbots differ in model size, provider, user-interface, and availability. In contrast, many previous studies have focused on a single model, often an earlier version of ChatGPT (<xref ref-type="bibr" rid="B11">11</xref>&#x2013;<xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B23">23</xref>). Second, we considered three distinct groups of physicians, allowing for diverse perspectives and detailed insights in addressing the assigned tasks. Last, we introduced a non-binary evaluation approach, using multiple accuracy categories to allow for more nuanced assessments.</p>
<p>Our study also has some limits. First, as LLMs may show a lack of reproducibility, they could produce different responses when presented with the same case multiple times, sometimes reversing the order of diagnoses. This issue was not explored in our research.</p>
<p>Second, to avoid potential learning or contamination effects across prompt repetitions, each vignette was submitted only once per LLM. However, this approach prevents the assessment of intra-model variability. Future work should include repeated sampling to better quantify the consistency and stability of LLM-generated outputs. Sequential inputs or follow-up questions could also be explored, to simulate more closely real clinical conversations and evaluate their impact on diagnostic reasoning performance.</p>
<p>Moreover, when analyzing the physicians&#x2019; responses, we did not consider factors like a distracting environment, focus level, and stress or fatigue, which may increase inaccuracies, especially at the end of the forms. On the other hand, the process of reasoning on a clinical vignette is different from reasoning in front of a real patient: the clinical impression &#x201C;at first sight&#x201D; is crucial to reach the correct diagnosis and could be difficult to reproduce by written description (<xref ref-type="bibr" rid="B24">24</xref>). Such limitations do not affect the responses provided by chatbots.</p>
<p>Furthermore, the varying number of cases across difficulty levels, with only 20 cases for the hardest ones, represents a limitation. Another limit is the non-homogeneity of the number of physicians per group. This may have affected the reliability of estimates for smaller groups, as they are more sensitive to outliers. In statistical analysis, physicians&#x0027; performances were summarized using the median score and compared with the absolute score of each chatbot. This difference in measurement may limit the accuracy of direct comparisons, affecting the generalizability of the results.</p>
<p>In conclusion, the results of our pilot study highlight the importance of understanding the diagnostic performance among different LLMs, especially in more complex PED clinical cases. Our observations suggest that certain LLMs, especially ChatGPT-4o and Gemini 1.5 Pro, have diagnostic efficacy similar to or even better than those of pediatricians. Due to their high level of accuracy, LLMs could serve as a valuable tool to support PED physicians in solving the most difficult pediatric emergency cases, and they can be a very useful tool for EM physicians for all degrees of difficulty of pediatric cases. However, LLMs should never substitute human clinical judgement.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability"><title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s6" sec-type="ethics-statement"><title>Ethics statement</title>
<p>Ethical approval was not required for the study involving humans in accordance with the local legislation and institutional requirements. Written informed consent to participate in this study was not required from the participants or the participants&#x0027; legal guardians/next of kin in accordance with the national legislation and the institutional requirements.</p>
</sec>
<sec id="s7" sec-type="author-contributions"><title>Author contributions</title>
<p>FDM: Writing &#x2013; original draft, Data curation, Conceptualization, Methodology. RB: Conceptualization, Investigation, Writing &#x2013; original draft. MC: Formal analysis, Data curation, Writing &#x2013; review &#x0026; editing. AGD: Validation, Conceptualization, Resources, Writing &#x2013; review &#x0026; editing. EC: Writing &#x2013; original draft, Methodology, Visualization, Conceptualization. EP: Writing &#x2013; review &#x0026; editing, Investigation. LB: Formal analysis, Data curation, Writing &#x2013; review &#x0026; editing. MF: Writing &#x2013; review &#x0026; editing, Visualization, Formal analysis, Data curation. GO: Project administration, Writing &#x2013; review &#x0026; editing, Supervision. CB: Writing &#x2013; review &#x0026; editing, Project administration, Supervision.</p>
</sec>
<sec id="s8" sec-type="funding-information"><title>Funding</title>
<p>The author(s) declare that no financial support was received for the research and/or publication of this article.</p>
</sec>
<sec id="s9" sec-type="COI-statement"><title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="ai-statement"><title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="s12" sec-type="other"><title>Correction note</title>
<p>A correction has been made to this article. Details can be found at: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fdgth.2025.1658635">10.3389/fdgth.2025.1658635</ext-link>.</p>
</sec>
<sec id="s13" sec-type="disclaimer"><title>Publisher&#x0027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11" sec-type="supplementary-material"><title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fdgth.2025.1624786/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fdgth.2025.1624786/full&#x0023;supplementary-material</ext-link></p>
<supplementary-material id="SD1" content-type="local-data">
<media mimetype="application" mime-subtype="vnd.openxmlformats-officedocument.wordprocessingml.document" xlink:href="Table1.docx"/></supplementary-material>
</sec>
<ref-list><title>References</title>
<ref id="B1"><label>1.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Minaee</surname><given-names>S</given-names></name><name><surname>Mikolov</surname><given-names>T</given-names></name><name><surname>Nikzad</surname><given-names>N</given-names></name><name><surname>Chenaghlu</surname><given-names>M</given-names></name><name><surname>Socher</surname><given-names>R</given-names></name><name><surname>Amatriain</surname><given-names>X</given-names></name><etal/></person-group> <article-title>Large language models: a survey. arXiv [preprint]</article-title>. (<year>2024</year>). <comment>Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2402.06196v3">https://arxiv.org/abs/2402.06196v3</ext-link> (<comment>Accessed June 23, 2025</comment>).</citation></ref>
<ref id="B2"><label>2.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Achiam</surname><given-names>J</given-names></name><name><surname>Adler</surname><given-names>S</given-names></name><name><surname>Agarwal</surname><given-names>S</given-names></name><name><surname>Ahmad</surname><given-names>L</given-names></name><name><surname>Akkaya</surname><given-names>I</given-names></name><name><surname>Aleman</surname><given-names>FL</given-names></name><etal/></person-group> <article-title>GPT-4 Technical Report. arXiv [preprint]</article-title>. (<year>2023</year>). <comment>Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2303.08774">https://arxiv.org/abs/2303.08774</ext-link> <comment>(Accessed August 31, 2024)</comment>.</citation></ref>
<ref id="B3"><label>3.</label><citation citation-type="book"><collab>OpenAI, Inc</collab>. <source>ChatGPT</source>. <comment>Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://chatgpt.com/">https://chatgpt.com/</ext-link> <comment>(Accessed August 31, 2024)</comment>.</citation></ref>
<ref id="B4"><label>4.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Anil</surname><given-names>R</given-names></name><name><surname>Borgeaud</surname><given-names>S</given-names></name><name><surname>Alayrac</surname><given-names>JB</given-names></name><name><surname>Yu</surname><given-names>J</given-names></name><name><surname>Soricut</surname><given-names>R</given-names></name><name><surname>Schalkwyk</surname><given-names>J</given-names></name><etal/></person-group> <article-title>Gemini: a family of highly capable multimodal models. arXiv [preprint]</article-title>. (<year>2023</year>). <comment>Available at</comment>: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2312.11805">https://arxiv.org/abs/2312.11805</ext-link> <comment>(Accessed August 31, 2024)</comment>.</citation></ref>
<ref id="B5"><label>5.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Touvron</surname><given-names>H</given-names></name><name><surname>Lavril</surname><given-names>T</given-names></name><name><surname>Izacard</surname><given-names>G</given-names></name><name><surname>Martinet</surname><given-names>X</given-names></name><name><surname>Lachaux</surname><given-names>MA</given-names></name><name><surname>Lacroix</surname><given-names>T</given-names></name><etal/></person-group> <article-title>LLaMA: Open and Efficient Foundation Language Models. arXiv [preprint]</article-title>. (<year>2023</year>). <comment>Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2302.13971">https://arxiv.org/abs/2302.13971</ext-link> <comment>(Accessed August 31, 2024)</comment>.</citation></ref>
<ref id="B6"><label>6.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Preiksaitis</surname><given-names>C</given-names></name><name><surname>Ashenburg</surname><given-names>N</given-names></name><name><surname>Bunney</surname><given-names>G</given-names></name><name><surname>Chu</surname><given-names>A</given-names></name><name><surname>Kabeer</surname><given-names>R</given-names></name><name><surname>Riley</surname><given-names>F</given-names></name><etal/></person-group> <article-title>The role of large language models in transforming emergency medicine: scoping review</article-title>. <source>JMIR Med Inform</source>. (<year>2024</year>) <volume>12</volume>:<fpage>e53787</fpage>. <pub-id pub-id-type="doi">10.2196/53787</pub-id><pub-id pub-id-type="pmid">38728687</pub-id></citation></ref>
<ref id="B7"><label>7.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Nazi</surname><given-names>ZA</given-names></name><name><surname>Peng</surname><given-names>W</given-names></name></person-group>. <article-title>Large language models in healthcare and medical domain: a review. arXiv [preprint]</article-title>. (<year>2023</year>). <comment>Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2401.06775">https://arxiv.org/abs/2401.06775</ext-link> <comment>(Accessed August 31, 2024)</comment>.</citation></ref>
<ref id="B8"><label>8.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Demirba&#x015F;</surname><given-names>KC</given-names></name><name><surname>Y&#x0131;ld&#x0131;z</surname><given-names>M</given-names></name><name><surname>Sayg&#x0131;l&#x0131;</surname><given-names>S</given-names></name><name><surname>Canpolat</surname><given-names>N</given-names></name><name><surname>Kasap&#x00E7;opur</surname><given-names>&#x00D6;</given-names></name></person-group>. <article-title>Artificial intelligence in pediatrics: learning to walk together</article-title>. <source>Turk Arch Pediatr</source>. (<year>2024</year>) <volume>59</volume>(<issue>2</issue>):<fpage>121</fpage>&#x2013;<lpage>30</lpage>. <pub-id pub-id-type="doi">10.5152/turkarchpediatr.2024.24002</pub-id></citation></ref>
<ref id="B9"><label>9.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tian</surname><given-names>S</given-names></name><name><surname>Jin</surname><given-names>Q</given-names></name><name><surname>Yeganova</surname><given-names>L</given-names></name><name><surname>Lai</surname><given-names>PT</given-names></name><name><surname>Zhu</surname><given-names>Q</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><etal/></person-group> <article-title>Opportunities and challenges for ChatGPT and large language models in biomedicine and health</article-title>. <source>Brief Bioinform</source>. (<year>2023</year>) <volume>25</volume>(<issue>1</issue>):<fpage>bbad493</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbad493</pub-id><pub-id pub-id-type="pmid">38168838</pub-id></citation></ref>
<ref id="B10"><label>10.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Omiye</surname><given-names>JA</given-names></name><name><surname>Gui</surname><given-names>H</given-names></name><name><surname>Rezaei</surname><given-names>SJ</given-names></name><name><surname>Zou</surname><given-names>J</given-names></name><name><surname>Daneshjou</surname><given-names>R</given-names></name></person-group>. <article-title>Large language models in medicine: the potentials and pitfalls</article-title>. <source>Ann Intern Med</source>. (<year>2024</year>) <volume>177</volume>(<issue>2</issue>):<fpage>210</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.7326/M23-2772</pub-id><pub-id pub-id-type="pmid">38285984</pub-id></citation></ref>
<ref id="B11"><label>11.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kanjee</surname><given-names>Z</given-names></name><name><surname>Crowe</surname><given-names>B</given-names></name><name><surname>Rodman</surname><given-names>A</given-names></name></person-group>. <article-title>Accuracy of a generative artificial intelligence model in a complex diagnostic challenge</article-title>. <source>JAMA</source>. (<year>2023</year>) <volume>330</volume>(<issue>1</issue>):<fpage>78</fpage>. <pub-id pub-id-type="doi">10.1001/jama.2023.8288</pub-id><pub-id pub-id-type="pmid">37318797</pub-id></citation></ref>
<ref id="B12"><label>12.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hirosawa</surname><given-names>T</given-names></name><name><surname>Harada</surname><given-names>Y</given-names></name><name><surname>Yokose</surname><given-names>M</given-names></name><name><surname>Sakamoto</surname><given-names>T</given-names></name><name><surname>Kawamura</surname><given-names>R</given-names></name><name><surname>Shimizu</surname><given-names>T</given-names></name></person-group>. <article-title>Diagnostic accuracy of differential-diagnosis lists generated by generative pretrained transformer 3 chatbot for clinical vignettes with common chief complaints: a pilot study</article-title>. <source>Int J Environ Res Public Health</source>. (<year>2023</year>) <volume>20</volume>(<issue>4</issue>):<fpage>3378</fpage>. <pub-id pub-id-type="doi">10.3390/ijerph20043378</pub-id><pub-id pub-id-type="pmid">36834073</pub-id></citation></ref>
<ref id="B13"><label>13.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hirosawa</surname><given-names>T</given-names></name><name><surname>Kawamura</surname><given-names>R</given-names></name><name><surname>Harada</surname><given-names>Y</given-names></name><name><surname>Mizuta</surname><given-names>K</given-names></name><name><surname>Tokumasu</surname><given-names>K</given-names></name><name><surname>Kaji</surname><given-names>Y</given-names></name><etal/></person-group> <article-title>ChatGPT-generated differential diagnosis lists for complex case&#x2013;derived clinical vignettes: diagnostic accuracy evaluation</article-title>. <source>JMIR Med Inform</source>. (<year>2023</year>) <volume>11</volume>:<fpage>e48808</fpage>. <pub-id pub-id-type="doi">10.2196/48808</pub-id><pub-id pub-id-type="pmid">37812468</pub-id></citation></ref>
<ref id="B14"><label>14.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hirosawa</surname><given-names>T</given-names></name><name><surname>Harada</surname><given-names>Y</given-names></name><name><surname>Mizuta</surname><given-names>K</given-names></name><name><surname>Sakamoto</surname><given-names>T</given-names></name><name><surname>Tokumasu</surname><given-names>K</given-names></name><name><surname>Shimizu</surname><given-names>T</given-names></name></person-group>. <article-title>Diagnostic performance of generative artificial intelligences for a series of complex case reports</article-title>. <source>Digit Health</source>. (<year>2024</year>) <volume>10</volume>:<fpage>20552076241265215</fpage>. <pub-id pub-id-type="doi">10.1177/20552076241265215</pub-id><pub-id pub-id-type="pmid">39229463</pub-id></citation></ref>
<ref id="B15"><label>15.</label><citation citation-type="book"><collab>ArtificialAnalysis</collab>. <source>Comparison of Models: Intelligence, Performance &#x0026; Price Analysis</source>. <comment>Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://artificialanalysis.ai/models">https://artificialanalysis.ai/models</ext-link> <comment>(Accessed August 31, 2024)</comment>.</citation></ref>
<ref id="B16"><label>16.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Chiang</surname><given-names>WL</given-names></name><name><surname>Zheng</surname><given-names>L</given-names></name><name><surname>Sheng</surname><given-names>Y</given-names></name><name><surname>Angelopoulos</surname><given-names>AN</given-names></name><name><surname>Li</surname><given-names>T</given-names></name><name><surname>Li</surname><given-names>D</given-names></name><etal/></person-group> <article-title>Chatbot Arena: an open platform for evaluating LLMs by human preference. arXiv [preprint]</article-title>. (<year>2024</year>). <comment>Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2403.04132">https://arxiv.org/abs/2403.04132</ext-link> <comment>(Accessed August 31, 2024)</comment>.</citation></ref>
<ref id="B17"><label>17.</label><citation citation-type="book"><collab>OpenAI, Inc</collab>. <source>Hello GPT-4o</source>. (<year>2024</year>). <comment>Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://openai.com/index/hello-gpt-4o/">https://openai.com/index/hello-gpt-4o/</ext-link> <comment>(Accessed August 31, 2024)</comment>.</citation></ref>
<ref id="B18"><label>18.</label><citation citation-type="book"><collab>OpenAI, Inc</collab>. <source>GPT-4o Mini: Advancing Cost-Efficient Intelligence</source>. (<year>2024</year>). <comment>Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://openai.com/index/gpt-4o-mini-advancing-cost-efficient-intelligence/">https://openai.com/index/gpt-4o-mini-advancing-cost-efficient-intelligence/</ext-link> <comment>(Accessed August 31, 2024)</comment>.</citation></ref>
<ref id="B19"><label>19.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Georgiev</surname><given-names>P</given-names></name><name><surname>Lei</surname><given-names>VI</given-names></name><name><surname>Burnell</surname><given-names>R</given-names></name><name><surname>Bai</surname><given-names>L</given-names></name><name><surname>Gulati</surname><given-names>A</given-names></name><name><surname>Tanzer</surname><given-names>G</given-names></name><etal/></person-group> <article-title>Gemini 1.5: unlocking multimodal understanding across millions of tokens of context. arXiv [preprint]</article-title>. (<year>2024</year>). <comment>Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2403.05530v5">https://arxiv.org/abs/2403.05530v5</ext-link> (<comment>Accessed June 23, 2025</comment>).</citation></ref>
<ref id="B20"><label>20.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Grattafiori</surname><given-names>A</given-names></name><name><surname>Dubey</surname><given-names>A</given-names></name><name><surname>Jauhri</surname><given-names>A</given-names></name><name><surname>Pandey</surname><given-names>A</given-names></name><name><surname>Kadian</surname><given-names>A</given-names></name><name><surname>Al-Dahle</surname><given-names>A</given-names></name><etal/></person-group> <article-title>The Llama 3 herd of models. arXiv [preprint]</article-title>. (<year>2024</year>). <comment>Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2407.21783v3">https://arxiv.org/abs/2407.21783v3</ext-link> (<comment>Accessed June 23, 2025</comment>).</citation></ref>
<ref id="B21"><label>21.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sezgin</surname><given-names>E</given-names></name></person-group>. <article-title>Artificial intelligence in healthcare: complementing, not replacing, doctors and healthcare providers</article-title>. <source>Digit Health</source>. (<year>2023</year>) <volume>9</volume>:<fpage>20552076231186520</fpage>. <pub-id pub-id-type="doi">10.1177/20552076231186520</pub-id><pub-id pub-id-type="pmid">37426593</pub-id></citation></ref>
<ref id="B22"><label>22.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clayton</surname><given-names>L</given-names></name><name><surname>Wells</surname><given-names>M</given-names></name><name><surname>Alter</surname><given-names>S</given-names></name><name><surname>Solano</surname><given-names>J</given-names></name><name><surname>Hughes</surname><given-names>P</given-names></name><name><surname>Shih</surname><given-names>R</given-names></name></person-group>. <article-title>Educational concepts: a longitudinal interleaved curriculum for emergency medicine residency training</article-title>. <source>JACEP Open</source>. (<year>2024</year>) <volume>5</volume>(<issue>3</issue>):<fpage>e13223</fpage>. <pub-id pub-id-type="doi">10.1002/emp2.13223</pub-id><pub-id pub-id-type="pmid">38903766</pub-id></citation></ref>
<ref id="B23"><label>23.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barile</surname><given-names>J</given-names></name><name><surname>Margolis</surname><given-names>A</given-names></name><name><surname>Cason</surname><given-names>G</given-names></name><name><surname>Kim</surname><given-names>R</given-names></name><name><surname>Kalash</surname><given-names>S</given-names></name><name><surname>Tchaconas</surname><given-names>A</given-names></name><etal/></person-group> <article-title>Diagnostic accuracy of a large language model in pediatric case studies</article-title>. <source>JAMA Pediatr</source>. (<year>2024</year>) <volume>178</volume>(<issue>3</issue>):<fpage>313</fpage>. <pub-id pub-id-type="doi">10.1001/jamapediatrics.2023.5750</pub-id><pub-id pub-id-type="pmid">38165685</pub-id></citation></ref>
<ref id="B24"><label>24.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Knack</surname><given-names>SKS</given-names></name><name><surname>Scott</surname><given-names>N</given-names></name><name><surname>Driver</surname><given-names>BE</given-names></name><name><surname>Prekker</surname><given-names>ME</given-names></name><name><surname>Black</surname><given-names>LP</given-names></name><name><surname>Hopson</surname><given-names>C</given-names></name><etal/></person-group> <article-title>Early physician gestalt versus usual screening tools for the prediction of sepsis in critically ill emergency patients</article-title>. <source>Ann Emerg Med</source>. (<year>2024</year>) <volume>84</volume>(<issue>3</issue>):<fpage>246</fpage>&#x2013;<lpage>58</lpage>. <pub-id pub-id-type="doi">10.1016/j.annemergmed.2024.02.009</pub-id><pub-id pub-id-type="pmid">38530675</pub-id></citation></ref>
<ref id="B25"><label>25.</label><citation citation-type="book"><collab>Google LLC</collab>. <source>Gemini</source>. <comment>Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://gemini.google.com/app">https://gemini.google.com/app</ext-link> <comment>(Accessed August 31, 2024)</comment>.</citation></ref>
<ref id="B26"><label>26.</label><citation citation-type="book"><collab>Google LLC</collab>. <source>Google AI Studio</source>. <comment>Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://aistudio.google.com/">https://aistudio.google.com/</ext-link> <comment>(Accessed August 31, 2024)</comment>.</citation></ref>
<ref id="B27"><label>27.</label><citation citation-type="book"><collab>Meta Platforms, Inc</collab>. <source>LLaMA 3 (8B) on Ollama</source>. (<year>2024</year>). <comment>Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://ollama.com/library/llama3:8b">https://ollama.com/library/llama3:8b</ext-link> <comment>(Accessed August 31, 2024)</comment>.</citation></ref></ref-list>
</back>
</article>