<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article article-type="brief-report" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Digit. Health</journal-id>
<journal-title-group>
<journal-title>Frontiers in Digital Health</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Digit. Health</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2673-253X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fdgth.2025.1633278</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Brief Research Report</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Artificial intelligence in thoracic surgery consultations: evaluating the concordance between a large language model and expert clinical decisions</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>D&#x00E9;niz</surname><given-names>Carlos</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="corresp" rid="cor1">&#x002A;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3076013/overview"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role></contrib>
<contrib contrib-type="author">
<name><surname>Marc&#x00E8;</surname><given-names>Judith</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role></contrib>
<contrib contrib-type="author">
<name><surname>Macia</surname><given-names>Iv&#x00E1;n</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Rivas</surname><given-names>Francisco</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role></contrib>
<contrib contrib-type="author">
<name><surname>Mu&#x00F1;oz</surname><given-names>Anna</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role></contrib>
<contrib contrib-type="author">
<name><surname>Paradela</surname><given-names>Marina</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Garc&#x00ED;a</surname><given-names>Samuel</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role></contrib>
<contrib contrib-type="author">
<name><surname>Moreno</surname><given-names>Camilo</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role></contrib>
<contrib contrib-type="author">
<name><surname>Serratosa</surname><given-names>Ines</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role></contrib>
<contrib contrib-type="author">
<name><surname>Garc&#x00ED;a</surname><given-names>Marta</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role></contrib>
<contrib contrib-type="author">
<name><surname>Rodr&#x00ED;guez-Martos</surname><given-names>Tania</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Ojanguren</surname><given-names>Amaia</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>Department of Thoracic Surgery, Hospital Universitari de Bellvitge, L&#x2019;Hospitalet de Llobregat</institution>, <city>Barcelona</city>, <country country="es">Spain</country></aff>
<aff id="aff2"><label>2</label><institution>Bellvitge Institute for Biomedical Research, L&#x2019;Hospitalet de Llobregat</institution>, <city>Barcelona</city>, <country country="es">Spain</country></aff>
<aff id="aff3"><label>3</label><institution>Department of Medical</institution> <institution>Oncology, Catalan Institute of Oncology, L&#x2019;Hospitalet de Llobregat</institution>, <city>Barcelona</city>, <country country="es">Spain</country></aff>
<aff id="aff4"><label>4</label><institution>Universitat de Barcelona (UB&#x2014;Barcelona University)</institution>, <city>Barcelona</city>, <country country="es">Spain</country></aff>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label><bold>Correspondence:</bold> Carlos D&#x00E9;niz <email xlink:href="mailto:cdenizar@gmail.com">cdenizar@gmail.com</email></corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-11-17"><day>17</day><month>11</month><year>2025</year></pub-date>
<pub-date publication-format="electronic" date-type="collection"><year>2025</year></pub-date>
<volume>7</volume><elocation-id>1633278</elocation-id>
<history>
<date date-type="received"><day>02</day><month>06</month><year>2025</year></date>
<date date-type="accepted"><day>21</day><month>10</month><year>2025</year></date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2025 D&#x00E9;niz, Marc&#x00E8;, Macia, Rivas, Mu&#x00F1;oz, Paradela, Garc&#x00ED;a, Moreno, Serratosa, Garc&#x00ED;a, Rodr&#x00ED;guez-Martos and Ojanguren.</copyright-statement>
<copyright-year>2025</copyright-year><copyright-holder>D&#x00E9;niz, Marc&#x00E8;, Macia, Rivas, Mu&#x00F1;oz, Paradela, Garc&#x00ED;a, Moreno, Serratosa, Garc&#x00ED;a, Rodr&#x00ED;guez-Martos and Ojanguren</copyright-holder><license><ali:license_ref start_date="2025-11-17">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p></license>
</permissions>
<abstract><sec><title>Background</title>
<p>Artificial intelligence (AI) and large language models (LLMs) are increasingly used in clinical workflows, but their real-world application in thoracic surgery decision-making remains underexplored.</p>
</sec><sec><title>Methods</title>
<p>This retrospective observational study assessed the concordance between diagnostic and therapeutic recommendations generated by Scholar GPT (based on GPT-4) and decisions made by board-certified thoracic surgeons. All outpatient consultations over one week in a tertiary care hospital were included. Each case was evaluated using a 6-point concordance scale (0&#x2013;5), developed to quantify agreement in diagnosis and treatment planning. This was a retrospective observational, single-centre analysis; two independent thoracic surgeons assigned the concordance score. We report descriptive statistics and used <italic>t</italic>-tests/ANOVA for continuous variables and chi-square tests for categorical variables. Given the exploratory design, no <italic>a priori</italic> sample-size calculation or power analysis was performed.</p>
</sec><sec><title>Results</title>
<p>A total of 81 consultations were analysed. The mean concordance score was 3.67&#x2009;&#x00B1;&#x2009;1.17. High concordance (scores 4&#x2013;5) occurred in 56.8&#x0025; of cases, particularly in oncological diagnoses such as mediastinal and pleural tumours. Lower concordance was observed in complex or functional conditions like metastatic lung disease and thoracic outlet syndrome. No significant differences were found between consultation modalities or visit types.</p>
</sec><sec><title>Conclusion</title>
<p>Scholar GPT demonstrated promising alignment with surgeon decisions in structured oncologic cases but showed variability in complex scenarios. While AI may assist in streamlining outpatient workflows, its use should remain complementary to expert clinical judgment. These findings are exploratory and should be interpreted with caution given the small sample size and single-centre, one-week design.</p>
</sec>
</abstract>
<kwd-group>
<kwd>artificial intelligence</kwd>
<kwd>large language models</kwd>
<kwd>thoracic surgery</kwd>
<kwd>clinical decision support</kwd>
<kwd>oncology</kwd>
<kwd>AI in medicine</kwd>
</kwd-group><funding-group>
<funding-statement>The author(s) declare that financial support was received for the research and/or publication of this article. The article processing charge (APC) was partially funded by the University of Barcelona Institutional Open Access Program.</funding-statement>
</funding-group>
<counts>
<fig-count count="2"/>
<table-count count="0"/><equation-count count="0"/><ref-count count="28"/><page-count count="7"/><word-count count="3870"/></counts><custom-meta-group><custom-meta><meta-name>section-at-acceptance</meta-name><meta-value>Health Informatics</meta-value></custom-meta></custom-meta-group>
</article-meta>
</front>
<body><sec id="s1" sec-type="intro"><title>Introduction</title>
<p>Thoracic surgery involves intricate clinical decision-making that requires the integration of patient history, imaging studies, and adherence to evolving clinical guidelines. The increasing complexity of medical data has driven the exploration of artificial intelligence (AI) and large language models (LLMs) as potential decision-support tools to assist clinicians in managing complex cases with greater efficiency and consistency (<xref ref-type="bibr" rid="B1">1</xref>&#x2013;<xref ref-type="bibr" rid="B3">3</xref>). Among these technologies, LLMs such as ChatGPT/Scholar GPT, based on GPT-4-class architectures, have shown strong performance on knowledge-based tasks and standardized examinations, suggesting an emerging role in supporting clinical reasoning (<xref ref-type="bibr" rid="B4">4</xref>).</p>
<p>Although early applications of AI in thoracic surgery have focused on imaging and tumor classification (<xref ref-type="bibr" rid="B5">5</xref>), there is a growing body of work examining LLMs beyond image-centric tasks in ambulatory clinical decision support&#x2014;covering diagnostic reasoning, triage, and guideline-concordant recommendations (<xref ref-type="bibr" rid="B6">6</xref>&#x2013;<xref ref-type="bibr" rid="B8">8</xref>). However, the application of LLMs in real-world surgical environments&#x2014;especially in thoracic surgery&#x2014;remains underexplored, where decisions depend on nuanced presentations, comorbidities, and patient preferences.</p>
<p>Moreover, integrating AI into clinical workflows raises critical questions about its concordance with human judgment and its reliability across heterogeneous scenarios. Prior reports highlight both promising alignment and meaningful discrepancies, underscoring the need for careful validation in contexts where multidisciplinary input remains essential (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B10">10</xref>). In parallel, potential risks&#x2014;including hallucinations, misplaced confidence, bias propagation, and over-reliance&#x2014;carry direct safety implications in low-concordance cases; the &#x201C;black-box&#x201D; nature of LLMs further complicates accountability and clinician trust, reinforcing the importance of explainability and human-in-the-loop oversight in high-stakes settings (<xref ref-type="bibr" rid="B11">11</xref>&#x2013;<xref ref-type="bibr" rid="B13">13</xref>).</p>
<p>Accordingly, we conducted a single-centre, exploratory evaluation of concordance between Scholar GPT and board-certified thoracic surgeons in routine outpatient consultations, aiming to identify clinical areas of higher and lower agreement and to delineate pragmatic considerations for safe integration.</p>
</sec>
<sec id="s2" sec-type="methods"><title>Materials and methods</title>
<sec id="s2a"><title>Study design and population</title>
<p>This retrospective observational study assessed the concordance between AI-generated recommendations from Scholar GPT and clinical decisions made by board-certified thoracic surgeons in a high-volume tertiary university hospital. All outpatient thoracic surgery consultations conducted over one week were included, without case selection or exclusion criteria, ensuring a representative sample of routine clinical practice. Visits were performed by certified thoracic surgeons; Six surgeons participated during the study week.</p>
<p>Data collected included patient demographics (age, sex), consultation type (first visit or follow-up), consultation modality (in-person or telemedicine), and the final clinical diagnosis established by the surgeon. The full distribution of first vs. follow-up and in-person vs. telemedicine is reported in Results.</p>
</sec>
<sec id="s2b"><title>Artificial intelligence model</title>
<p>Scholar GPT, (OpenAI, version accessed January 2025) a large language model based on GPT-4 architecture, was selected due to its ability to access and synthesize information from biomedical databases, including PubMed and evidence-based clinical guidelines. The model was accessed via a dedicated medical consultation interface during <italic>January</italic> 2025, with output traceability. The system had no connectivity to the hospital EHR and no local fine-tuning with institutional data.</p>
</sec>
<sec id="s2c"><title>Procedure and data collection</title>
<p>For each consultation, the attending surgeon documented the primary reason for the visit, relevant clinical history, imaging results, and the diagnostic and therapeutic plan. The same clinical scenario was entered into Scholar GPT, prompting the AI to provide a diagnostic assessment and treatment recommendation. AI responses were recorded verbatim without modifications. The AI received a structured <italic>text</italic> summary (history, comorbidities/medication, and key findings transcribed from imaging reports); no image files were uploaded, and no external oncology notes were provided. Physical examination findings, when available, were included as text.</p>
<p>Additionally, the number of clarification questions generated by the AI before formulating a recommendation was documented to evaluate the model&#x0027;s reasoning process. Across the 81 consultations, the mean number of clarification turns was <italic>1.02</italic>. Two independent thoracic surgeons reviewed the AI output and assigned a concordance score according to the predefined scale.</p>
</sec>
<sec id="s2d"><title>Concordance scale and evaluation</title>
<p>A specific 6-point ordinal concordance scale (0&#x2013;5) was developed to assess the level of agreement between AI-generated recommendations and the surgeons&#x0027; decisions, focusing on both diagnosis and treatment alignment:
<list list-type="bullet">
<list-item>
<p>0: No diagnosis or treatment proposed.</p></list-item>
<list-item>
<p>1: Correct diagnosis, no treatment recommendation.</p></list-item>
<list-item>
<p>2: Correct diagnosis with limited treatment alignment (&#x2264;25&#x0025;).</p></list-item>
<list-item>
<p>3: Correct diagnosis with partial treatment alignment (&#x2248;50&#x0025;).</p></list-item>
<list-item>
<p>4: Correct diagnosis with near-complete treatment alignment (&#x2248;75&#x0025;).</p></list-item>
<list-item>
<p>5: Complete agreement in both diagnosis and treatment.</p></list-item>
</list>This scale allowed a detailed evaluation of the AI&#x0027;s clinical performance, distinguishing between correct diagnostic proposals and the adequacy of therapeutic suggestions.</p>
</sec>
<sec id="s2e"><title>Statistical analysis</title>
<p>Descriptive statistics were used to summarize the distribution of concordance scores between Scholar GPT&#x0027;s recommendations and the surgeons&#x0027; decisions. Normality of continuous variables was assessed using the Shapiro&#x2013;Wilk test. As the concordance scores did not follow a normal distribution (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.001), non-parametric tests were employed. Comparisons between groups were performed using the Mann&#x2013;Whitney <italic>U</italic>-test for two-group comparisons, while categorical variables were assessed with chi-square tests. Additionally, subgroup analyses were performed to compare concordance scores based on diagnosis type, consultation modality (in-person vs. telemedicine), and visit type (first-time vs. follow-up). Statistical significance was set at <italic>p</italic>&#x2009;&#x003C;&#x2009;0.05. Given the exploratory scope and one-week sampling, no <italic>a priori</italic> sample-size/power calculation was undertaken. Formal normality testing and inter-rater reliability metrics (e.g., ICC) were not computed; these are acknowledged as limitations and targets for future work.</p>
</sec>
<sec id="s2f"><title>Ethical considerations</title>
<p>This study was retrospective and based on anonymized clinical data. All patients had previously signed a general institutional consent authorizing the use of their anonymized data for research purposes. According to institutional policy, no additional ethical approval was required. No patient-identifiable information was entered into the AI system.</p>
</sec>
</sec>
<sec id="s3" sec-type="results"><title>Results</title>
<p>A total of 81 thoracic surgery outpatient consultations were analysed to assess the concordance between Scholar GPT&#x0027;s recommendations and the clinical decisions made by thoracic surgeons. The mean concordance score on the 0&#x2013;5 scale was 3.67&#x2009;&#x00B1;&#x2009;1.17, with individual scores ranging from 0 (no diagnosis or treatment proposed) to 5 (complete agreement in diagnosis and treatment). The distribution of concordance scores is shown in <xref ref-type="fig" rid="F1">Figure&#x00A0;1</xref>.</p>
<fig id="F1" position="float"><label>Figure&#x00A0;1</label>
<caption><p>Distribution of concordance scores between Scholar GPT and thoracic surgeons across the 81 outpatient consultations.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fdgth-07-1633278-g001.tif"><alt-text content-type="machine-generated">Bar chart depicting the number of outpatient consultations related to ScholarGPT accuracy scores from zero to five. Scores three, four, and five show higher consultation numbers, peaking at score three with nearly thirty consultations.</alt-text>
</graphic>
</fig>
<p>High concordance (scores 4&#x2013;5) was observed in 56.8&#x0025; of cases (<italic>n</italic>&#x2009;&#x003D;&#x2009;46), indicating substantial agreement between the AI-generated recommendations and the clinical decisions. Moderate concordance (score 3) occurred in 34.6&#x0025; of consultations (<italic>n</italic>&#x2009;&#x003D;&#x2009;28), while low concordance (scores 0&#x2013;2) was found in 8.6&#x0025; of cases (<italic>n</italic>&#x2009;&#x003D;&#x2009;7). First visits accounted for 8/81 (9.9&#x0025;) and follow-up visits for 73/81 (90.1&#x0025;); in-person consultations were 39/81 (48.1&#x0025;) and telemedicine 42/81 (51.9&#x0025;). No statistically significant differences in concordance were observed across visit type (Mann&#x2013;Whitney <italic>U</italic>&#x2009;&#x003D;&#x2009;292.5, <italic>p</italic>&#x2009;&#x003D;&#x2009;1.000) or consultation modality (Mann&#x2013;Whitney <italic>U</italic>&#x2009;&#x003D;&#x2009;908.0, <italic>p</italic>&#x2009;&#x003D;&#x2009;0.381).</p>
<p>The AI model performed notably well in oncology-related cases, achieving perfect alignment in diagnoses such as pectus excavatum/carinatum (<italic>n</italic>&#x2009;&#x003D;&#x2009;2; mean score 5.0), mediastinal neoplasms (<italic>n</italic>&#x2009;&#x003D;&#x2009;1; mean 5.0), and pleural neoplasms (<italic>n</italic>&#x2009;&#x003D;&#x2009;1; mean 5.0). In contrast, lower concordance was identified in conditions requiring nuanced clinical judgment or complex decision-making, including: thymoma (<italic>n</italic>&#x2009;&#x003D;&#x2009;3; mean 2.0), pulmonary metastases (<italic>n</italic>&#x2009;&#x003D;&#x2009;3; mean 2.3), and thoracic outlet syndrome (<italic>n</italic>&#x2009;&#x003D;&#x2009;4; mean 3.0). Among primary lung cancer presentations (<italic>n</italic>&#x2009;&#x003D;&#x2009;42), the mean concordance was 3.83. The variability in accuracy scores according to diagnosis is represented in <xref ref-type="fig" rid="F2">Figure&#x00A0;2</xref>. Given the small numerators in some subgroups, these estimates should be interpreted cautiously.</p>
<fig id="F2" position="float"><label>Figure&#x00A0;2</label>
<caption><p>Accuracy scores of Scholar GPT by diagnosis. The plot illustrates higher concordance in oncological cases and greater variability in functional or complex diagnoses.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fdgth-07-1633278-g002.tif"><alt-text content-type="machine-generated">Box plot showing accuracy scores by ChatGPT for various diagnoses. The y-axis represents accuracy ranging from zero to five. Diagnoses include malignant pleural effusion, lung cancer, and more, listed on the x-axis. Most boxes reveal scores between three and five, with variations in whisker length indicating different accuracy ranges across diagnoses.</alt-text>
</graphic>
</fig>
<p>Additionally, Scholar GPT demonstrated high efficiency, requiring a mean of 1.02 clarification questions to reach its diagnostic and therapeutic recommendations, with most cases resolved after a single prompt. Typical clarification themes included: (i) staging details (e.g., PET-CT/SUV and nodal stations), (ii) pulmonary function metrics (FEV1, DLCO, and predicted postoperative values), (iii) perioperative risk modifiers (anticoagulation/antiplatelet therapy, frailty/performance status), and (iv) occasional requests for low-yield tests (e.g., tumor markers) in otherwise straightforward scenarios. For instance, in a lung cancer case, the AI asked: &#x201C;<italic>What are the specific results of the PET-CT, including SUVmax values and any evidence of mediastinal or distant metastasis?</italic>&#x201D;.</p>
<p>Safety-relevant review did not identify AI recommendations that would be clearly harmful if executed verbatim; however, in low-concordance scenarios some suggestions could plausibly lead to delayed work-up or suboptimal management (e.g., unnecessary testing or premature therapy). These observations reinforce the need for clinician oversight and multidisciplinary team (MDT) review when AI advice diverges from standard pathways.</p>
</sec>
<sec id="s4" sec-type="discussion"><title>Discussion</title>
<p>The integration of large language models (LLMs) into clinical workflows represents a significant advancement in decision support, especially in resource-constrained outpatient settings. Our study explored the use of Scholar GPT as a supportive tool in thoracic surgery consultations and revealed moderate to high concordance between its recommendations and those made by expert clinicians, particularly in oncological cases. These results suggest that LLMs may offer valuable assistance when standard guideline-based decisions are required. This aligns with prior evidence that LLMs perform strongly on knowledge-based tasks and clinical reasoning benchmarks (<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B6">6</xref>) and with observations that foundation models tend to excel on well-specified, structured problems (<xref ref-type="bibr" rid="B7">7</xref>).</p>
<p>Recent literature demonstrates the expanding role of AI tools, including LLMs like ChatGPT and Med-PaLM, in improving diagnostic accuracy, triage, and treatment selection across various specialties (<xref ref-type="bibr" rid="B5">5</xref>, <xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B10">10</xref>). For instance, AI has shown utility in radiology, pathology, and oncology by offering second-opinion diagnoses and reducing interobserver variability (<xref ref-type="bibr" rid="B14">14</xref>). In their study on foundation models for generalist medical AI, Moor et al. reported that such models demonstrate strong performance in standard tasks but often underperform in more nuanced clinical decision-making scenarios (<xref ref-type="bibr" rid="B15">15</xref>). However, in surgical fields such as thoracic surgery, real-world validation of LLMs remains scarce.</p>
<p>Our findings align with prior works reporting strong AI performance in well-defined oncological scenarios, where diagnostic and treatment pathways are often governed by standardized guidelines (<xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B17">17</xref>). Conversely, lower agreement was observed in functional conditions, metastatic disease, or rare syndromes, echoing previous concerns about AI limitations in complex or ambiguous clinical contexts. Rajpurkar et al. emphasized that although LLMs show high accuracy on benchmark datasets, their reliability in clinical deployment still demands rigorous validation and attention to prompt sensitivity (<xref ref-type="bibr" rid="B2">2</xref>).</p>
<sec id="s4a"><title>Clinical safety implications and the black box problem</title>
<p>A critical aspect that requires thorough discussion, as highlighted by the reviewers, is the clinical safety implications of AI decisions, particularly the potential risks in low-concordance cases. While our safety-relevant review did not identify AI recommendations that would be clearly harmful if executed verbatim, the potential for suboptimal management in low-concordance scenarios represents a significant concern that extends beyond our immediate findings.</p>
<p>Recent comprehensive evaluations have demonstrated that current LLMs are not ready for autonomous clinical decision-making (<xref ref-type="bibr" rid="B18">18</xref>). Hager et al. found that LLMs tend to make hasty decisions without consistently following diagnostic guidelines, often omitting essential physical examinations and misinterpreting basic laboratory results&#x2014;fundamental errors that pose serious risks to patient safety without extensive clinician supervision (<xref ref-type="bibr" rid="B18">18</xref>). This finding is particularly relevant to our study, as it suggests that even in cases where we observed moderate concordance, the underlying reasoning process may be fundamentally flawed.</p>
<p>The &#x201C;black box&#x201D; nature of LLMs further complicates their integration into surgical decision-making contexts. Unlike traditional clinical decision support tools where the reasoning pathway can be traced and validated, LLMs operate through complex neural networks that make their decision-making process opaque and uninterpretable (<xref ref-type="bibr" rid="B19">19</xref>). This lack of transparency poses significant challenges for surgical practice, where understanding the rationale behind a recommendation is crucial for patient safety and medicolegal accountability. As Xu et al. argue, the unexplainability feature of medical AI systems may cause harm that is currently underestimated, particularly when these systems make incomprehensible mistakes that are difficult to detect (<xref ref-type="bibr" rid="B19">19</xref>).</p>
<p>The implications of this opacity are particularly concerning in thoracic surgery, where decisions often involve high-stakes interventions with significant morbidity and mortality risks. When an AI system recommends a specific surgical approach or suggests delaying intervention, surgeons need to understand the underlying reasoning to make informed decisions about patient care. The inability to interrogate the AI&#x0027;s decision-making process undermines the fundamental principle of evidence-based medicine and may lead to either inappropriate reliance on AI recommendations or complete rejection of potentially valuable insights.</p>
</sec>
<sec id="s4b"><title>Strengths and limitations</title>
<p>This study offers a real-life evaluation of LLM-assisted consultation in a high-volume outpatient service, contributing to a growing body of work emphasizing AI&#x0027;s role in enhancing healthcare efficiency (<xref ref-type="bibr" rid="B20">20</xref>). However, important limitations must be noted, and their impact on result interpretation requires careful analysis.</p>
<p>First, LLMs operate without direct access to physical examination findings or dynamic patient interaction, factors essential for nuanced decision-making in surgical practice. This limitation is particularly significant in thoracic surgery, where physical examination findings such as respiratory mechanics, chest wall deformities, and lymph node palpation often provide crucial diagnostic information that cannot be captured in text-based summaries.</p>
<p>Second, the lack of interpretability in LLM outputs&#x2014;commonly referred to as the &#x201C;black box&#x201D; problem&#x2014;poses a barrier to clinician trust and accountability (<xref ref-type="bibr" rid="B21">21</xref>). The absence of baseline intra-group concordance assessment (AI reproducibility and inter-surgeon agreement) as control measures, as noted by the reviewers, represents a significant methodological limitation that prevents us from establishing the reliability of our concordance measurements.</p>
<p>Third, our study lacks formal normality testing before applying <italic>t</italic>-tests and the absence of inter-rater correlation coefficient analysis. These methodological deficiencies impact the interpretation of our results and should be addressed in future studies.</p>
<p>Furthermore, AI outputs may vary based on prompt phrasing, regional practices, and training data, as demonstrated in comparative analyses between generalist LLMs and domain-specific models (<xref ref-type="bibr" rid="B22">22</xref>, <xref ref-type="bibr" rid="B23">23</xref>). Although Scholar GPT achieved high scores in our study, its performance cannot be generalized without further multi-centre validation. The small sample size of 81 cases is insufficient for multiple subgroup analyses, resulting in inadequate statistical power for many of our diagnostic-specific comparisons.</p>
<p>A major limitation of this study is the absence of inter-rater reliability assessment. While two independent thoracic surgeons assigned concordance scores, only the final consensus scores were recorded, preventing calculation of inter-rater agreement statistics such as Cohen&#x0027;s kappa or ICC. This represents a significant methodological deficiency that limits confidence in the reliability of our primary outcome measure.</p>
</sec>
<sec id="s4c"><title>Applications in thoracic surgery</title>
<p>Despite these limitations, the potential applications of AI in thoracic surgery continue to expand. Recent comprehensive reviews have highlighted AI&#x0027;s growing role in enhancing diagnostic accuracy, surgical precision, and postoperative care in thoracic surgery (<xref ref-type="bibr" rid="B24">24</xref>). AI applications now span the entire perioperative period, from preoperative risk assessment and surgical planning to intraoperative guidance and postoperative monitoring (<xref ref-type="bibr" rid="B5">5</xref>).</p>
<p>In the preoperative phase, AI has demonstrated promise in imaging analysis, tumor classification, and surgical candidacy assessment. During surgery, AI-assisted augmented reality systems have shown feasibility in robotic lung surgery, achieving high accuracy in gesture recognition and potentially improving surgical precision (<xref ref-type="bibr" rid="B25">25</xref>). Postoperatively, AI assists in pathology assessment, complication prediction, and long-term outcome modelling.</p>
<p>However, as noted in recent scoping reviews, significant challenges remain, including the need for larger, more diverse datasets, standardization of AI applications across institutions, and development of robust validation frameworks specific to thoracic surgery (<xref ref-type="bibr" rid="B26">26</xref>).</p>
</sec>
<sec id="s4d"><title>Implications and future directions</title>
<p>LLMs may serve as cognitive aids, particularly in settings with limited subspecialty coverage, such as remote clinics or after-hours consultations. Yet, AI should complement&#x2014;not replace&#x2014;clinical judgment, especially in patient-specific scenarios involving comorbidities or psychosocial factors (<xref ref-type="bibr" rid="B27">27</xref>).</p>
<p>Future efforts should focus on fine-tuning LLMs using thoracic surgery-specific datasets, integrating multimodal inputs (e.g., imaging, labs), and enabling real-time feedback loops from clinical users (<xref ref-type="bibr" rid="B11">11</xref>). Ethical frameworks and regulatory guidelines will also be essential to ensure responsible AI deployment in patient care (<xref ref-type="bibr" rid="B28">28</xref>). Most importantly, future studies must address the methodological limitations identified in our work, including larger sample sizes, multi-centre designs, formal inter-rater reliability assessments, and comprehensive safety evaluations.</p>
<p>The development of explainable AI systems that can provide transparent reasoning for their recommendations will be crucial for gaining clinician trust and ensuring safe integration into surgical practice. Until these challenges are addressed, LLMs should be used with extreme caution in clinical decision-making, with mandatory human oversight and validation of all AI-generated recommendations.</p>
</sec>
</sec>
<sec id="s5" sec-type="conclusions"><title>Conclusion</title>
<p>Large language models, such as Scholar GPT, demonstrate high concordance with thoracic surgeons in outpatient decision-making, particularly in oncological cases. However, its performance was more variable in complex or functional diagnoses, highlighting the indispensable role of human expertise in nuanced clinical scenarios. While AI-based decision-support systems have the potential to enhance clinical efficiency, their integration should focus on augmenting, rather than replacing, expert decision-making. Future advancements should prioritize real-time clinical feedback and multimodal patient data integration to optimize the reliability and applicability of AI-assisted decision-making in thoracic surgery.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability"><title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s7" sec-type="ethics-statement"><title>Ethics statement</title>
<p>This study was retrospective and based on anonymized clinical data. All patients had previously signed a general institutional consent authorizing the use of their anonymized data for research purposes. According to institutional policy, no additional ethical approval was required.</p>
</sec>
<sec id="s8" sec-type="author-contributions"><title>Author contributions</title>
<p>CD: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. JM: Investigation, Resources, Writing &#x2013; review &#x0026; editing. IM: Data curation, Methodology, Validation, Writing &#x2013; review &#x0026; editing. FR: Conceptualization, Data curation, Writing &#x2013; review &#x0026; editing. AM: Investigation, Supervision, Writing &#x2013; review &#x0026; editing. MP: Formal analysis, Resources, Validation, Writing &#x2013; review &#x0026; editing. SG: Methodology, Supervision, Writing &#x2013; review &#x0026; editing. CM: Conceptualization, Investigation, Writing &#x2013; review &#x0026; editing. IS: Investigation, Methodology, Writing &#x2013; review &#x0026; editing. MG: Data curation, Methodology, Writing &#x2013; review &#x0026; editing. TR-M: Investigation, Writing &#x2013; review &#x0026; editing. AO: Methodology, Resources, Supervision, Writing &#x2013; review &#x0026; editing.</p>
</sec>
<sec id="s10" sec-type="COI-statement"><title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="ai-statement"><title>Generative AI statement</title>
<p>The author(s) declare that Generative AI was used in the creation of this manuscript. The author(s) verify and take full responsibility for the use of generative AI in the preparation of this manuscript. Generative AI was used solely to assist with improving the English language and grammar in the writing process, as none of the authors are native English speakers. No content, analysis, or original ideas were generated by AI, and all scientific interpretations and conclusions are the sole responsibility of the authors.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec id="s12" sec-type="disclaimer"><title>Publisher&#x0027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list><title>References</title>
<ref id="B1"><label>1.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Topol</surname> <given-names>EJ</given-names></name></person-group>. <article-title>High-performance medicine: the convergence of human and artificial intelligence</article-title>. <source>Nat Med</source>. (<year>2019</year>) <volume>25</volume>:<fpage>44</fpage>&#x2013;<lpage>56</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-018-0300-7</pub-id><pub-id pub-id-type="pmid">30617339</pub-id></mixed-citation></ref>
<ref id="B2"><label>2.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rajpurkar</surname> <given-names>P</given-names></name> <name><surname>Chen</surname> <given-names>E</given-names></name> <name><surname>Banerjee</surname> <given-names>O</given-names></name> <name><surname>Topol</surname> <given-names>EJ</given-names></name></person-group>. <article-title>AI In health and medicine</article-title>. <source>Nat Med</source>. (<year>2022</year>) <volume>28</volume>:<fpage>31</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-021-01614-0</pub-id><pub-id pub-id-type="pmid">35058619</pub-id></mixed-citation></ref>
<ref id="B3"><label>3.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>KH</given-names></name> <name><surname>Beam</surname> <given-names>AL</given-names></name> <name><surname>Kohane</surname> <given-names>IS</given-names></name></person-group>. <article-title>Artificial intelligence in healthcare</article-title>. <source>Nat Biomed Eng</source>. (<year>2018</year>) <volume>2</volume>:<fpage>719</fpage>&#x2013;<lpage>31</lpage>. <pub-id pub-id-type="doi">10.1038/s41551-018-0305-z</pub-id><pub-id pub-id-type="pmid">31015651</pub-id></mixed-citation></ref>
<ref id="B4"><label>4.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Singhal</surname> <given-names>K</given-names></name> <name><surname>Azizi</surname> <given-names>S</given-names></name> <name><surname>Tu</surname> <given-names>T</given-names></name> <name><surname>Mahdavi</surname> <given-names>SS</given-names></name> <name><surname>Wei</surname> <given-names>J</given-names></name> <name><surname>Chung</surname> <given-names>HW</given-names></name><etal/></person-group> <article-title>Large language models encode clinical knowledge</article-title>. <source>Nature</source>. (<year>2023</year>) <volume>620</volume>:<fpage>172</fpage>&#x2013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id><pub-id pub-id-type="pmid">37438534</pub-id></mixed-citation></ref>
<ref id="B5"><label>5.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bellini</surname> <given-names>V</given-names></name> <name><surname>Pinotti</surname> <given-names>E</given-names></name> <name><surname>Sozzi</surname> <given-names>M</given-names></name> <name><surname>Bignami</surname> <given-names>E</given-names></name></person-group>. <article-title>Artificial intelligence in thoracic surgery: a narrative review</article-title>. <source>J Thorac Dis</source>. (<year>2021</year>) <volume>13</volume>:<fpage>6487</fpage>&#x2013;<lpage>502</lpage>. <pub-id pub-id-type="doi">10.21037/jtd-21-761</pub-id></mixed-citation></ref>
<ref id="B6"><label>6.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>P</given-names></name> <name><surname>Bubeck</surname> <given-names>S</given-names></name> <name><surname>Petro</surname> <given-names>J</given-names></name></person-group>. <article-title>Benefits, limits, and risks of GPT-4 as an AI chatbot for medicine</article-title>. <source>N Engl J Med</source>. (<year>2023</year>) <volume>388</volume>:<fpage>1233</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMsr2214184</pub-id><pub-id pub-id-type="pmid">36988602</pub-id></mixed-citation></ref>
<ref id="B7"><label>7.</label><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Bommasani</surname> <given-names>R</given-names></name> <name><surname>Hudson</surname> <given-names>DA</given-names></name> <name><surname>Adeli</surname> <given-names>E</given-names></name> <name><surname>Altman</surname> <given-names>R</given-names></name> <name><surname>Arora</surname> <given-names>S</given-names></name> <name><surname>von Arx</surname> <given-names>S</given-names></name><etal/></person-group> <comment>On the opportunities and risks of foundation models</comment>. <comment>arXiv preprint arXiv:2108.07258</comment>. (<year>2021</year>).</mixed-citation></ref>
<ref id="B8"><label>8.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Thirunavukarasu</surname> <given-names>AJ</given-names></name> <name><surname>Ting</surname> <given-names>DSJ</given-names></name> <name><surname>Elangovan</surname> <given-names>K</given-names></name> <name><surname>Gutierrez</surname> <given-names>L</given-names></name> <name><surname>Tan</surname> <given-names>TF</given-names></name> <name><surname>Ting</surname> <given-names>DSW</given-names></name></person-group>. <article-title>Large language models in medicine</article-title>. <source>Nat Med</source>. (<year>2023</year>) <volume>29</volume>:<fpage>1930</fpage>&#x2013;<lpage>40</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id><pub-id pub-id-type="pmid">37460753</pub-id></mixed-citation></ref>
<ref id="B9"><label>9.</label><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Nori</surname> <given-names>H</given-names></name> <name><surname>King</surname> <given-names>N</given-names></name> <name><surname>McKinney</surname> <given-names>SM</given-names></name> <name><surname>Carignan</surname> <given-names>D</given-names></name> <name><surname>Horvitz</surname> <given-names>E</given-names></name></person-group>. <comment>Capabilities of GPT-4 on medical challenge problems</comment>. <comment>arXiv preprint arXiv:2303.13375</comment>. (<year>2023</year>).</mixed-citation></ref>
<ref id="B10"><label>10.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tu</surname> <given-names>T</given-names></name> <name><surname>Azizi</surname> <given-names>S</given-names></name> <name><surname>Driess</surname> <given-names>D</given-names></name> <name><surname>Schaekermann</surname> <given-names>M</given-names></name> <name><surname>Amin</surname> <given-names>M</given-names></name> <name><surname>Chang</surname> <given-names>PC</given-names></name><etal/></person-group> <article-title>Towards generalist biomedical AI</article-title>. <source>N Engl J Med</source>. (<year>2023</year>) <volume>389</volume>:<fpage>1180</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMra2204778</pub-id><pub-id pub-id-type="pmid">37754283</pub-id></mixed-citation></ref>
<ref id="B11"><label>11.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Amann</surname> <given-names>J</given-names></name> <name><surname>Blasimme</surname> <given-names>A</given-names></name> <name><surname>Vayena</surname> <given-names>E</given-names></name> <name><surname>Frey</surname> <given-names>D</given-names></name> <name><surname>Madai</surname> <given-names>VI</given-names></name></person-group>. <article-title>Explainability for artificial intelligence in healthcare: a multidisciplinary perspective</article-title>. <source>BMC Med Inform Decis Mak</source>. (<year>2020</year>) <volume>20</volume>:<fpage>310</fpage>. <pub-id pub-id-type="doi">10.1186/s12911-020-01332-6</pub-id><pub-id pub-id-type="pmid">33256715</pub-id></mixed-citation></ref>
<ref id="B12"><label>12.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rudin</surname> <given-names>C</given-names></name></person-group>. <article-title>Stop explaining black box machine learning models for high stakes decisions and use interpretable models instead</article-title>. <source>Nat Mach Intell</source>. (<year>2019</year>) <volume>1</volume>:<fpage>206</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1038/s42256-019-0048-x</pub-id><pub-id pub-id-type="pmid">35603010</pub-id></mixed-citation></ref>
<ref id="B13"><label>13.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ghassemi</surname> <given-names>M</given-names></name> <name><surname>Oakden-Rayner</surname> <given-names>L</given-names></name> <name><surname>Beam</surname> <given-names>AL</given-names></name></person-group>. <article-title>The false hope of current approaches to explainable artificial intelligence in health care</article-title>. <source>Lancet Digit Health</source>. (<year>2021</year>) <volume>3</volume>:<fpage>e745</fpage>&#x2013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1016/S2589-7500(21)00208-9</pub-id><pub-id pub-id-type="pmid">34711379</pub-id></mixed-citation></ref>
<ref id="B14"><label>14.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Br&#x00FC;gge</surname> <given-names>E</given-names></name> <name><surname>Ricchizzi</surname> <given-names>S</given-names></name> <name><surname>Arenbeck</surname> <given-names>M</given-names></name> <name><surname>Keller</surname> <given-names>MN</given-names></name> <name><surname>Dittrich</surname> <given-names>F</given-names></name> <name><surname>Seyfarth</surname> <given-names>S</given-names></name><etal/></person-group> <article-title>Large language models improve clinical decision making of medical students through patient simulation and structured feedback: a randomized controlled trial</article-title>. <source>BMC Med Educ</source>. (<year>2024</year>) <volume>24</volume>:<fpage>1399</fpage>. <pub-id pub-id-type="doi">10.1186/s12909-024-06399-7</pub-id></mixed-citation></ref>
<ref id="B15"><label>15.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Moor</surname> <given-names>M</given-names></name> <name><surname>Banerjee</surname> <given-names>O</given-names></name> <name><surname>Abad</surname> <given-names>ZSH</given-names></name> <name><surname>Krumholz</surname> <given-names>HM</given-names></name> <name><surname>Leskovec</surname> <given-names>J</given-names></name> <name><surname>Topol</surname> <given-names>EJ</given-names></name><etal/></person-group> <article-title>Foundation models for generalist medical artificial intelligence</article-title>. <source>Nature</source>. (<year>2023</year>) <volume>616</volume>:<fpage>259</fpage>&#x2013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-023-05881-4</pub-id><pub-id pub-id-type="pmid">37045921</pub-id></mixed-citation></ref>
<ref id="B16"><label>16.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gaber</surname> <given-names>F</given-names></name> <name><surname>Ahmed</surname> <given-names>A</given-names></name> <name><surname>Abdelrazek</surname> <given-names>M</given-names></name> <name><surname>Grundy</surname> <given-names>J</given-names></name> <name><surname>Susilo</surname> <given-names>W</given-names></name></person-group>. <article-title>Evaluating large language model workflows in clinical decision support</article-title>. <source>NPJ Digit Med</source>. (<year>2025</year>) <volume>8</volume>:<fpage>11</fpage>. <pub-id pub-id-type="doi">10.1038/s41746-025-01684-1</pub-id><pub-id pub-id-type="pmid">39762352</pub-id></mixed-citation></ref>
<ref id="B17"><label>17.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>J</given-names></name> <name><surname>Zhou</surname> <given-names>Z</given-names></name> <name><surname>Lyu</surname> <given-names>H</given-names></name> <name><surname>Wang</surname> <given-names>Z</given-names></name></person-group>. <article-title>Large language models-powered clinical decision support: enhancing or replacing human expertise?</article-title> <source>Intell Med</source>. (<year>2025</year>) <volume>5</volume>:<fpage>14</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1016/j.imed.2025.01.002</pub-id></mixed-citation></ref>
<ref id="B18"><label>18.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hager</surname> <given-names>P</given-names></name> <name><surname>Jungmann</surname> <given-names>F</given-names></name> <name><surname>Holland</surname> <given-names>R</given-names></name> <name><surname>Bhagat</surname> <given-names>K</given-names></name> <name><surname>Hubrecht</surname> <given-names>I</given-names></name> <name><surname>Knauer</surname> <given-names>M</given-names></name><etal/></person-group> <article-title>Evaluation and mitigation of the limitations of large language models in clinical decision-making</article-title>. <source>Nat Med</source>. (<year>2024</year>) <volume>30</volume>:<fpage>2613</fpage>&#x2013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-024-03097-1</pub-id><pub-id pub-id-type="pmid">38965432</pub-id></mixed-citation></ref>
<ref id="B19"><label>19.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>H</given-names></name> <name><surname>Shuttleworth</surname> <given-names>KMJ</given-names></name></person-group>. <article-title>Medical artificial intelligence and the black box problem: a view based on the ethical principle of &#x201C;do no harm&#x201D;</article-title>. <source>Intell Med</source>. (<year>2024</year>) <volume>4</volume>:<fpage>77</fpage>&#x2013;<lpage>85</lpage>. <pub-id pub-id-type="doi">10.1016/j.imed.2023.08.001</pub-id></mixed-citation></ref>
<ref id="B20"><label>20.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lammert</surname> <given-names>J</given-names></name> <name><surname>Ravi</surname> <given-names>N</given-names></name> <name><surname>Rajpurkar</surname> <given-names>P</given-names></name></person-group>. <article-title>Expert-guided large language models for clinical decision support</article-title>. <source>JAMA Netw Open</source>. (<year>2024</year>) <volume>7</volume>:<fpage>e2439297</fpage>. <pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.39297</pub-id></mixed-citation></ref>
<ref id="B21"><label>21.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Winkler</surname> <given-names>PW</given-names></name> <name><surname>Ravi</surname> <given-names>N</given-names></name> <name><surname>Rajpurkar</surname> <given-names>P</given-names></name></person-group>. <article-title>Risks, limitations, safety and verification of medical AI systems</article-title>. <source>Nat Med</source>. (<year>2025</year>) <volume>31</volume>:<fpage>45</fpage>&#x2013;<lpage>52</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-024-03298-8</pub-id><pub-id pub-id-type="pmid">39833407</pub-id></mixed-citation></ref>
<ref id="B22"><label>22.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Toal</surname> <given-names>M</given-names></name> <name><surname>Hill</surname> <given-names>C</given-names></name> <name><surname>Quinn</surname> <given-names>M</given-names></name> <name><surname>O&#x0027;Neill</surname> <given-names>C</given-names></name> <name><surname>Molloy</surname> <given-names>EJ</given-names></name> <name><surname>Denton</surname> <given-names>E</given-names></name><etal/></person-group> <article-title>Large language Models&#x2019; clinical decision-making on when to perform a kidney biopsy: comparative study</article-title>. <source>J Med Internet Res</source>. (<year>2025</year>) <volume>27</volume>:<fpage>e73603</fpage>. <pub-id pub-id-type="doi">10.2196/73603</pub-id><pub-id pub-id-type="pmid">40966592</pub-id></mixed-citation></ref>
<ref id="B23"><label>23.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McCoy</surname> <given-names>LG</given-names></name> <name><surname>Nagaraj</surname> <given-names>S</given-names></name> <name><surname>Morgado</surname> <given-names>F</given-names></name> <name><surname>Harish</surname> <given-names>V</given-names></name> <name><surname>Das</surname> <given-names>S</given-names></name> <name><surname>Celi</surname> <given-names>LA</given-names></name></person-group>. <article-title>What do medical students think of AI? Implications for curricula and ethics</article-title>. <source>Acad Med</source>. (<year>2020</year>) <volume>95</volume>:<fpage>1858</fpage>&#x2013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.1097/ACM.0000000000003537</pub-id></mixed-citation></ref>
<ref id="B24"><label>24.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Leivaditis</surname> <given-names>V</given-names></name> <name><surname>Maniatopoulos</surname> <given-names>AA</given-names></name> <name><surname>Lausberg</surname> <given-names>H</given-names></name> <name><surname>Koletsis</surname> <given-names>E</given-names></name> <name><surname>Prokakis</surname> <given-names>C</given-names></name> <name><surname>Koletsis</surname> <given-names>N</given-names></name><etal/></person-group> <article-title>Artificial intelligence in thoracic surgery: a review bridging innovation and clinical practice for the next generation of surgical care</article-title>. <source>J Clin Med</source>. (<year>2025</year>) <volume>14</volume>:<fpage>2729</fpage>. <pub-id pub-id-type="doi">10.3390/jcm14082729</pub-id><pub-id pub-id-type="pmid">40283559</pub-id></mixed-citation></ref>
<ref id="B25"><label>25.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sadeghi</surname> <given-names>AH</given-names></name> <name><surname>Bakhuis</surname> <given-names>W</given-names></name> <name><surname>van der Sande</surname> <given-names>L</given-names></name> <name><surname>Goos</surname> <given-names>T</given-names></name> <name><surname>Goense</surname> <given-names>L</given-names></name> <name><surname>Akbari Aghdam</surname> <given-names>K</given-names></name><etal/></person-group> <article-title>Artificial intelligence-assisted augmented reality robotic lung surgery: navigating the future of thoracic surgery</article-title>. <source>JTCVS Tech</source>. (<year>2024</year>) <volume>26</volume>:<fpage>121</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1016/j.xjtc.2024.05.003</pub-id><pub-id pub-id-type="pmid">39156519</pub-id></mixed-citation></ref>
<ref id="B26"><label>26.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Seastedt</surname> <given-names>KP</given-names></name> <name><surname>Moukheiber</surname> <given-names>D</given-names></name> <name><surname>Muniappan</surname> <given-names>A</given-names></name> <name><surname>Gaissert</surname> <given-names>HA</given-names></name> <name><surname>Wright</surname> <given-names>CD</given-names></name> <name><surname>Lanuti</surname> <given-names>M</given-names></name><etal/></person-group> <article-title>A scoping review of artificial intelligence applications in thoracic surgery</article-title>. <source>Eur J Cardiothorac Surg</source>. (<year>2022</year>) <volume>61</volume>:<fpage>239</fpage>&#x2013;<lpage>47</lpage>. <pub-id pub-id-type="doi">10.1093/ejcts/ezab422</pub-id><pub-id pub-id-type="pmid">34601587</pub-id></mixed-citation></ref>
<ref id="B27"><label>27.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Abbaker</surname> <given-names>N</given-names></name> <name><surname>Minervini</surname> <given-names>F</given-names></name> <name><surname>Guttadauro</surname> <given-names>A</given-names></name> <name><surname>Solli</surname> <given-names>P</given-names></name> <name><surname>Cioffi</surname> <given-names>U</given-names></name></person-group>. <article-title>The future of artificial intelligence in thoracic surgery for non-small cell lung cancer treatment: a narrative review</article-title>. <source>Front Oncol</source>. (<year>2024</year>) <volume>14</volume>:<fpage>1347464</fpage>. <pub-id pub-id-type="doi">10.3389/fonc.2024.1347464</pub-id><pub-id pub-id-type="pmid">38414748</pub-id></mixed-citation></ref>
<ref id="B28"><label>28.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ong</surname> <given-names>JCL</given-names></name> <name><surname>Bharath</surname> <given-names>A</given-names></name> <name><surname>Sharma</surname> <given-names>A</given-names></name> <name><surname>Kulkarni</surname> <given-names>S</given-names></name> <name><surname>Subramanian</surname> <given-names>V</given-names></name> <name><surname>Garg</surname> <given-names>T</given-names></name><etal/></person-group> <article-title>Ethical and regulatory challenges of large language models in medicine</article-title>. <source>Lancet Digit Health</source>. (<year>2024</year>) <volume>6</volume>:<fpage>e366</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1016/S2589-7500(24)00061-X</pub-id></mixed-citation></ref></ref-list>
<fn-group>
<fn id="n1" fn-type="custom" custom-type="edited-by"><p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/544643/overview">Steffen Pauws</ext-link>, Tilburg University, Netherlands</p></fn>
<fn id="n2" fn-type="custom" custom-type="reviewed-by"><p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1457786/overview">Zhixing Song</ext-link>, Tsinghua University, China</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1514548/overview">Yuexiong Yi</ext-link>, Wuhan University, China</p></fn>
</fn-group>
</back>
</article>