<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Oncol.</journal-id>
<journal-title>Frontiers in Oncology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Oncol.</abbrev-journal-title>
<issn pub-type="epub">2234-943X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fonc.2024.1513608</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Oncology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Feasibility of large language models for CEUS LI-RADS categorization of small liver nodules in patients at risk for hepatocellular carcinoma</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Huang</surname>
<given-names>Jiayan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2912206"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yang</surname>
<given-names>Rui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1096222"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Huang</surname>
<given-names>Xiaotong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1273058"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zeng</surname>
<given-names>Keyu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2544562"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Yan</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2912846"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Luo</surname>
<given-names>Jun</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Lyshchik</surname>
<given-names>Andrej</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/932493"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Lu</surname>
<given-names>Qiang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1201244"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>West China Hospital of Sichuan University</institution>, <addr-line>Chengdu</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Ultrasound, Affiliated Hospital of Panzhihua University</institution>, <addr-line>Panzhihua</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Ultrasound, Sichuan Academy of Medical Sciences and Sichuan Provincial People&#x2019;s Hospital</institution>, <addr-line>Chengdu</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Thomas Jefferson University Hospital , Jefferson University Hospitals</institution>, <addr-line>Philadelphia, PA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Abdullah Esmail, Houston Methodist Hospital, United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Jonathan Soldera, University of Caxias do Sul, Brazil</p>
<p>Shao Bo Duan, Henan Provincial People&#x2019;s Hospital, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Andrej Lyshchik, <email xlink:href="mailto:Andrej.Lyshchik@jefferson.edu">Andrej.Lyshchik@jefferson.edu</email>; Qiang Lu, <email xlink:href="mailto:luqiang@scu.edu.cn">luqiang@scu.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>18</day>
<month>12</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>14</volume>
<elocation-id>1513608</elocation-id>
<history>
<date date-type="received">
<day>18</day>
<month>10</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>22</day>
<month>11</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Huang, Yang, Huang, Zeng, Liu, Luo, Lyshchik and Lu</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Huang, Yang, Huang, Zeng, Liu, Luo, Lyshchik and Lu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Large language models (LLMs) offer opportunities to enhance radiological applications, but their performance in handling complex tasks remains insufficiently investigated</p>
</sec>
<sec>
<title>Purpose</title>
<p>To evaluate the performance of LLMs integrated with Contrast-enhanced Ultrasound Liver Imaging Reporting and Data System (CEUS LI-RADS) in diagnosing small (&#x2264;20mm) hepatocellular carcinoma (sHCC) in high-risk patients.</p>
</sec>
<sec>
<title>Materials and Methods</title>
<p>From November 2014 to December 2023, high-risk HCC patients with untreated small (&#x2264;20mm) focal liver lesions (sFLLs), were included in this retrospective study. ChatGPT-4.0, ChatGPT-4o, ChatGPT-4o mini, and Google Gemini were integrated with imaging features from structured CEUS LI-RADS reports to assess their diagnostic performance for sHCC. The diagnostic efficacy of LLMs for small HCC were compared using McNemar test.</p>
</sec>
<sec>
<title>Results</title>
<p>The final population consisted of 403 high-risk patients (52 years &#xb1; 11, 323 men). ChatGPT-4.0 and ChatGPT-4o demonstrated substantial to almost perfect intra-agreement for CEUS LI-RADS categorization (&#x3ba; values: 0.76-1.0 and 0.7-0.94, respectively), outperforming ChatGPT-4o mini (&#x3ba; values: 0.51-0.72) and Google Gemini (&#x3ba; values: -0.04-0.47). ChatGPT-4.0 had higher sensitivity in detecting sHCC than ChatGPT-4o (83%-89% vs. 70%-78%, <italic>p</italic> &lt; 0.02) with comparable specificity (76%-90% vs. 83%-86%, <italic>p</italic> &gt; 0.05). Compared to human readers, ChatGPT-4.0 showed superior sensitivity (83%-89% vs. 63%-78%, <italic>p</italic> &lt; 0.004) and comparable specificity (76%-90% vs. 90%-95%, <italic>p</italic> &gt; 0.05) in diagnosing sHCC.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>LLM integrated with CEUS LI-RADS offers potential tool in diagnosing sHCC for high-risk patients. ChatGPT-4.0 demonstrated satisfactory consistency in CEUS LI-RADS categorization, offering higher sensitivity in diagnosing sHCC while maintaining comparable specificity to that of human readers.</p>
</sec>
</abstract>
<kwd-group>
<kwd>hepatocellular carcinoma (HCC)</kwd>
<kwd>large language model (LLM)</kwd>
<kwd>diagnosis</kwd>
<kwd>CEUS (Contrast-enhanced ultrasound)</kwd>
<kwd>ultrasound</kwd>
</kwd-group>
<counts>
<fig-count count="5"/>
<table-count count="4"/>
<equation-count count="0"/>
<ref-count count="25"/>
<page-count count="11"/>
<word-count count="5231"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Gastrointestinal Cancers: Hepato Pancreatic Biliary Cancers</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<title>Introduction</title>
<p>Liver cancer is the sixth most common cancer and the third leading cause of cancer-related deaths worldwide, with over 830,000 deaths in 2020 and rising mortality rates (<xref ref-type="bibr" rid="B1">1</xref>). Among all the pathological subtypes of liver cancer, hepatocellular carcinoma (HCC) accounts for the majority of cases. However, due to the complex dual blood supply to liver and the multistage process of HCC, radiological diagnosis of HCC remains challenging (<xref ref-type="bibr" rid="B2">2</xref>). Notably, the early diagnosis of HCC, especially for tumors measuring 2 cm or smaller in diameter (small HCC), due to it offers more treatment options, reduced risk of complications and better prognosis (<xref ref-type="bibr" rid="B3">3</xref>).</p>
<p>The Contrast-enhanced Ultrasound Liver Imaging Reporting and Data System (CEUS LI-RADS) released by American College of Radiology (ACR)aims to improve the accuracy and consistency of HCC diagnosis in patients at high-risk (<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B5">5</xref>). The implementation of structured reporting in radiology plays a pivotal role in improving communication, fostering collaboration among medical practitioners, and standardizing reporting language across institutions (<xref ref-type="bibr" rid="B6">6</xref>). Moreover, the characterization of focal liver lesions (FLLs) using CEUS LI-RADS, based on imaging features derived from B-mode and multiphasic CEUS enhancement patterns, establishes a robust foundation for the application of artificial intelligence (AI) in imaging diagnosis. Large Language Models (LLM) represent a specialized AI application that focuses on comprehending and generating text resembling human-like language. Recently conducted research on the interaction strategy between humans and LLM has shown promising results in terms of LLM-agreement and diagnostic accuracy for predicting benign and malignant thyroid nodules using the ACR Thyroid Imaging Reporting and Data System (TI-RADS) (<xref ref-type="bibr" rid="B7">7</xref>).</p>
<p>Amid significant advancements in LLMs, AI chatbots like ChatGPT have gained increasing attention across various fields (<xref ref-type="bibr" rid="B8">8</xref>). The Chatbot (primarily ChatGPT 4.0) has showed potential in transforming unstructured free-text reports into organized formats (<xref ref-type="bibr" rid="B6">6</xref>, <xref ref-type="bibr" rid="B9">9</xref>). However, the impressive capability of LLMs in rapid and standardized language processing notwithstanding, concerns have increasingly arisen regarding the assessment of their agreement and accuracy in generating prompt output. The subjective question-answering format may inadvertently be influenced by existing biases and disparities, thereby overlooking crucial aspects of transparency and accuracy when investigating the applications of LLMs (<xref ref-type="bibr" rid="B7">7</xref>, <xref ref-type="bibr" rid="B10">10</xref>). Notably, LLMs such as ChatGPT 4.0 do not support concurrent recognition or processing of multiple images. Given that this approach is more practical for handling natural language, the medical application of LLMs, particularly in processing radiology reports, represents a critical and cutting-edge area of research at the forefront of LLM advancements (<xref ref-type="bibr" rid="B6">6</xref>&#x2013;<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B11">11</xref>). Despite the promising efficacy of LLMs demonstrated by a rapidly emerging plethora of studies, their potential in aiding real-world clinical applications remains controversial.</p>
<p>To our knowledge, the performance of LLMs in classifying FLLs based on CEUS LI-RADS reports, particularly regarding their output LLMs-agreement and accuracy, has not been previously reported. Thus, the purpose of our study was to evaluate the intra- and inter-agreement of four publicly available LLMs (Google Gemini, ChatGPT-4.0, ChatGPT-4o, and ChatGPT-4o mini) in CEUS LI-RADS categorization using structured CEUS reports from patients with small FLLs. Moreover, the diagnostic accuracy of LLMs in diagnosing small HCC was also investigated, using a composite reference standard as previously described (<xref ref-type="bibr" rid="B12">12</xref>).</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<title>Materials and methods</title>
<p>This retrospective study was approved by the ethics committee of West China Hospital of Sichuan University, and written informed consent was waived. To consecutively collect participants, 172 samples in this study were drawn from our previous investigation (<xref ref-type="bibr" rid="B12">12</xref>). Ultrasound images of these patients were used to develop structured reports for LLMs, while in the earlier research, they were used for evaluating the diagnostic accuracy of CEUS LI-RADS (version 2017).</p>
<sec id="s2_1">
<title>Study design</title>
<p>This study investigated four publicly available LLM chatbots, including three versions of ChatGPT (ChatGPT-4.0, ChatGPT-4o, ChatGPT-4o mini) and Google Gemini. Ultrasound images of small focal liver lesions (sFLLs) were assessed by six certified ultrasound radiologists with 3 to 20 years of liver CEUS experience using CEUS LI-RADS (Version 2017). Structured reports were then rendered and entered into LLMs for prompt CEUS LI-RADS categorization. Three output rounds (with an extra round if completely inconsistent) were performed to evaluate the intra-agreement of LLM in sFLLs categorization, with a majority vote deciding the final CEUS LI-RADS category. To avoid space-time impact, outputs were spaced three days apart. The diagnostic efficacy of LLMs for sHCC was compared by analyzing reports from readers with varying expertise. Furthermore, the best-performing LLM was compared to human readers and a convolutional neural network (CNN) model in diagnosing sHCC (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Graphical representation of study design. Overall, LLMs were integrated with structured reports based on the CEUS LI-RADS for diagnosing sHCC (the top box). First, the LLMs' agreement was evaluated by comparing intra-LLM consistency, with the most frequently voted category used for further inter-LLMs agreement assessment (the middle box). Second, the diagnostic performance of the LLMs was assessed in comparison to human readers utilizing CEUS LI-RADS (Ver. 2017) for diagnosing sHCC, as well as compared to a CNN model (the bottom box). LLMs, large language models; CEUS LI-RADS, Contrast-enhanced Ultrasound Liver Imaging Reporting and Data System; sHCC, small hepatocellular carcinoma; CNN, convolutional neural network.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-14-1513608-g001.tif"/>
</fig>
</sec>
<sec id="s2_2">
<title>Patient selection and reference standard</title>
<p>Consecutive patients underwent hepatic CEUS examinations were retrospectively collected from November 2014 to December 2023. The inclusion criteria were: <italic>(a)</italic> aged 18 or older; <italic>(b)</italic> HCC risk factors involving cirrhosis or chronic hepatitis B (HBV); <italic>(c)</italic> untreated hepatic nodules &#x2264;20 mm on imaging (ultrasound, CT or MRI); <italic>(d)</italic> less than two sFLLs, with the larger tumor selected for analysis to minimize the impact of multiple injections on contrast enhancement. Exclusion criteria included: <italic>(a)</italic> indeterminate pathology, contrast-enhanced CT/MRI results, or incomplete follow-up; <italic>(b)</italic> poor-quality ultrasound images. HCC risk factors were defined as any cause of cirrhosis and/or HBV, per the American Association for the Study of Liver Diseases guidelines (<xref ref-type="bibr" rid="B13">13</xref>). Patients with a history of HCC treatment were excluded to minimize the influence from post-treatment changes.</p>
<p>This study used a composite reference standard as previously described in our earlier investigation (<xref ref-type="bibr" rid="B12">12</xref>). In brief, all lesions were diagnosed by histopathology, while contrast-enhanced CT/MRI were used for LR-1 or LR-5 nodules. Diagnosis for LR-2, LR-3, and LR-4 nodules involved imaging follow-up (&#x2265;12 months), pathology, or multidisciplinary recommendations, while LR-M lesions were diagnosed by histopathology. The processing of the CEUS examination is detailed in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material Appendix S1</bold>
</xref>.</p>
</sec>
<sec id="s2_3">
<title>CEUS LI-RADS category assignment by LLMs</title>
<p>After independent review of the images by radiologists, a separate radiologist translated the structured reports, including patients&#x2019; clinical and ultrasound characteristics, from Chinese to English. Then, the reports were input into ChatGPT-4o mini (<xref ref-type="bibr" rid="B14">14</xref>), ChatGPT-4o (<xref ref-type="bibr" rid="B15">15</xref>), ChatGPT-4.0 (<xref ref-type="bibr" rid="B16">16</xref>) and Google Gemini (<xref ref-type="bibr" rid="B17">17</xref>) for prompt CEUS LI-RADS classification. LLMs were integrated with CEUS LI-RADS reports to evaluate their agreement and diagnostic performance in diagnosing sHCC, since LLMs currently cannot interpret multiple images directly. The CEUS LI-RADS classification process for human readers and LLMs is detailed in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material Appendix S2</bold>
</xref> and <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>.</p>
</sec>
<sec id="s2_4">
<title>End-to-end CNN model of small FLLs</title>
<p>End-to-end CNN models involving baseline and multiphase CEUS images were developed for sHCC diagnosis. Patients were randomly divided into training (59.8%, 241 of 403) and validation sets (40.2%, 162 of 403). The diagnostic efficacy of the CNN for sHCC was evaluated using the validation cohort. Algorithm and network structures were established as described in previous investigations (<xref ref-type="bibr" rid="B7">7</xref>, <xref ref-type="bibr" rid="B18">18</xref>). Due to the characterization of FLLs according to CEUS LI-RADS, which incorporates imaging features from both baseline and multiphasic CEUS images, results were derived by integrating outputs from each modality using a weighted averaging method. The CNN was developed using ultrasound images of sFLLs, with tumor segmentation performed by a radiologist with three years of liver CEUS experience, utilizing MITK Workbench (<ext-link ext-link-type="uri" xlink:href="https://docs.mitk.org/nightly/index.html">https://docs.mitk.org/nightly/index.html</ext-link>). The model was built and executed in Python (version 3.10.3; Python Software Foundation, Wilmington, Delaware, USA). Detailed information regarding network structures and parameters is provided in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material Appendix S3</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S2</bold>
</xref> and <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1</bold>
</xref>.</p>
</sec>
<sec id="s2_5">
<title>Statistical analysis</title>
<p>Fleiss&#x2019; Kappa and Cohen&#x2019;s Kappa tests were used to assess intra- and inter-LLM agreements, respectively. A best-of-three strategy was used to identify the preferred LLM category for sFLLs. Agreement strength was classified using the Landis and Koch scale: 0-0.20 as poor; 0.21-0.40 as fair; 0.41-0.60 as moderate; 0.61-0.80 as substantial; and 0.81-1.00 as almost perfect.</p>
<p>LR-5 category is designated for predicting HCC according to CEUS LI-RADS, whereas LR-4, LR-5, and LR-M are categorized as indicative of malignancy (<xref ref-type="bibr" rid="B19">19</xref>). The diagnostic performance of LLMs, human readers, and CNN strategy for sHCC was assessed by calculating sensitivity, specificity, accuracy and area under the receiver operating characteristic curve (AUC) based on standard procedures (<xref ref-type="bibr" rid="B12">12</xref>). Sensitivity, specificity, and accuracy were compared among LLMs, between LLMs and human readers, and between LLMs and CNNs using the McNemar test. AUCs for sHCC were compared among LLMs, human readers, and CNN using DeLong test.</p>
<p>Statistical analyses were conducted using R packages (R 4.1.2 [Puppy Cup], The R Foundation, Vienna, Austria) and MedCalc software (MedCalc22.030, Ostend, Belgium). A P-value less than.05 indicated statistical significance.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>Results</title>
<sec id="s3_1">
<title>Patients and liver nodule characteristics</title>
<p>A total of 1612 representative ultrasound images, including B-mode, arterial phase, portal phase, and late phase images (one representative image per phase), were obtained from 403 patients at risks of HCC with sFLLs (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). Of the 403 patients (mean age, 52.3 years &#xb1; 10.8; age range, 21&#x2013;81 years), 323 (80.1%) were men. The mean size of sFLLs was 16.1 mm &#xb1; 3.4. Clinical features of patients involving age, gender, liver disease etiology, nodule size, and pathological results are exhibited in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. Based on the composite reference standard, 263 liver nodules were proved by pathology, 65 by follow-up, and 75 by contrast enhanced CT or MRI (including 42 HCC and 33 hemangioma). The median follow-up period was 15.2 months (range 12&#x2013;41 months). The constitution of 403 sFLLs and the distribution of CEUS LI-RADS categories, as determined by human readers and LMMs, are depicted in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Study population flowchart. US, ultrasound; HCC, hepatocellular carcinoma; FLL, focal liver lesion.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-14-1513608-g002.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Clinicopathological characteristics of patients.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Characteristic</th>
<th valign="top" align="left">Value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" colspan="2" align="left">Sex</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Men</td>
<td valign="top" align="left">323 (80.1)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Woman</td>
<td valign="top" align="left">80 (19.9)</td>
</tr>
<tr>
<td valign="top" align="left">Mean age (y)<sup>*</sup>
</td>
<td valign="top" align="left">52.3 &#xb1; 10.8 (21&#x2013;81)</td>
</tr>
<tr>
<td valign="top" align="left">Mean nodule size (mm)<sup>*</sup>
</td>
<td valign="top" align="left">16.2 &#xb1; 3.4 (0.7&#x2013;2)</td>
</tr>
<tr>
<th valign="top" colspan="2" align="left">Liver disease etiologic cause</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;HBV</td>
<td valign="top" align="left">374 (92.8)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;HCV</td>
<td valign="top" align="left">11 (2.7)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;HBV and HCV</td>
<td valign="top" align="left">6 (1.5)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;PBC</td>
<td valign="top" align="left">1 (0.2)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Alcohol</td>
<td valign="top" align="left">4 (1)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Unknown etiology</td>
<td valign="top" align="left">7 (1.7)</td>
</tr>
<tr>
<td valign="top" align="left">Cirrhosis</td>
<td valign="top" align="left">162 (40.2)</td>
</tr>
<tr>
<th valign="top" colspan="2" align="left">Pathologic Analysis</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;HCC</td>
<td valign="top" align="left">223 (55.3)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;Well differentiated</td>
<td valign="top" align="left">5 (1.2)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;Moderately differentiated</td>
<td valign="top" align="left">161 (40)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;Poorly differentiated</td>
<td valign="top" align="left">57 (14.1)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;DN/RN</td>
<td valign="top" align="left">21 (5.2)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;FNH</td>
<td valign="top" align="left">2 (0.5)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Hemangioma</td>
<td valign="top" align="left">3 (0.7)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;ICC</td>
<td valign="top" align="left">8 (2)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;cHCC-CCA</td>
<td valign="top" align="left">2 (0.5)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Metastasis</td>
<td valign="top" align="left">1 (0.2)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Reactive lymphoid hyperplasia</td>
<td valign="top" align="left">1 (0.2)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Biliary adenoma</td>
<td valign="top" align="left">1 (0.2)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;NEN</td>
<td valign="top" align="left">1 (0.2)</td>
</tr>
<tr>
<th valign="top" colspan="2" align="left">No pathologic analysis</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Contrast-enhanced CT or MRI</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;HCC</td>
<td valign="top" align="left">42 (10.4)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;Hemangioma</td>
<td valign="top" align="left">33 (8.2)</td>
</tr>
<tr>
<th valign="top" colspan="2" align="left">Follow-up</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;&lt; 50% size increase in 12 months</td>
<td valign="top" align="left">62 (15.4)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;&#x2265;50% size increase in 12 months</td>
<td valign="top" align="left">3 (0.7)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Unless otherwise indicated, data are liver nodules or patients (n=403) and data in parentheses are percentages. Mean data are &#xb1; standard deviation. HBV, hepatitis B virus; HCV, hepatitis C virus; PBC, primary biliary cirrhosis; HCC, hepatocellular carcinoma; DN, dysplastic nodule; RN, regenerative nodule; FNH, focal nodular hyperplasia; ICC, intrahepatic cholangiocarcinoma; cHCC-CCA, combined hepatocellular-cholangiocarcinoma; NEN, neuroendocrine neoplasm.</p>
</fn>
<fn>
<p>
<sup>*</sup>Data in parentheses are range.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Pathological composition and CEUS LI-RADS categories of liver nodules classified by human reader and LLMs. <bold>(A)</bold> Pie chart depicts the pathological composition of 403 small FLLs according to the reference standard. <bold>(B)</bold> Horizontal stacked bar chart illustrates the distribution of CEUS LI-RADS categories for small FLLs as assigned by both LLMs and human readers using CEUS LI-RADS category. CEUS LI-RADS, Contrast-enhanced Ultrasound Liver Imaging Reporting and Data System; FLL, focal liver lesions; LLMs, large language models; HCC, hepatocellular carcinoma; FNH, focal nodular hyperplasia; ICC, intrahepatic cholangiocarcinoma; DN, dysplastic Nodule; RN, regenerative Nodule; cHCC-ICC, combined hepatocellular-cholangiocarcinoma.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-14-1513608-g003.tif"/>
</fig>
</sec>
<sec id="s3_2">
<title>Intra- and inter-LLM Agreement on CEUS LI-RADS categorization for small FLLs</title>
<p>The distributions of intra-LLM agreement and inter-LLM agreement are presented in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. ChatGPT-4.0 and ChatGPT-4o demonstrated substantial to almost perfect intra-agreement in the CEUS LI-RADS classification assignment among radiologists with varying levels of liver CEUS experience (&#x3ba; value = 0.76-1[95% CI: 0.69, 1], and 0.7-0.94 [95% CI: 0.55, 0.99] for ChatGPT-4.0, and ChatGPT 4o, respectively). There was moderate to substantial intra-agreement for ChatGPT-4o mini, with &#x3ba; values ranging from 0.51 to 0.72 (95% CI: 0.33 to 0.79). However, apart from moderate agreement for a junior radiologist, Google Gemini demonstrated poor to fair intra-LLM agreement for the other radiologists (&#x3ba; value = -0.04 to 0.47 [95% CI: -0.32, 0.6]).</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Intra-LLMs and Inter-LLMs agreements for CEUS LI-RADS category assignments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="3" align="center">Agreement evaluation</th>
<th valign="top" colspan="6" align="center">Human-LLM Interaction</th>
</tr>
<tr>
<th valign="top" colspan="2" align="center">Junior Radiologist</th>
<th valign="top" colspan="2" align="center">Senior Radiologist</th>
<th valign="top" colspan="2" align="center">Expert Radiologist</th>
</tr>
<tr>
<th valign="top" align="center">1</th>
<th valign="top" align="center">2</th>
<th valign="top" align="center">1</th>
<th valign="top" align="center">2</th>
<th valign="top" align="center">1</th>
<th valign="top" align="center">2</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="top" colspan="7" align="left">
<italic>Intra-LLM agreement<sup>*</sup>
</italic>
</th>
</tr>
<tr>
<td valign="top" align="left">Gemini</td>
<td valign="top" align="left">-0.04 (-0.32-0.5)</td>
<td valign="top" align="left">0.47 (0.36-0.6)</td>
<td valign="top" align="left">0.3 (0.15-0.46)</td>
<td valign="top" align="left">-0.03 (-0.1-0.1)</td>
<td valign="top" align="left">0.26 (0.12-0.41)</td>
<td valign="top" align="left">0.07 (-0.1-0.28)</td>
</tr>
<tr>
<td valign="top" align="left">GPT-4o mini</td>
<td valign="top" align="left">0.71 (0.35, 0.93)</td>
<td valign="top" align="left">0.72 (0.65, 0.79)</td>
<td valign="top" align="left">0.68 (0.56, 0.77)</td>
<td valign="top" align="left">0.59 (0.49, 0.69)</td>
<td valign="top" align="left">0.62 (0.5, 0.72)</td>
<td valign="top" align="left">0.51 (0.33, 0.67)</td>
</tr>
<tr>
<td valign="top" align="left">GPT-4.0</td>
<td valign="top" align="left">1 (1-1)</td>
<td valign="top" align="left">0.84 (0.8-0.88)</td>
<td valign="top" align="left">0.88 (0.83-0.9)</td>
<td valign="top" align="left">0.76 (0.69-0.83)</td>
<td valign="top" align="left">0.9 (0.86-0.93)</td>
<td valign="top" align="left">0.95 (0.92-0.97)</td>
</tr>
<tr>
<td valign="top" align="left">GPT-4o</td>
<td valign="top" align="left">0.94 (0.82-0.99)</td>
<td valign="top" align="left">0.78 (0.7-0.84)</td>
<td valign="top" align="left">0.73 (0.63-0.8)</td>
<td valign="top" align="left">0.8 (0.74-0.85)</td>
<td valign="top" align="left">0.85 (0.79-0.9)</td>
<td valign="top" align="left">0.7 (0.55-0.81)</td>
</tr>
<tr>
<th valign="top" colspan="7" align="left">
<italic>Inter-LLM agreement<sup>&#x2020;</sup>
</italic>
</th>
</tr>
<tr>
<td valign="top" align="left">Gemini vs GPT-4o mini</td>
<td valign="top" align="left">-0.14 (-0.7, 0.6)</td>
<td valign="top" align="left">0.55 (0.41, 0.67)</td>
<td valign="top" align="left">0.13 (-0.1, 0.36)</td>
<td valign="top" align="left">-0.3 (-0.47, -0.1)</td>
<td valign="top" align="left">-0.05 (-0.3, 0.2)</td>
<td valign="top" align="left">-0.14 (-0.4, 0.2)</td>
</tr>
<tr>
<td valign="top" align="left">Gemini vs GPT-4.0</td>
<td valign="top" align="left">0.01 (-0.63-0.67)</td>
<td valign="top" align="left">0.27 (0.09-0.4)</td>
<td valign="top" align="left">0.06 (-0.18-0.3)</td>
<td valign="top" align="left">-0.2 (-0.4-0.02)</td>
<td valign="top" align="left">0.14 (-0.1-0.35)</td>
<td valign="top" align="left">0.15 (-0.16-0.44)</td>
</tr>
<tr>
<td valign="top" align="left">Gemini vs GPT-4o</td>
<td valign="top" align="left">0.04 (-0.61-0.68)</td>
<td valign="top" align="left">0.18 (-0.01-0.4)</td>
<td valign="top" align="left">0.23 (-0.01-0.4)</td>
<td valign="top" align="center">-0.05 (-0.2-0.2)</td>
<td valign="top" align="left">0.04 (-0.2-0.27)</td>
<td valign="top" align="left">0.16 (-0.15-0.44)</td>
</tr>
<tr>
<td valign="top" align="left">GPT-4o mini vs GPT-4.0</td>
<td valign="top" align="left">0.41 (-0.3, 0.84)</td>
<td valign="top" align="left">0.35 (0.18, 0.5)</td>
<td valign="top" align="left">0.55 (0.36, 0.69)</td>
<td valign="top" align="left">0.48 (0.32, 0.62)</td>
<td valign="top" align="left">0.27 (0.05, 0.47)</td>
<td valign="top" align="left">0.34 (0.04, 0.58)</td>
</tr>
<tr>
<td valign="top" align="left">GPT-4o mini vs GPT-4o</td>
<td valign="top" align="left">0.66 (0.04, 0.92)</td>
<td valign="top" align="left">0.19 (-0.01-0.4)</td>
<td valign="top" align="left">0.47 (0.26, 0.63)</td>
<td valign="top" align="left">0.34 (0.16, 0.5)</td>
<td valign="top" align="left">0.25 (0.02, 0.45)</td>
<td valign="top" align="left">0.16 (-0.15, 0.4)</td>
</tr>
<tr>
<td valign="top" align="left">GPT-4.0 vs GPT-4o</td>
<td valign="top" align="left">0.86 (0.50-0.97)</td>
<td valign="top" align="left">0.7 (0.58-0.78)</td>
<td valign="top" align="left">0.83 (0.7-0.89)</td>
<td valign="top" align="left">0.71 (0.6-0.8)</td>
<td valign="top" align="left">0.69 (0.55-0.79)</td>
<td valign="top" align="left">0.63 (0.4-0.78)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Three levels of radiologists individually generated 403 CEUS LI-RSDS reports, along with a prompt output of a CEUS LI-RADS category. Data are &#x3ba; values, and data in parentheses are 95% CIs. LLM, large language model; CEUS LI-RADS, contrast-enhanced US Liver Imaging Reporting and Data System.</p>
</fn>
<fn>
<p>
<sup>*</sup>Kappa values calculated as described by Fleiss&#x2019; &#x3ba; for the Intra-LLM agreement and their 95% CIs.</p>
</fn>
<fn>
<p>
<sup>&#x2020;</sup>Kappa values calculated as described by Cohen&#x2019; &#x3ba; for the Inter-LLM agreement and their 95% CIs.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>As for the inter-LLM agreement evaluation, GPT-4.0 and GPT-4o achieved substantial to almost perfect agreement for both the junior and senior radiologists (&#x3ba; value = 0.7-0.86 [95% CI: 0.58, 0.97]), and substantial agreement for the expert radiologists (&#x3ba; value = 0.63-0.69 [95% CI: 0.4, 0.79]), respectively. ChatGPT-4o mini showed fair to moderate agreement with ChatGPT-4.0 (&#x3ba; value = 0.27-0.55 [95% CI: 0.05, 0.69]) involving all readers, whereas poor to substantial agreement with ChatGPT-4o (&#x3ba; value = 0.16-0.66 [95% CI: -0.15, 0.92]). There was poor to fair agreement between Google Gemini and ChatGPT, including version 4o mini, 4.0 and 4o, with &#x3ba; values ranging from -0.3 to 0.55 (95% CI: -0.47 to 0.67), regardless of the radiologist&#x2019;s expertise.</p>
</sec>
<sec id="s3_3">
<title>Diagnostic efficacy of ChatGPT-4.0 and ChatGPT-4o in predicting small HCC</title>
<p>The diagnostic performance of LLMs in diagnosing sHCC is shown in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> and <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>. Since ChatGPT-4.0 showed comparable intra-LLM and superior inter-LLM agreement to other LLMs, its diagnostic performance was evaluated in greater detail. In a human-LLM interaction context, ChatGPT-4.0 demonstrated superior sensitivity compared to ChatGPT-4o across all readers levels, achieving 83% [95% CI: 73%, 90%] versus 70% [95% CI: 59%, 79%] for junior radiologists (<italic>p</italic> = 0.007), 86% [95% CI: 78%, 92%] versus 77% [95% CI: 68%, 84%] for senior radiologists (<italic>p</italic> = 0.02), and 90% [95% CI: 81%, 95%] versus 78% [95% CI: 67%, 87%] for expert radiologists (<italic>p</italic> = 0.004), respectively. However, ChatGPT-4.0 and ChatGPT-4o exhibited comparable specificity in differentiating sHCC from non-HCC. Regarding diagnostic accuracy, ChatGPT-4.0 demonstrated superior performance compared to ChatGPT 4o for senior (87% [95% CI: 81%, 92%] vs 79% [95% CI: 72%, 85%], <italic>p</italic> = 0.009) and expert radiologists (90% [95% CI: 83%, 94%] vs 80% [95% CI: 72%, 87%], <italic>p</italic> = 0.001). However, they showed comparable performance for junior radiologists (81% [95% CI: 73%, 88%] vs 74% [95% CI: 65%, 82%], <italic>p</italic> = 0.12). Similarly, the AUC for ChatGPT-4.0 was higher for senior and expert radiologists (<italic>p</italic> = 0.01 and <italic>p</italic> = .001, respectively), but comparable for junior radiologists, when compared with ChatGPT-4o.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Comparison of ChatGPT 4.0 and ChatGPT 4o in predicting small HCC versus Non-HCC.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Diagnostic Performance</th>
<th valign="top" colspan="3" align="center">Human-LLM Interaction</th>
</tr>
<tr>
<th valign="top" align="center">Junior Radiologist</th>
<th valign="top" align="center">Senior Radiologist</th>
<th valign="top" align="center">Expert Radiologist</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="top" colspan="4" align="left">Sensitivity (%)</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;GPT-4.0</td>
<td valign="top" align="center">83 (72/87)</td>
<td valign="top" align="center">86 (93/108)</td>
<td valign="top" align="center">89 (69/77)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;GPT-4o</td>
<td valign="top" align="center">70 (61/87)</td>
<td valign="top" align="center">77 (83/108)</td>
<td valign="top" align="center">78 (60/77)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;<italic>p</italic> Value</td>
<td valign="top" align="center">0.007</td>
<td valign="top" align="center">0.02</td>
<td valign="top" align="center">0.004</td>
</tr>
<tr>
<th valign="top" colspan="4" align="left">Specificity (%)</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;GPT-4.0</td>
<td valign="top" align="center">76 (22/29)</td>
<td valign="top" align="center">89 (56/63)</td>
<td valign="top" align="center">90 (35/39)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;GPT-4o</td>
<td valign="top" align="center">86 (25/29)</td>
<td valign="top" align="center">83 (52/63)</td>
<td valign="top" align="center">85 (33/39)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;<italic>p</italic> Value</td>
<td valign="top" align="center">0.38</td>
<td valign="top" align="center">0.34</td>
<td valign="top" align="center">0.50</td>
</tr>
<tr>
<th valign="top" colspan="4" align="left">Accuracy (%)</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;GPT-4.0</td>
<td valign="top" align="center">81(94/116)</td>
<td valign="top" align="center">87 (149/171)</td>
<td valign="top" align="center">90 (104/116)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;GPT-4o</td>
<td valign="top" align="center">74 (86/116)</td>
<td valign="top" align="center">79 (135/171)</td>
<td valign="top" align="center">80 (93/116)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;<italic>p</italic> Value</td>
<td valign="top" align="center">0.12</td>
<td valign="top" align="center">0.009</td>
<td valign="top" align="center">0.001</td>
</tr>
<tr>
<th valign="top" colspan="4" align="left">AUC</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;GPT-4.0</td>
<td valign="top" align="center">0.79 (0.71, 0.86)</td>
<td valign="top" align="center">0.88 (0.82, 0.92)</td>
<td valign="top" align="center">0.89 (0.83, 0.95)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;GPT-4o</td>
<td valign="top" align="left">0.78 (0.70, 0.85)</td>
<td valign="top" align="center">0.79 (0.73, 0.86)</td>
<td valign="top" align="center">0.81 (0.73, 0.88)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;<italic>p</italic> Value</td>
<td valign="top" align="center">0.79</td>
<td valign="top" align="center">0.01</td>
<td valign="top" align="center">0.001</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Data in parentheses for the sensitivity, specificity and accuracy are numerator/denominator; data in parentheses for the AUC are 95% confidence intervals. <italic>p</italic> values present the comparison of performance between ChatGPT 4.0 and ChatGPT 4o in predicting small HCC using the same input generated by the same radiologist. LLM, large language model; HCC, hepatocellular carcinoma, AUC = area under a receiver operating characteristic curve.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_4">
<title>Performance of human-LLM interaction, CEUS LI-RADS and CNN strategy in diagnosing small HCC</title>
<p>
<xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> shows the diagnostic effectiveness for sHCC by ChatGPT-4.0, human readers using CEUS LI-RADS, and a CNN strategy. ChatGPT-4.0 achieved significantly higher sensitivities of 83% (95% CI: 73%, 90%), 86% (95% CI: 78%, 92%), and 89% (95% CI: 81%, 95%) for junior, senior, and expert radiologists, respectively, compared to human readers with corresponding liver CEUS expertise, who demonstrated sensitivities of 63% (95% CI: 52%, 73%) (<italic>p</italic> &lt;.001), 69% (95% CI: 60%, 78%) (<italic>p</italic> &lt; 0.001), and 78% (95% CI: 67%, 86%) (<italic>p</italic> = 0.004). Besides, ChatGPT-4.0 had similar specificity (76%-90% [95% CI: 56%, 97%] vs 90%-95% [95% CI: 73%, 99%]) to that of human readers, with all P-values above.05. As for accuracy, ChatGPT-4.0 achieved 81% (95% CI: 73%, 88%) for the junior radiologist and 87% (95% CI: 81%, 91%) for the senior radiologist, outperforming human readers who showed accuracies of 70% (95% CI: 61%, 78%) for the junior radiologist (<italic>p</italic> = 0.007) and 79% (95% CI: 72%, 85%) for the senior radiologist (<italic>p</italic> = 0.004), respectively. Notably, ChatGPT-4.0 showed comparable accuracy to that of the expert radiologist (90% [95% CI: 83%, 94%] vs 84% [95% CI: 76%, 90%], <italic>p</italic> =0.07) and AUC (0.89 [95% CI: 0.83, 0.95] vs 0.86 [95% CI: 0.79, 0.92], <italic>p</italic> = 0.20). Examples of LLMs for the CEUS LI-RADS category for sHCC are presented in <xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4</bold>
</xref> and <xref ref-type="fig" rid="f5">
<bold>5</bold>
</xref>.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Diagnostic performance of ChatGPT-4.0, human reader, and US Images-based CNN model in predicting small HCC versus Non-HCC.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Diagnostic Performance</th>
<th valign="middle" align="center">SEN (%)</th>
<th valign="middle" align="center">
<italic>p</italic> Value</th>
<th valign="middle" align="center">SPE (%)</th>
<th valign="middle" align="center">
<italic>p</italic> Value</th>
<th valign="middle" align="center">ACC (%)</th>
<th valign="middle" align="center">
<italic>p</italic> Value</th>
<th valign="middle" align="center">AUC<sup>&#x2021;</sup>
</th>
<th valign="middle" align="center">
<italic>p</italic> Value</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="top" colspan="3" align="left">
<italic>ChatGPT-4.0 vs Human Reader</italic>
</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
</tr>
<tr>
<td valign="top" colspan="3" align="left">&#x2003;ChatGPT-4.0</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;Junior Radiologist</td>
<td valign="top" align="left">83 (72/87)</td>
<td valign="top" align="left">&lt;0.001<sup>*</sup>
</td>
<td valign="top" align="left">76 (22/29)</td>
<td valign="top" align="left">0.13<sup>*</sup>
</td>
<td valign="top" align="left">81 (94/116)</td>
<td valign="top" align="left">0.007<sup>*</sup>
</td>
<td valign="top" align="left">0.79 (0.71, 0.86)</td>
<td valign="top" align="left">0.46<sup>*</sup>
</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;Senior Radiologist</td>
<td valign="top" align="left">86 (93/108)</td>
<td valign="top" align="left">&lt;0.001<sup>*</sup>
</td>
<td valign="top" align="left">89 (56/63)</td>
<td valign="top" align="left">0.13<sup>*</sup>
</td>
<td valign="top" align="left">87 (149/171)</td>
<td valign="top" align="left">0.004<sup>*</sup>
</td>
<td valign="top" align="left">0.88 (0.82, 0.92)</td>
<td valign="top" align="left">0.03<sup>*</sup>
</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;Expert Radiologist</td>
<td valign="top" align="left">89 (69/77)</td>
<td valign="top" align="left">0.004<sup>*</sup>
</td>
<td valign="top" align="left">90 (35/39)</td>
<td valign="top" align="left">0.50<sup>*</sup>
</td>
<td valign="top" align="left">90 (104/116)</td>
<td valign="top" align="left">0.07<sup>*</sup>
</td>
<td valign="top" align="left">0.89 (0.83, 0.95)</td>
<td valign="top" align="left">0.20<sup>*</sup>
</td>
</tr>
<tr>
<td valign="top" colspan="3" align="left">&#x2003;Human Reader with CEUS LI-RADS</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;Junior Radiologist</td>
<td valign="top" align="left">63 (55/87)</td>
<td valign="top" align="left"/>
<td valign="top" align="left">90 (26/29)</td>
<td valign="top" align="left"/>
<td valign="top" align="left">70 (81/116)</td>
<td valign="top" align="left"/>
<td valign="top" align="left">0.76 (0.68, 0.84)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;Senior Radiologist</td>
<td valign="top" align="left">69 (75/108)</td>
<td valign="top" align="left"/>
<td valign="top" align="left">95 (60/63)</td>
<td valign="top" align="left"/>
<td valign="top" align="left">79 (135/171)</td>
<td valign="top" align="left"/>
<td valign="top" align="left">0.82 (0.76, 0.88)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;Expert Radiologist</td>
<td valign="top" align="left">78 (60/77)</td>
<td valign="top" align="left"/>
<td valign="top" align="left">95 (60/77)</td>
<td valign="top" align="left"/>
<td valign="top" align="left">84 (97/116)</td>
<td valign="top" align="left"/>
<td valign="top" align="left">0.86 (0.79, 0.92)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<th valign="top" colspan="2" align="left">
<italic>ChatGPT-4.0 vs CNN</italic>
</th>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
<th valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;ChatGPT-4.0</td>
<td valign="top" align="left">86 (94/110)</td>
<td valign="top" align="left">0.004<sup>&#x2020;</sup>
</td>
<td valign="top" align="left">87 (45/52)</td>
<td valign="top" align="left">&lt;0.001<sup>&#x2020;</sup>
</td>
<td valign="top" align="left">86 (139/162)</td>
<td valign="top" align="left">0.01<sup>&#x2020;</sup>
</td>
<td valign="top" align="left">0.86 (0.8, 0.91)</td>
<td valign="top" align="left">&lt;0.001<sup>&#x2020;</sup>
</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;&#x2003;CNN</td>
<td valign="top" align="left">96 (106/110)</td>
<td valign="top" align="left"/>
<td valign="top" align="left">29 (15/52)</td>
<td valign="top" align="left"/>
<td valign="top" align="left">75 (121/162)</td>
<td valign="top" align="left"/>
<td valign="top" align="left">0.63 (0.55, 0.7)</td>
<td valign="top" align="left"/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Unless otherwise indicated, data in parentheses are numerators/denominators. CNN, convolutional neural network; HCC, hepatocellular carcinoma; SEN, sensitivity; SPE, specificity; ACC, accuracy; AUC, area under a receiver operating characteristic curve.</p>
</fn>
<fn>
<p>
<sup>*</sup>
<italic>p</italic> values present the comparison of performance between ChatGPT-4.0 and the human reader who generated the original structured CEUS LI-RADS reports.</p>
</fn>
<fn>
<p>
<sup>&#x2020;</sup>
<italic>p</italic>values are for comparing the diagnostic performance between ChatGPT-4.0 and CNN model.</p>
</fn>
<fn>
<p>
<sup>&#x2021;</sup>Data in parentheses are 95% confidence intervals.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Responses for a CEUS LI-RADS LR-M category small liver lesion classified by LLMs. This lesion was classified as CEUS LI-RADS LR-5 category by ChatGPT-4o <bold>(A)</bold>, ChatGPT-4.0 <bold>(B)</bold>, Google Gemini <bold>(D)</bold>, however, it was assigned to LR-4 by ChatGPT-4o mini <bold>(C)</bold>. The lesion was confirmed as a moderately-differentiated HCC by histopathology. CEUS LI-RADS, Contrast-enhanced Ultrasound Liver Imaging Reporting and Data System; LLMs, large language models; HCC, hepatocellular carcinoma.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-14-1513608-g004.tif"/>
</fig>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Responses for a CEUS LI-RADS LR-4 category small liver lesion classified by LLMs. The lesion was classified as CEUS LI-RADS LR-5 category by ChatGPT-4o <bold>(A)</bold> and Google Gemini <bold>(D)</bold>. However, it was categorized as LR-4 and LR-3 by ChatGPT-4.0 <bold>(B)</bold> and ChatGPT-4o mini, respectively. The lesion was confirmed as a poorly-differentiated HCC by histopathology. CEUS LI-RADS, Contrast-enhanced Ultrasound Liver Imaging Reporting and Data System; LLMs, large language models; HCC, hepatocellular carcinoma.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-14-1513608-g005.tif"/>
</fig>
<p>The CNN model showed higher sensitivity at 96% (95% CI: 91%, 97%) compared to 86% (95% CI: 77%, 91%) for ChatGPT-4.0 with CEUS LI-RADS (<italic>p</italic> = 0.004), but lower specificity at 29% [95% CI: 82%, 87%] versus 87% [95% CI: 74%, 94%] (<italic>p =</italic> &lt; 0.001). Moreover, ChatGPT-4.0 exhibited superior accuracy and AUC compared to the CNN model, with an accuracy of 86% (95% CI: 79%, 91%) versus 75% (95% CI: 67%, 81%, <italic>p</italic> = 0.01), and an AUC of 0.86 (95% CI: 0.80, 0.91) versus 0.63 (95% CI: 0.55, 0.70, <italic>p</italic> &lt; 0.001).</p>
<p>Additionally, the diagnostic performance of ChatGPT-4.0, human readers using CEUS LI-RADS, and a CNN model for malignant sFLLs was investigated, as shown in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S3</bold>
</xref>. <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S4</bold>
</xref> presents the performance of ChatGPT-4o, ChatGPT-4o mini and Genimi in differentiating malignant from benign sFLLs.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<title>Discussion</title>
<p>In this study, we investigated the intra- and inter-agreement, as well as the diagnostic accuracy of four popular large language models (LLMs) in diagnosing small hepatocellular carcinoma (sHCC) in high-risk patients. ChatGPT-4.0 and ChatGPT-4o showed substantial to almost perfect intra-agreement (&#x3ba; = 0.76-1 and 0.7-0.94, respectively) and higher inter-agreement than other LLMs (&#x3ba; = 0.63-0.86). In human-LLM interactions using CEUS LI-RADS, ChatGPT-4.0 demonstrated comparable specificity (76%-90%) across radiologists with varying levels of liver CEUS expertise, similar to ChatGPT-4o (83%-86%). However, ChatGPT-4.0 outperformed ChatGPT-4o with a sensitivity of 83%-89% versus 70%-78%, <italic>p &#x2264;</italic> 0.02. Notably, ChatGPT-4.0 demonstrated superior sensitivity, ranging from 83% to 89% compared to 63% to 78% for human readers (<italic>p</italic> &#x2264; 0.004), in diagnosing sHCC. Moreover, ChatGPT-4.0 with CEUS LI-RADS outperformed CNN models in predicting sHCC with AUC of 0.86 versus 0.63 (<italic>p</italic> &lt; 0.001). Overall, ChatGPT-4o mini and Google Gemini showed poor intra- and inter-LLM agreement and lower diagnostic efficacy in diagnosing sHCC compared to ChatGPT-4.0 and ChatGPT-4o.</p>
<p>Currently, the primary focus of LLMs in the diagnostic imaging field is on processing text data, though research is now extending these models to multimodal tasks (such as combining image and text processing) (<xref ref-type="bibr" rid="B20">20</xref>, <xref ref-type="bibr" rid="B21">21</xref>). The CEUS LI-RADS released by ACR provides a diagnostic framework for assessing the risk of HCC in patients at risk. However, imaging early-stage HCC, particularly lesions under 2 cm is challenging (<xref ref-type="bibr" rid="B22">22</xref>). We previously determined that CEUS LI-RADS effectively characterizes sFLLs (<xref ref-type="bibr" rid="B12">12</xref>), while the interaction between LLMs and CEUS LI-RADS in diagnosing liver nodules, especially sFLLs, remains unexplored. By removing spatiotemporal interference factors, we found that ChatGPT-4.0 and ChatGPT-4o achieved superior repeatability in CEUS LI-RADS categorization among the four LLMs. GPT-4o mini is characterized by faster processing and superior intelligence compared to ChatGPT-3.5. However, similar to Google Gemini, it demonstrates poor reproducibility in CEUS LI-RADS classification. This is of considerable significance because the stable and reliable grasp of the CEUS LI-RADS system by LLMs could potentially establish a foundation for their clinical diagnostic applications.</p>
<p>Notably, human-LLM interaction with ChatGPT-4.0 outperformed the radiologists who generated the original structured CEUS LI-RADS reports in sensitivity and accuracy in diagnosis sHCC. Interestingly, we observed that, 15.8% (43 of 272) of sHCC were classified as LR-M by human readers, and of these, 88.4% (38 of 43) were assigned to LR-5 by ChatGPT-4.0. Previous studies have shown that early washout within 60 seconds&#x2014;an essential LR-M feature&#x2014;is the main factor causing many HCC cases to be classified as LR-M (<xref ref-type="bibr" rid="B23">23</xref>&#x2013;<xref ref-type="bibr" rid="B25">25</xref>). In the study by Zheng et&#xa0;al, the investigators found that of 354 LR-M nodules, 224 (63%) were HCC (<xref ref-type="bibr" rid="B23">23</xref>). By recategorizing nodules displaying early washout and without punched-out before 5 minutes into LR-5, the sensitivity could elevate from 75% (1141 of 1513) to 85% (1283 of 1513) (P &lt;.001), and accuracy from 81% to 87% (P &lt;.001). Although we have demonstrated that ChatGPT-4.0 understands the CEUS LI-RADS system and recognizes early washout as a typical feature of LR-M, it continues to classify nodules showing hyper-enhancement in arterial phase followed by early washout as LR-5. ChatGPT-4.0 seems to incorporate recent research and does not strictly classify a case as LR-M if washout occurs within 60 seconds. This could be the core reason why ChatGPT-4.0 demonstrated higher sensitivity compared to human readers using CEUS LI-RADS criteria. Considering the &#x201c;black box&#x201d; nature of LLMs, ongoing efforts are essential to address issues such as clearly explaining how decisions are derived. The aforementioned finding highlights the valuable role of LLMs in enhancing the accuracy of diagnoses made by radiologists, not only for senior ultrasound practitioners but also for senior practitioners to make more comprehensive judgements.</p>
<p>Despite the innovative application of LLM in diagnosing sHCC with CEUS LI-RADS, our study has some limitations. First, the LLM task relied only on structured CEUS LI-RADS reports, limiting access to full imaging and clinical information about the patients. Second, the sample sizes for certain CEUS LI-RADS categories, especially LR-2, were small, likely due to the infrequent use of CEUS for liver nodules under 1 cm in routine practice. Third, our study focused on &#x2264;2 cm sFLLs in high-risk patients, which limited the number of participants, and the lack of sufficient follow-up led to additional exclusions. The COVID-19 pandemic and related control measures further reduced patient enrollment, with only 5, 11, and 15 patients being included in 2020, 2021, and 2022, respectively.</p>
<p>In conclusion, large language models (LLMs) showed significant potential in diagnosing small hepatocellular carcinoma (sHCC) in high-risk patients when integrated with the CEUS LI-RADS. ChatGPT-4.0 and ChatGPT-4o demonstrated satisfactory reproducibility with ChatGPT-4.0 outperforming ChatGPT-4o, ChatGPT-4o Mini, and Google Gemini in diagnostic efficacy. It is worth noting that ChatGPT-4.0 identified the &#x2018;early washout&#x2019; feature would not rule out LR-5, which may be the core reason for its superior sensitivity and accuracy in detecting sHCC compared to human readers using CEUS LI-RADS. This highlights the need for continuous data review, model refinement, and improved transparency and explainability of LLM decision-making.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>. Further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>JH: Methodology, Resources, Software, Writing &#x2013; original draft. RY: Resources, Software, Writing &#x2013; original draft. XH: Data curation, Investigation, Writing &#x2013; review &amp; editing. KZ: Formal analysis, Writing &#x2013; review &amp; editing. YL: Formal analysis, Writing &#x2013; review &amp; editing. JL: Resources, Software, Writing &#x2013; review &amp; editing. AL: Supervision, Validation, Writing &#x2013; review &amp; editing. QL: Conceptualization, Funding acquisition, Supervision, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This study was supported by the National Natural Science Foundation of China (Grant No. 82171952).</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We extend our sincere gratitude to Mr. Xin Zhang, M.D. for his statistical analysis and construction of CNN models in this study. <xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4</bold>
</xref> and <xref ref-type="fig" rid="f5">
<bold>5</bold>
</xref> are prompt responses generated by ChatGPT-4o (OpenAI; <ext-link ext-link-type="uri" xlink:href="https://platform.openai.com/docs/models/gpt-4o/">https://platform.openai.com/docs/models/gpt-4o/</ext-link>), ChatGPT-4.0 (OpenAI; <ext-link ext-link-type="uri" xlink:href="https://platform.openai.com/docs/models/gpt-4-and-gpt-4-turbo/">https://platform.openai.com/docs/models/gpt-4-and-gpt-4-turbo/</ext-link>), ChatGPT-4o mini (OpenAI; <ext-link ext-link-type="uri" xlink:href="https://platform.openai.com/docs/models/gpt-4o-mini/">https://platform.openai.com/docs/models/gpt-4o-mini/</ext-link>), and Google Gemini (Google; <ext-link ext-link-type="uri" xlink:href="https://gemini.google.com/">https://gemini.google.com/</ext-link>). We used ChatGPT-4.0 to refine the language and proofread the manuscript. All authors assume full responsibility for the publication&#x2019;s content.</p>
</ack>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="sa" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The authors declare that Generative AI was used in the creation of this manuscript. In this study, we used LLMs to render prompt responses for CEUS LI-RADS categorization. <xref ref-type="fig" rid="f4"><bold>Figures 4</bold></xref>, <xref ref-type="fig" rid="f5"><bold>5</bold></xref> show the prompt responses generated by ChatGPT-4o (OpenAI; https://platform.openai.com/docs/models/gpt-4o/), ChatGPT-4.0 (OpenAI; https://platform.openai.com/docs/models/gpt-4-and-gpt-4-turbo/), ChatGPT-4o mini (OpenAI; <uri xlink:href="https://platform.openai.com/docs/models/gpt-4o-mini/">https://platform.openai.com/docs/models/gpt-4o-mini/</uri>), and Google Gemini (Google; https://gemini.google.com/). We used ChatGPT-4.0 to refine the language and proofread the manuscript.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fonc.2024.1513608/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fonc.2024.1513608/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singal</surname> <given-names>AG</given-names>
</name>
<name>
<surname>Kanwal</surname> <given-names>F</given-names>
</name>
<name>
<surname>Llovet</surname> <given-names>JM</given-names>
</name>
</person-group>. <article-title>Global trends in hepatocellular carcinoma epidemiology: implications for screening, prevention and therapy</article-title>. <source>Nat Rev Clin Oncol</source>. (<year>2023</year>) <volume>20</volume>:<page-range>864&#x2013;84</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41571-023-00825-3</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Quaglia</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Hepatocellular carcinoma: a review of diagnostic challenges for the pathologist</article-title>. <source>J Hepatocell Carcinoma</source>. (<year>2018</year>) <volume>5</volume>:<fpage>99</fpage>&#x2013;<lpage>108</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2147/JHC.S159808</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reig</surname> <given-names>M</given-names>
</name>
<name>
<surname>Forner</surname> <given-names>A</given-names>
</name>
<name>
<surname>Rimola</surname> <given-names>J</given-names>
</name>
<name>
<surname>F&#xe0;brega</surname> <given-names>JF</given-names>
</name>
<name>
<surname>Burrel</surname> <given-names>M</given-names>
</name>
<name>
<surname>Criadoet</surname> <given-names>AG</given-names>
</name>
<etal/>
</person-group>. <article-title>BCLC strategy for prognosis prediction and treatment recommendation: The 2022 update</article-title>. <source>J Hepatol</source>. (<year>2022</year>) <volume>76</volume>:<page-range>681&#x2013;93</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jhep.2021.11.018</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wilson</surname> <given-names>SR</given-names>
</name>
<name>
<surname>Lyshchik</surname> <given-names>A</given-names>
</name>
<name>
<surname>Piscaglia</surname>
</name>
<name>
<surname>Cosgrove</surname> <given-names>D</given-names>
</name>
<name>
<surname>Jang</surname> <given-names>HJ</given-names>
</name>
<name>
<surname>Sirlin</surname> <given-names>C</given-names>
</name>
<etal/>
</person-group>. <article-title>CEUS LI-RADS: algorithm, implementation, and key differences from CT/MRI</article-title>. <source>Abdom Radiol (NY)</source>. (<year>2018</year>) <volume>43</volume>:<page-range>127&#x2013;42</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00261-017-1250-0</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Piscaglia</surname> <given-names>F</given-names>
</name>
<name>
<surname>Wilson</surname> <given-names>SR</given-names>
</name>
<name>
<surname>Lyshchik</surname>
</name>
<name>
<surname>Cosgrove</surname> <given-names>D</given-names>
</name>
<name>
<surname>Dietrich</surname> <given-names>CF</given-names>
</name>
<name>
<surname>Jang</surname> <given-names>HJ</given-names>
</name>
<etal/>
</person-group>. <article-title>American College of Radiology Contrast Enhanced Ultrasound Liver Imaging Reporting and Data System (CEUS LI-RADS) for the diagnosis of Hepatocellular Carcinoma: a pictorial essay</article-title>. <source>Ultraschall Med</source>. (<year>2017</year>) <volume>38</volume>:<page-range>320&#x2013;4</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1055/s-0042-124661</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gu</surname> <given-names>K</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>JH</given-names>
</name>
<name>
<surname>Shin</surname> <given-names>J</given-names>
</name>
<name>
<surname>Hwang</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Min</surname> <given-names>JH</given-names>
</name>
<name>
<surname>Jeong</surname> <given-names>WK</given-names>
</name>
<etal/>
</person-group>. <article-title>Using GPT-4 for LI-RADS feature extraction and categorization with multilingual free-text reports</article-title>. <source>Liver Int</source>. (<year>2024</year>) <volume>44</volume>:<page-range>1578&#x2013;87</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/liv.15891</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>SH</given-names>
</name>
<name>
<surname>Tong</surname> <given-names>WJ</given-names>
</name>
<name>
<surname>Li</surname> <given-names>MD</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>HT</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>XZ</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>ZR</given-names>
</name>
<etal/>
</person-group>. <article-title>Collaborative enhancement of consistency and accuracy in US diagnosis of thyroid nodules using large language models</article-title>. <source>Radiology</source>. (<year>2024</year>) <volume>310</volume>:<fpage>e232255</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1148/radiol.232255</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhayana</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>Chatbots and large language models in radiology: A practical primer for clinical and research applications</article-title>. <source>Radiology</source>. (<year>2024</year>) <volume>310</volume>:<fpage>e232756</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1148/radiol.232756</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adams</surname> <given-names>LC</given-names>
</name>
<name>
<surname>Truhn</surname> <given-names>D</given-names>
</name>
<name>
<surname>Busch</surname> <given-names>F</given-names>
</name>
<name>
<surname>Kader</surname> <given-names>A</given-names>
</name>
<name>
<surname>Niehues</surname> <given-names>SM</given-names>
</name>
<name>
<surname>Makowski</surname> <given-names>MR</given-names>
</name>
<etal/>
</person-group>. <article-title>Leveraging GPT-4 for <italic>post hoc</italic> transformation of free-text radiology reports into structured reporting: A multilingual feasibility study</article-title>. <source>Radiology</source>. (<year>2023</year>) <volume>307</volume>:<fpage>e230725</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1148/radiol.230725</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhayana</surname> <given-names>R</given-names>
</name>
<name>
<surname>Krishna</surname> <given-names>S</given-names>
</name>
<name>
<surname>Bleakney</surname> <given-names>RR</given-names>
</name>
</person-group>. <article-title>Performance of chatGPT on a radiology board-style examination: insights into current strengths and limitations</article-title>. <source>Radiology</source>. (<year>2023</year>) <volume>307</volume>:<fpage>e230582</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1148/radiol.230582</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cozzi</surname> <given-names>A</given-names>
</name>
<name>
<surname>Pinker</surname> <given-names>K</given-names>
</name>
<name>
<surname>Hidber</surname>
</name>
<name>
<surname>Zhang</surname> <given-names>TY</given-names>
</name>
<name>
<surname>Bonomo</surname> <given-names>L</given-names>
</name>
<name>
<surname>Gullo</surname> <given-names>RL</given-names>
</name>
<etal/>
</person-group>. <article-title>BI-RADS category assignments by GPT-3.5, GPT-4, and google bard: A multilanguage study</article-title>. <source>Radiology</source>. (<year>2024</year>) <volume>311</volume>:<fpage>e232133</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1148/radiol.232133</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>JY</given-names>
</name>
<name>
<surname>Li</surname> <given-names>JW</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>L</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>YJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Diagnostic Accuracy of CEUS LI-RADS for the Characterization of Liver Nodules 20 mm or Smaller in Patients at Risk for Hepatocellular Carcinoma</article-title>. <source>Radiology</source>. (<year>2020</year>) <volume>294</volume>:<page-range>329&#x2013;39</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1148/radiol.2019191086</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bruix</surname> <given-names>J</given-names>
</name>
<name>
<surname>Sherman</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Management of hepatocellular carcinoma</article-title>. <source>Hepatology</source>. (<year>2005</year>) <volume>42</volume>:<page-range>1208&#x2013;36</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/hep.20933</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="web">
<person-group person-group-type="author">
<collab>GPT-4o mini</collab>
</person-group>. <source>OpenAI</source>. Available online at: <uri xlink:href="https://platform.openai.com/docs/models/gpt-4o-mini/">https://platform.openai.com/docs/models/gpt-4o-mini/</uri> (Accessed July 5-16, 2024).</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="web">
<person-group person-group-type="author">
<collab>GPT-4o</collab>
</person-group>. <source>OpenAI</source>. Available online at: <uri xlink:href="https://platform.openai.com/docs/models/gpt-4o/">https://platform.openai.com/docs/models/gpt-4o/</uri> (Accessed July 5-16, 2024).</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="web">
<person-group person-group-type="author">
<collab>GPT-4</collab>
</person-group>. <source>OpenAI</source>. Available online at: <uri xlink:href="https://platform.openai.com/docs/models/gpt-4-and-gpt-4-turbo/">https://platform.openai.com/docs/models/gpt-4-and-gpt-4-turbo/</uri> (Accessed July 5-16, 2024).</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="web">
<person-group person-group-type="author">
<collab>Google Genimi</collab>
</person-group>. Available online at: <uri xlink:href="https://gemini.google.com/">https://gemini.google.com/</uri> (Accessed July 5-16, 2024).</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname> <given-names>S</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Lv</surname> <given-names>W</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>LZ</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>Deep learning-based artificial intelligence model to assist thyroid nodule diagnosis and management: a multicentre diagnostic study</article-title>. <source>Lancet Digit Health</source>. (<year>2021</year>) <volume>3</volume>:<page-range>e250&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S2589-7500(21)00041-8</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khanna</surname> <given-names>G</given-names>
</name>
<name>
<surname>Chavhan</surname> <given-names>GB</given-names>
</name>
<name>
<surname>Schooler</surname> <given-names>GR</given-names>
</name>
<name>
<surname>Fraum</surname> <given-names>TJ</given-names>
</name>
<name>
<surname>Alazraki</surname> <given-names>AL</given-names>
</name>
<name>
<surname>Squires</surname> <given-names>HJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Diagnostic performance of LI-RADS version 2018 for evaluation of pediatric hepatocellular carcinoma</article-title>. <source>Radiology</source>. (<year>2021</year>) <volume>299</volume>:<page-range>190&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1148/radiol.2021203559</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>HT</given-names>
</name>
<name>
<surname>Li</surname> <given-names>CY</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>QY</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>YJ</given-names>
</name>
</person-group>. <source>Visual instruction tuning</source>. (<year>2023</year>), arXiv:2304.08485. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2304.08485</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Awadalla</surname> <given-names>A</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>I</given-names>
</name>
<name>
<surname>Gardner</surname> <given-names>J</given-names>
</name>
<name>
<surname>Hessel</surname> <given-names>J</given-names>
</name>
<name>
<surname>Hanafy</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>WR</given-names>
</name>
<etal/>
</person-group>. <source>Openflamingo: An open-source framework for training large autoregressive vision-language models</source>. (<year>2023</year>), arXiv:2308.01390. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2304.08485</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nakashima</surname> <given-names>O</given-names>
</name>
<name>
<surname>Sugihara</surname> <given-names>S</given-names>
</name>
<name>
<surname>Kage</surname> <given-names>M</given-names>
</name>
<name>
<surname>Kojiro</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Pathomorphologic characteristics of small hepatocellular carcinoma a: a special reference to small hepatocellular carcinoma with indistinct margins</article-title>. <source>Hepatology</source>. (<year>1995</year>) <volume>22</volume>:<page-range>101&#x2013;5</page-range>.</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname> <given-names>W</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>XB</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>JW</given-names>
</name>
<name>
<surname>Han</surname> <given-names>F</given-names>
</name>
<name>
<surname>Li</surname> <given-names>F</given-names>
</name>
<etal/>
</person-group>. <article-title>Evaluation of contrast-enhanced US LI-RADS version 2017: application on 2020 liver nodules in patients with hepatitis B infection</article-title>. <source>Radiology</source>. (<year>2019</year>) <volume>294</volume>:<fpage>299</fpage>&#x2013;<lpage>307</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1148/radiol.2019190878</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>JY</given-names>
</name>
<name>
<surname>Li</surname> <given-names>JW</given-names>
</name>
<name>
<surname>Ling</surname> <given-names>WW</given-names>
</name>
<name>
<surname>Li</surname> <given-names>T</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>JB</given-names>
</name>
<etal/>
</person-group>. <article-title>Can contrast enhanced ultrasound differentiate intrahepatic cholangiocarcinoma from hepatocellular carcinoma</article-title>. <source>World J Gastroenterol</source>. (<year>2020</year>) <volume>26</volume>:<page-range>3938&#x2013;51</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3748/wjg.v26.i27.3938</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>F</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Han</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>W</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>YN</given-names>
</name>
<etal/>
</person-group>. <article-title>Distinguishing intrahepatic cholangiocarcinoma from hepatocellular carcinoma in patients with and without risks: the evaluation of the LR-M criteria of contrast-enhanced ultrasound liver imaging reporting and data system version 2017</article-title>. <source>Eur Radiol</source>. (<year>2020</year>) <volume>30</volume>:<page-range>461&#x2013;70</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00330-019-06317-2</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>