<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Oncol.</journal-id>
<journal-title>Frontiers in Oncology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Oncol.</abbrev-journal-title>
<issn pub-type="epub">2234-943X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fonc.2025.1613818</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Oncology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Performance of large language models in the differential diagnosis of benign and malignant biliary stricture</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Kang</surname>
<given-names>Chenxi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3028656/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Jing</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yang</surname>
<given-names>Xintian</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1226637/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ren</surname>
<given-names>Gui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Linhui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Wei</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Xin</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Lei</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Shang</surname>
<given-names>Guochen</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2425811/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hong</surname>
<given-names>Jianglong</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wan</surname>
<given-names>Bingnian</given-names>
</name>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Du</surname>
<given-names>Yu</given-names>
</name>
<xref ref-type="aff" rid="aff8">
<sup>8</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zeng</surname>
<given-names>Wei</given-names>
</name>
<xref ref-type="aff" rid="aff9">
<sup>9</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Yaling</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Tongxin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2057675/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lou</surname>
<given-names>Lijun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Luo</surname>
<given-names>Hui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liang</surname>
<given-names>Shuhui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Lv</surname>
<given-names>Yong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1610510/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Pan</surname>
<given-names>Yanglin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Xijing Hospital of Digestive Diseases, Air Force Medical University</institution>, <addr-line>Xi&#x2019;an</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Gastroenterology, People&#x2019;s Liberation Army Joint Logistics Support Force 940th Hospital</institution>, <addr-line>Lanzhou, Gansu</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Gastroenterology, Third People&#x2019;s Hospital of Gansu Province</institution>, <addr-line>Lanzhou, Gansu</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Gastroenterology, Ankang Traditional Chinese Medicine Hospital</institution>, <addr-line>Ankang</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Tongji Medical College, Huazhong University of Science and Technology</institution>, <addr-line>Wuhan, Hubei</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>First Affiliated Hospital of Anhui Medical University</institution>, <addr-line>Hefei, Anhui</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>Yantai Ludong Hospital, Shandong Provincial Hospital Group</institution>, <addr-line>Yantai, Shandong</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff8">
<sup>8</sup>
<institution>Department of Gastroenterology, Qinzhou Second People&#x2019;s Hospital</institution>, <addr-line>Qinzhou</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff9">
<sup>9</sup>
<institution>Xiang&#x2019;an Hospital, Xiamen University</institution>, <addr-line>Xiamen, Fujian</addr-line>,&#xa0;<country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Xin-Rong Yang, Fudan University, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Lei Xu, The First Affiliated Hospital of Xi&#x2019;an Jiaotong University, China</p>
<p>Balu Bhasuran, Florida State University, United States</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Yong Lv, <email xlink:href="mailto:lvyongxj@yeah.net">lvyongxj@yeah.net</email>; Yanglin Pan, <email xlink:href="mailto:yanglinpan@hotmail.com">yanglinpan@hotmail.com</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>07</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>15</volume>
<elocation-id>1613818</elocation-id>
<history>
<date date-type="received">
<day>17</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>06</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Kang, Li, Yang, Ren, Zhang, Wang, Liu, Wang, Shang, Hong, Wan, Du, Zeng, Liu, Li, Lou, Luo, Liang, Lv and Pan</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Kang, Li, Yang, Ren, Zhang, Wang, Liu, Wang, Shang, Hong, Wan, Du, Zeng, Liu, Li, Lou, Luo, Liang, Lv and Pan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Distinguishing benign from malignant biliary strictures remains challenging. Large Language Models (LLMs) show promise in enhancing diagnostic accuracy. This study aimed to evaluate the performances of ten LLMs in the differential diagnosis of benign and malignant biliary strictures.</p>
</sec>
<sec>
<title>Methods</title>
<p>Consecutive patients with biliary strictures undergoing endoscopic retrograde cholangiopancreatography (ERCP) at Xijing Hospital between January and December 2024 were retrospectively analyzed. Ten LLMs were systematically prompted with standardized clinical, laboratory, and imaging data. Performance was compared against tumor markers (CA19-9, CEA), a new multivariable clinical model, and ten independent pancreaticobiliary exoerienced physicians. Subgroup analyses assessed hilar (n=29) versus non-hilar strictures. Gold-standard diagnosis relied on histopathology and &#x2265;3-month follow-up.</p>
</sec>
<sec>
<title>Results</title>
<p>Among the 159 included patients (83 benign, 76 malignant), four LLMs (Kimi, Deepseek-R1, Claude-3.5S, Llama-3.1), the clinical model (AUC:0.83), and six physicians achieved &gt;80% accuracy. Kimi demonstrated superior accuracy (87%), significantly outperforming 70% of physicians (7/10, p&lt;0.01). Three other LLMs (Deepseek-R1:83%, Claude-3.5S:82%, Llama-3.1:81%) and the clinical model performed comparably to physicians (78-84%, p&gt;0.05), collectively surpassing tumor markers (CA19&#x2013;9 accuracy:66%, CEA:71%). Physicians demonstrated higher accuracy for hilar strictures (87% vs. 79% for non-hilar, p&lt;0.001). LLMs showed similar performance across stricture locations (hilar:64-95%; non-hilar:62-88%, p&gt;0.05). For hilar strictures, 7/10 physicians achieved significantly higher accuracy (87-90%) than 8/10 LLMs (64-84%, p&lt;0.05).</p>
</sec>
<sec>
<title>Conclusions</title>
<p>Using clinical, lab, and imaging data, some LLMs achieved diagnostic accuracy comparable to or exceeding clinical models and experienced physicians for differentiating benign versus malignant strictures. However, for hilar strictures, LLM performance was inferior to over half of the physicians.</p>
</sec>
</abstract>
<kwd-group>
<kwd>large language model</kwd>
<kwd>biliary stricture</kwd>
<kwd>cholangiocarcinoma</kwd>
<kwd>prediction model</kwd>
<kwd>diagnosis</kwd>
</kwd-group>
<contract-num rid="cn001">2022YFC25051002022YFC2505100 2022YFC2505100</contract-num>
<contract-num rid="cn002">8237311782373117, 8237061982370619</contract-num>
<contract-sponsor id="cn001">National Key Research and Development Program of China<named-content content-type="fundref-id">10.13039/501100012166</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content>
</contract-sponsor>
<counts>
<fig-count count="4"/>
<table-count count="3"/>
<equation-count count="0"/>
<ref-count count="47"/>
<page-count count="13"/>
<word-count count="5466"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Gastrointestinal Cancers: Hepato Pancreatic Biliary Cancers</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<title>Introduction</title>
<p>Biliary strictures, characterized by abnormal bile duct narrowing, can significantly obstruct bile flow. An estimated 54-87% of biliary strictures are malignant (<xref ref-type="bibr" rid="B1">1</xref>&#x2013;<xref ref-type="bibr" rid="B5">5</xref>), arising from local or metastatic cancers. Benign biliary strictures have heterogeneous etiologies including surgical bile duct injury, chronic pancreatitis, or chronic cholangiopathies (e.g., primary sclerosing cholangitis) (<xref ref-type="bibr" rid="B6">6</xref>). Benign and malignant strictures differ significantly in management and prognosis. Benign strictures are typically managed by endoscopic dilation, stenting, or surgery (<xref ref-type="bibr" rid="B7">7</xref>&#x2013;<xref ref-type="bibr" rid="B9">9</xref>), while malignant strictures require aggressive approaches including surgical resection, palliative drainage, and systemic therapy (<xref ref-type="bibr" rid="B10">10</xref>&#x2013;<xref ref-type="bibr" rid="B12">12</xref>). Thus, accurately distinguishing between them is crucial for guiding treatment and prognostic assessment.</p>
<p>Characteristics of biliary strictures are typically assessed via brushing cytology, forceps biopsy, or cholangioscopic biopsy during endoscopic retrograde cholangiopancreatography (ERCP) (<xref ref-type="bibr" rid="B13">13</xref>). Brush cytology and forceps biopsy demonstrate diagnostic accuracies of 15&#x2013;80% (<xref ref-type="bibr" rid="B14">14</xref>, <xref ref-type="bibr" rid="B15">15</xref>), while cholangioscopic biopsy achieves 70&#x2013;87% (<xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B16">16</xref>). Endoscopic ultrasound-guided (EUS) fine needle aspiration (FNA) or fine needle biopsy (FNB) shows favorable accuracy, particularly for extrinsic mass-related strictures (<xref ref-type="bibr" rid="B1">1</xref>). Advanced techniques like intraductal ultrasonography (IDUS) (<xref ref-type="bibr" rid="B17">17</xref>), probe-based confocal laser endomicroscopy (pCLE) (<xref ref-type="bibr" rid="B18">18</xref>), and optical coherence tomography (OCT) (<xref ref-type="bibr" rid="B19">19</xref>), offer enhanced precision but are limited by invasiveness, cost, and availability.</p>
<p>Significant differences also exist in common clinical parameters between benign and malignant strictures, including age of onset, duration of liver function abnormalities, previous surgeries, and tumor markers. These noninvasive, accessible parameters facilitate convenient prediction. Wang et&#xa0;al. used CA50, CA19-9, and AFP to achieve an AUC of 0.879 (95% CI: 0.841&#x2013;0.917) (<xref ref-type="bibr" rid="B20">20</xref>), while Zhang et&#xa0;al. combined MRI with inflammatory markers for an AUC of 0.802 (95% CI: 0.719&#x2013;0.870) (<xref ref-type="bibr" rid="B18">18</xref>). Though promising, these models require further validation.</p>
<p>LLMs including Deepseek-R1, GPT-4T, Claude-3.5S, and Llama-3.1 show potential in improving diagnostic accuracy across medicine (<xref ref-type="bibr" rid="B21">21</xref>&#x2013;<xref ref-type="bibr" rid="B24">24</xref>). Trained on vast medical data, LLMs may assist preliminary diagnosis by reducing interpretational variability. However, their utility for differentiating biliary strictures&#x2014;a specialized, less common condition&#x2014;remains unknown. We hypothesize that LLMs leveraging common clinical data (manifestations, blood tests, imaging) could aid this differentiation.</p>
<p>In this study, we aimed to evaluate the diagnostic performance of ten distinct LLMs for diagnosing benign versus malignant biliary strictures, comparing performance against tumor markers, a novel clinical model, and ten experienced pancreaticobiliary specialists.</p>
</sec>
<sec id="s2">
<title>Methods</title>
<sec id="s2_1">
<title>Study design</title>
<p>This retrospective study evaluated the diagnostic performance of ten distinct large language models (LLMs) in differentiating between benign and malignant biliary strictures. The study protocol was approved by the Ethics Committee of Xijing Hospital. The written informed consent was obtained from all the patients or their next of kin.</p>
</sec>
<sec id="s2_2">
<title>Patients</title>
<p>Consecutive patients aged &#x2265;18 years admitted to Xijing Hospital for biliary stricture evaluation between January and December 2024 were eligible. Inclusion criteria required a definitive etiological diagnosis confirmed by pathological examination and regular follow-up exceeding 3 months. Patients with incomplete medical records were excluded. Malignancy diagnosis relied on pathological results obtained via endoscopic retrograde cholangiopancreatography (ERCP), percutaneous transhepatic cholangiographic drainage (PTCD), endoscopic ultrasound (EUS), biopsy, or surgery. Benign strictures required confirmation by benign pathology and absence of progression over &#x2265;3 months.</p>
</sec>
<sec id="s2_3">
<title>Data collection</title>
<p>Demographic, clinical, imaging, and pathological data were extracted from electronic medical records. An independent physician standardized data inputs for ten LLMs, ensuring unbiased differentiation. Case data was structured uniformly, excluding diagnostic conclusions, and included clinical presentation, history, imaging reports (CT, MRI, MRCP), and lab results (complete blood count, liver and renal function tests, lipids, coagulation, tumor/inflammatory markers). Any indications of benignancy or malignancy from the reports were removed. A flowchart of the overall study design is shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>, illustrating the process from patient selection to diagnostic performance evaluation.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Flowchart of overall study design. LLM, Large Language Model; MLM, Machine Learning Model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1613818-g001.tif">
<alt-text content-type="machine-generated">Flowchart illustrating the process of evaluating biliary stricture cases using Large Language Models (LLMs), Machine Learning Models (MLMs), and experienced physicians. It includes inclusion and exclusion criteria, preprocessing of cases, and structured steps for LLMs, MLMs, and physicians. The analysis involves diagnostic performance metrics and agreement comparisons between models, measured by Kappa Value. The chart emphasizes standardized input and output collection for consistent analysis.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_4">
<title>Large language models for differential diagnosis of biliary strictures</title>
<p>Ten LLMs were selected for evaluation for reproducibility, including 1) mainstream commercial models with proven medical reasoning capabilities in prior studies (GPT-4T, GPT-4o, Gemini-1.5 pro, Claude-3.5S), 2) models developed by Chinese companies to align with the Chinese-language clinical data (Kimi, ERNIE-4, Qwen-2, GLM-4); 3) open-source models (Deepseek-R1, Llama-3.1) (<xref ref-type="bibr" rid="B25">25</xref>, <xref ref-type="bibr" rid="B26">26</xref>). To ensure consistency and reproducibility, a structured query approach was used. A case-specific prompt simulated consultation with an experienced pancreaticobiliary specialist: &#x201c;Supposing you are an experienced physician specializing in pancreaticobiliary diseases, when encountering a case with biliary stricture, please provide a tentative judgment indicating whether the cause of the stricture is more likely to be malignant or benign.&#x201d; (Detailed prompt provided in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Method S1</bold>
</xref>). For each LLM, a new conversation session was initiated to eliminate contextual carryover. Probabilistic LLM outputs (e.g., &#x201c;likely benign,&#x201d; &#x201c;possibly malignant&#x201d;) were converted into binary outcomes (benign/malignant) using predefined rules. Responses indicating malignancy (e.g., &#x201c;likely malignant,&#x201d; &#x201c;probably malignant,&#x201d; &#x201c;suggestive of malignancy&#x201d;) were classified as &#x201c;malignant.&#x201d; Responses indicating benignancy (e.g., &#x201c;likely benign,&#x201d; &#x201c;probably benign,&#x201d; &#x201c;suggestive of a benign process&#x201d;) were classified as &#x201c;benign.&#x201d; Explicit statements (&#x201c;benign&#x201d;/&#x201d;malignant&#x201d;) were directly categorized.</p>
<p>All queries employed identical deterministic parameters: temperature=0.0 (output consistency), max tokens=10240, and consistent system prompts. Queries were executed in January 2025 using contemporaneous model versions. The evaluated LLMs included: Deepseek-R1, GPT-4T, GPT-4o, Claude-3.5S, Gemini-1.5 Pro, Kimi, Llama-3.1 405B, ERNIE-4.0-Turbo-8K, Qwen-2-72B, and ChatGLM-4-9B (details in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1</bold>
</xref>). Chain-of-thought (CoT) outputs were extracted for reasoning pattern analysis. Cases were categorized by diagnostic accuracy (correct/incorrect), and reasoning traces were independently assessed by two gastroenterologists using predefined clinical logic criteria (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S3</bold>
</xref>).</p>
</sec>
<sec id="s2_5">
<title>Development of a clinical prediction model</title>
<p>A multivariable logistic regression model incorporating key demographic, clinical, laboratory, and imaging parameters was developed. The modeling process initiated with univariable screening to identify candidate variables (p &lt; 0.10), followed by multivariable regression analysis. Variables were retained multivariable models based on statistical significance (p &lt; 0.05) or established clinical relevance for variables approaching significance. Collinearity was assessed using variance inflation factors (VIF) and Spearman correlation coefficients (|&#x3c1;| &gt; 0.6), with clinical importance determining variable selection when collinearity was detected. Bidirectional stepwise selection (forward/backward) using Akaike (AIC) and Bayesian (BIC) information criteria optimized the model. Final model selection prioritized minimized AIC/BIC and maximized predictive performance, with internal validation implemented through bootstrapping (1,000 resamples). CA19&#x2013;9 and CEA were separately assessed as independent diagnostic markers.</p>
</sec>
<sec id="s2_6">
<title>Experienced physicians&#x2019; evaluation</title>
<p>Ten experienced pancreaticobiliary specialists (each with &#x2265;10 years of clinical practice) independently evaluated comprehensive clinical summaries identical to those processed by the LLMs. All diagnostic predictions regarding benign or malignant status were made without intercommunication among physicians to ensure independent assessment.</p>
</sec>
<sec id="s2_7">
<title>Outcome</title>
<p>The primary outcome was diagnostic accuracy that was calculated as the average of sensitivity and specificity to account for class imbalance (range: 0-1, higher values superior). For LLMs, the highest accuracy from duplicate assessments was selected. Secondary performance metrics included sensitivity (proportion of true positives correctly identified), specificity (proportion of true negatives correctly identified), positive predictive value (PPV, proportion of positive predictions that were correct), negative predictive value (NPV, proportion of negative predictions that were correct), and F1-score (harmonic mean of precision and sensitivity providing balanced assessment).</p>
</sec>
<sec id="s2_8">
<title>Statistical analysis</title>
<p>Continuous variables are presented as mean &#xb1; standard deviation (normally distributed) or median [interquartile range] (non-normal distributions), while categorical variables are expressed as proportions (%) with odds ratios (OR) and 95% confidence intervals (CI) for association analyses. McNemar and DeLong tests were used to compare performance metrics between models. Confidence intervals for accuracy, sensitivity, and specificity were calculated using the Clopper-Pearson method, with proportion differences reported in pairwise comparisons. Subgroup analysis evaluated LLM diagnostic performance by stricture location (hilar vs. non-hilar). Internal concordance of LLMs was assessed through weighted Cohen&#x2019;s kappa (&#x3ba;) with 95% CI based on duplicate assessments, interpreted as: &#x3ba; &#x2264; 0.20 (slight concordance), 0.21-0.40 (fair), 0.41-0.60 (moderate), 0.61-0.80 (substantial), and 0.81-1.00 (almost perfect). Pairwise comparisons of classification accuracy employed McNemar&#x2019;s test with Holm-Bonferroni correction for multiple comparisons (family-wise &#x3b1; = 0.05; significance threshold: adjusted p &lt; 0.05). All statistical tests were two-sided. Statistical analyses were conducted using R version 4.3.1.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>Results</title>
<sec id="s3_1">
<title>Patient demographics and clinical characteristics</title>
<p>During the study period, 270 patients were diagnosed with biliary stricture, of whom 111 were excluded according to inclusion/exclusion criteria, resulting in a final cohort of 159 patients (83 benign, 76 malignant). Baseline demographic and clinical characteristics are summarized in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. Compared with benign stricture patients, those with malignant lesions were older (65.13 vs. 57.19 years), exhibited higher bilirubin (149.14 vs. 78.15 &#x3bc;mol/L, p &lt; 0.001) and CA19&#x2013;9 levels (2888.33 vs. 246.32 U/mL, p = 0.003), a higher proportion of lymph node enlargement (47.4% vs. 19.3%, p &lt; 0.001), and more frequent hilar strictures (26.3% vs. 10.8%, p = 0.020).</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Patient demographics, clinical characteristics, and diagnostic markers.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Variables</th>
<th valign="top" align="left">Overall (n=159)</th>
<th valign="top" align="left">Benign (n=83)</th>
<th valign="top" align="left">Malignant (n=76)</th>
<th valign="top" align="left">P value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Age (year)</td>
<td valign="top" align="left">60.99 (14.03)</td>
<td valign="top" align="left">57.19 (16.22)</td>
<td valign="top" align="left">65.13 (9.70)</td>
<td valign="top" align="left">&lt;0.001</td>
</tr>
<tr>
<td valign="top" align="left">Male, n (%)</td>
<td valign="top" align="left">66 (41.5)</td>
<td valign="top" align="left">35 (42.2)</td>
<td valign="top" align="left">31 (40.8)</td>
<td valign="top" align="left">0.988</td>
</tr>
<tr>
<td valign="top" align="left">BMI (kg/m<sup>2</sup>)</td>
<td valign="top" align="left">20.91 (3.43)</td>
<td valign="top" align="left">21.55 (3.20)</td>
<td valign="top" align="left">20.17 (3.58)</td>
<td valign="top" align="left">0.075</td>
</tr>
<tr>
<td valign="top" align="left">Prior surgical history *, n (%)</td>
<td valign="top" align="left">85 (53.5)</td>
<td valign="top" align="left">51 (61.4)</td>
<td valign="top" align="left">34 (44.7)</td>
<td valign="top" align="left">0.051</td>
</tr>
<tr>
<td valign="top" align="left">Disease duration &lt; 1month &#x2020;, n (%)</td>
<td valign="top" align="left">135 (61.4)</td>
<td valign="top" align="left">69 (57.0)</td>
<td valign="top" align="left">66 (66.7)</td>
<td valign="top" align="left">0.005</td>
</tr>
<tr>
<th valign="top" colspan="5" align="left">Traditional tumor markers</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;CA125 (U/ml)</td>
<td valign="top" align="left">48.36 (131.48)</td>
<td valign="top" align="left">26.23 (51.27)</td>
<td valign="top" align="left">72.53 (180.03)</td>
<td valign="top" align="left">0.026</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;CA19-9 (U/ml)</td>
<td valign="top" align="left">1435.23 (6681.62)</td>
<td valign="top" align="left">246.32 (1854.95)</td>
<td valign="top" align="left">2888.33 (9574.67)</td>
<td valign="top" align="left">0.003</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;CEA (ng/ml)</td>
<td valign="top" align="left">15.00 (77.00)</td>
<td valign="top" align="left">3.22 (3.21)</td>
<td valign="top" align="left">29.40 (113.39)</td>
<td valign="top" align="left">0.012</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;AFP (ng/ml)</td>
<td valign="top" align="left">5.72 (17.88)</td>
<td valign="top" align="left">5.04 (15.03)</td>
<td valign="top" align="left">6.47 (20.62)</td>
<td valign="top" align="left">0.616</td>
</tr>
<tr>
<th valign="top" colspan="5" align="left">Whole blood tests</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;White blood cell (cell/&#x3bc;L)</td>
<td valign="top" align="left">6570 (3050)</td>
<td valign="top" align="left">6410 (2420)</td>
<td valign="top" align="left">6740 (3620)</td>
<td valign="top" align="left">0.491</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Hemoglobin (g/dL)</td>
<td valign="top" align="left">12.20 (2.05)</td>
<td valign="top" align="left">12.75 (1.90)</td>
<td valign="top" align="left">11.67 (2.06)</td>
<td valign="top" align="left">0.002</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Platelet count (&#xd7; 103/&#x3bc;L)</td>
<td valign="top" align="left">214.06 (79.79)</td>
<td valign="top" align="left">206.84 (76.57)</td>
<td valign="top" align="left">221.94 (82.95)</td>
<td valign="top" align="left">0.235</td>
</tr>
<tr>
<th valign="top" colspan="5" align="left">Liver/renal function tests</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Total bilirubin (&#x3bc;mol/L)</td>
<td valign="top" align="left">112.52 (119.36)</td>
<td valign="top" align="left">78.15 (105.85)</td>
<td valign="top" align="left">149.14 (122.78)</td>
<td valign="top" align="left">&lt;0.001</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Albumin (g/dL)</td>
<td valign="top" align="left">3.78 (0.68)</td>
<td valign="top" align="left">3.94 (0.66)</td>
<td valign="top" align="left">3.61 (0.66)</td>
<td valign="top" align="left">0.002</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Creatinine (mg/dL)</td>
<td valign="top" align="left">0.85 (0.18)</td>
<td valign="top" align="left">0.84 (0.18)</td>
<td valign="top" align="left">0.85 (0.17)</td>
<td valign="top" align="left">0.668</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Sodium (mEq/L)</td>
<td valign="top" align="left">141.96 (2.83)</td>
<td valign="top" align="left">141.92 (2.84)</td>
<td valign="top" align="left">142.00 (2.84)</td>
<td valign="top" align="left">0.858</td>
</tr>
<tr>
<th valign="top" colspan="5" align="left">Coagulation profiles</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;International normalized ratio</td>
<td valign="top" align="left">1.09 (0.16)</td>
<td valign="top" align="left">1.10 (0.18)</td>
<td valign="top" align="left">1.09 (0.14)</td>
<td valign="top" align="left">0.605</td>
</tr>
<tr>
<th valign="top" colspan="5" align="left">Lipid profiles</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Total cholesterol (mg/dL)</td>
<td valign="top" align="left">169.50 (41.70)</td>
<td valign="top" align="left">171.81 (44.79)</td>
<td valign="top" align="left">167.18 (37.84)</td>
<td valign="top" align="left">0.477</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Triglyceride (mg/dL)</td>
<td valign="top" align="left">111.50 (86.73)</td>
<td valign="top" align="left">110.62 (73.45)</td>
<td valign="top" align="left">112.39 (100.00)</td>
<td valign="top" align="left">0.859</td>
</tr>
<tr>
<th valign="top" colspan="5" align="left">Inflammatory markers</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;CRP (mg/dL)</td>
<td valign="top" align="left">2.02 (3.18)</td>
<td valign="top" align="left">2.21 (3.48)</td>
<td valign="top" align="left">1.81 (2.83)</td>
<td valign="top" align="left">0.431</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;IL6 (pg/ml)</td>
<td valign="top" align="left">33.43 (363.27)</td>
<td valign="top" align="left">3.20 (2.72)</td>
<td valign="top" align="left">66.45 (525.24)</td>
<td valign="top" align="left">0.274</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;PCT (ng/ml)</td>
<td valign="top" align="left">1.14 (4.06)</td>
<td valign="top" align="left">0.60 (1.79)</td>
<td valign="top" align="left">1.73 (5.53)</td>
<td valign="top" align="left">0.08</td>
</tr>
<tr>
<td valign="top" align="left">Immune abnormalities&#x2021;, n (%)</td>
<td valign="top" align="left">26 (16.4)</td>
<td valign="top" align="left">11 (13.3)</td>
<td valign="top" align="left">15 (19.7)</td>
<td valign="top" align="left">0.174</td>
</tr>
<tr>
<td valign="top" align="left">Lymph node enlargement, n (%)</td>
<td valign="top" align="left">52 (32.7)</td>
<td valign="top" align="left">16 (19.3)</td>
<td valign="top" align="left">36 (47.4)</td>
<td valign="top" align="left">&lt;0.001</td>
</tr>
<tr>
<td valign="top" align="left">Biliary stricture sites, n (%)</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left">0.020</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Hilar</td>
<td valign="top" align="left">29 (18.2)</td>
<td valign="top" align="left">9 (10.8)</td>
<td valign="top" align="left">20 (26.3)</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Non-hilar</td>
<td valign="top" align="left">130 (81.8)</td>
<td valign="top" align="left">74 (89.2)</td>
<td valign="top" align="left">56 (73.7)</td>
<td valign="top" align="left"/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Data are mean (standard deviation) or numbers (percentages) unless otherwise specified. BMI, Body Mass Index; CA19-9, Carbohydrate Antigen 199; CEA, Carcinoembryonic Antigen; AFP, Alpha-Fetoprotein; CA125, Carbohydrate Antigen 125; CRP, C-Reactive Protein; IL6, Interleukin6; PCT, Procalcitonin.</p>
</fn>
<fn>
<p>*Surgical history was defined as a history of liver transplantation and biliary surgery.</p>
</fn>
<fn>
<p>&#x2020;Disease duration &lt; 1 month was defined as a period less than one month from the onset of symptoms.</p>
</fn>
<fn>
<p>&#x2021;Immune abnormalities were defined as the presence of IgG4 subclass abnormalities, autoimmune diseases, or abnormalities detected in a series of autoantibody tests.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_2">
<title>Clinical data-driven diagnostic model performance</title>
<p>A stepwise multivariable logistic regression was employed to develop the diagnostic model, with the final model retaining age, CA19-9, CEA, disease duration &lt;1 month, C-reactive protein (CRP), surgical history, and lymph node enlargement (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>). <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref> illustrates the impact of hyperparameter &#x3bb; on model accuracy under L2 regularization, with the optimal &#x3bb; value of 149.3 determined via 10-fold cross-validation to maximize test-set accuracy. The model achieved an AUC of 0.83 (95% CI: 0.70&#x2013;0.96), accuracy of 0.83, sensitivity of 0.83, and specificity of 0.82 (<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>, <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>), outperforming tumor markers alone. CA19&#x2013;9 showed an AUC of 0.77 (95% CI: 0.69&#x2013;0.84), accuracy of 0.66, specificity of 0.93, and sensitivity of 0.39; CEA yielded an AUC of 0.66 (95% CI: 0.58&#x2013;0.75), accuracy of 0.71, specificity of 0.59, and sensitivity of 0.83. Internal validation via bootstrapping (1000 iterations) confirmed consistent AUC of 0.83 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S2</bold>
</xref>).</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Selection of variables based on univariate and multivariate logistic regression analysis.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" rowspan="2" align="left">Variables</th>
<th valign="top" colspan="2" align="left">Univariate Logistic Regression</th>
<th valign="top" colspan="2" align="left">Multivariate Logistic Regression</th>
</tr>
<tr>
<th valign="top" align="left">OR (95%CI)</th>
<th valign="top" align="left">P</th>
<th valign="top" align="left">OR (95%CI)</th>
<th valign="top" align="left">P</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Age &gt; 55 (year)</td>
<td valign="top" align="left">4.53 (2.09, 9.8)</td>
<td valign="top" align="left">&lt;0.001</td>
<td valign="top" align="left">4.68 (1.65, 14.51) &#xa7;</td>
<td valign="top" align="left">0.005</td>
</tr>
<tr>
<td valign="top" align="left">Male</td>
<td valign="top" align="left">1.06 (0.56, 1.99)</td>
<td valign="top" align="left">0.86</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">Prior Surgical history *</td>
<td valign="top" align="left">0.51 (0.27, 0.96)</td>
<td valign="top" align="left">0.04</td>
<td valign="top" align="left">0.74 (0.27, 1.99) &#xa7;1</td>
<td valign="top" align="left">0.549</td>
</tr>
<tr>
<td valign="top" align="left">Disease duration &lt; 1mouth &#x2020;</td>
<td valign="top" align="left">2.79 (1.4, 5.57)</td>
<td valign="top" align="left">&lt;0.001</td>
<td valign="top" align="left">3.48 (1.32, 9.79) &#xa7;</td>
<td valign="top" align="left">0.014</td>
</tr>
<tr>
<th valign="top" colspan="5" align="left">Traditional tumor markers</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;CA125 &gt; 20 (U/ml)</td>
<td valign="top" align="left">3.79 (1.95, 7.35)</td>
<td valign="top" align="left">&lt;0.001</td>
<td valign="top" align="left">1.74 (0.64, 4.81)</td>
<td valign="top" align="left">0.278</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;CA19-9 &gt; 30(U/ml)</td>
<td valign="top" align="left">6.98 (3.33, 14.64)</td>
<td valign="top" align="left">&lt;0.001</td>
<td valign="top" align="left">5.20 (1.81, 16.16) &#xa7;</td>
<td valign="top" align="left">0.003</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;CEA &gt; 5 (ng/ml)</td>
<td valign="top" align="left">8.37 (3.24, 21.63)</td>
<td valign="top" align="left">&lt;0.001</td>
<td valign="top" align="left">4.99 (1.58, 18.14) &#xa7;</td>
<td valign="top" align="left">0.009</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;AFP &gt; 4 (ng/ml)</td>
<td valign="top" align="left">0.44 (0.23, 0.87)</td>
<td valign="top" align="left">0.02</td>
<td valign="top" align="left">1.32 (0.48, 3.73)</td>
<td valign="top" align="left">0.596</td>
</tr>
<tr>
<th valign="top" colspan="5" align="left">Whole blood tests</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;White blood cell &gt; 6000 (cell/&#x3bc;L)</td>
<td valign="top" align="left">0.52 (0.28, 0.99)</td>
<td valign="top" align="left">0.05</td>
<td valign="top" align="left">0.53 (0.21, 1.32)</td>
<td valign="top" align="left">0.179</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Hemoglobin &gt; 12.5 (g/dL)</td>
<td valign="top" align="left">0.50 (0.26, 0.96)</td>
<td valign="top" align="left">0.04</td>
<td valign="top" align="left">0.71 (0.30, 2.23)</td>
<td valign="top" align="left">0.487</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Platelet count &gt; 250 (&#xd7; 103/&#x3bc;L)</td>
<td valign="top" align="left">1.51 (0.79, 2.90)</td>
<td valign="top" align="left">0.21</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<th valign="top" colspan="5" align="left">Liver/renal function tests</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Total bilirubin &gt; 2 (mg/dL)</td>
<td valign="top" align="left">5.14 (2.55, 10.39)</td>
<td valign="top" align="left">&lt;0.001</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Albumin &gt; 4 (g/dL)</td>
<td valign="top" align="left">0.38 (0.20, 0.75)</td>
<td valign="top" align="left">&lt;0.001</td>
<td valign="top" align="left">0.53 (0.18, 1.49)</td>
<td valign="top" align="left">0.23</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Creatinine &gt; 0.75 (mg/dL)</td>
<td valign="top" align="left">1.67 (0.82, 3.40)</td>
<td valign="top" align="left">0.15</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Sodium &gt; 140 (mEq/L)</td>
<td valign="top" align="left">1.51 (0.80, 2.85)</td>
<td valign="top" align="left">0.21</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<th valign="top" colspan="5" align="left">Coagulation profiles</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;International normalized ratio &gt; 1</td>
<td valign="top" align="left">2.04 (0.94, 4.47)</td>
<td valign="top" align="left">0.07</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<th valign="top" colspan="5" align="left">Inflammatory markers</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;CRP &gt; 0.6 (mg/dL)</td>
<td valign="top" align="left">0.42 (0.21, 0.82)</td>
<td valign="top" align="left">0.01</td>
<td valign="top" align="left">0.23 (0.08, 0.63)</td>
<td valign="top" align="left">0.006</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;IL6 &gt; 1 (pg/ml)</td>
<td valign="top" align="left">2.52 (1.07, 5.92)</td>
<td valign="top" align="left">0.03</td>
<td valign="top" align="left">2.72 (0.81, 9.81)</td>
<td valign="top" align="left">0.112</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;PCT &gt; 0.1 (ng/ml)</td>
<td valign="top" align="left">1.77 (0.94, 3.33)</td>
<td valign="top" align="left">0.08</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">Immune abnormal &#x2021;</td>
<td valign="top" align="left">1.61 (0.69, 3.76)</td>
<td valign="top" align="left">0.27</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="left">Lymph node enlargement</td>
<td valign="top" align="left">3.77 (1.86, 7.64)</td>
<td valign="top" align="left">&lt;0.001</td>
<td valign="top" align="left">2.90 (0.97, 9.16) &#xa7;</td>
<td valign="top" align="left">0.06</td>
</tr>
<tr>
<td valign="top" align="left">Biliary stricture sites: Hilar</td>
<td valign="top" align="left">0.34 (0.14, 0.80)</td>
<td valign="top" align="left">0.01</td>
<td valign="top" align="left">0.50 (0.13, 1.77)</td>
<td valign="top" align="left">0.29</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>CA19-9, Carbohydrate Antigen 199; CEA, Carcinoembryonic Antigen; AFP, Alpha-Fetoprotein; CA125, Carbohydrate Antigen 125; CRP, C-Reactive Protein; IL6, Interleukin6; PCT, Procalcitonin; OR, odds ratio; CI, Confidence Interval; NA, Not Applicable.</p>
</fn>
<fn>
<p>*Surgical history was defined as a history of liver transplantation and biliary tract surgery.</p>
</fn>
<fn>
<p>&#x2020;Disease duration was defined as a period less than one month from the onset of symptoms.</p>
</fn>
<fn>
<p>&#x2021;Immune abnormal was defined as the presence of immunoglobulin subclass 4 abnormalities, or having an autoimmune disease, or abnormalities in a series of autoantibody tests.</p>
</fn>
<fn>
<p>&#xa7;Variables included in the clinical model.</p>
</fn>
<fn>
<p>&#xa7;1Variables included based on clinical relevance despite borderline univariate p-values (p&lt;0.05).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Performance metrics of LLMs, EPs, clinical model and tumor markers.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Predictors</th>
<th valign="top" align="left">Acc, (95%CI)</th>
<th valign="top" align="left">Sens</th>
<th valign="top" align="left">Spec</th>
<th valign="top" align="left">F1</th>
<th valign="top" align="left">PPV</th>
<th valign="top" align="left">NPV</th>
<th valign="top" align="left">AUC, (95%CI)</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="top" colspan="8" align="left">LLMs</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;GPT-4T</td>
<td valign="top" align="left">0.79 (0.71, 0.84)</td>
<td valign="top" align="left">0.59 (0.51, 0.64)</td>
<td valign="top" align="left">0.99 (0.92, 1.00)</td>
<td valign="top" align="left">0.74</td>
<td valign="top" align="left">0.98</td>
<td valign="top" align="left">0.69</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;GPT-4o</td>
<td valign="top" align="left">0.66 (0.57, 0.72)</td>
<td valign="top" align="left">0.34 (0.30, 0.39)</td>
<td valign="top" align="left">0.99 (0.92, 1.00)</td>
<td valign="top" align="left">0.50</td>
<td valign="top" align="left">0.97</td>
<td valign="top" align="left">0.58</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Claude-3.5S</td>
<td valign="top" align="left">0.82 (0.74, 0.87)</td>
<td valign="top" align="left">0.71 (0.65, 0.76)</td>
<td valign="top" align="left">0.92 (0.88, 0.97)</td>
<td valign="top" align="left">0.80</td>
<td valign="top" align="left">0.91</td>
<td valign="top" align="left">0.74</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Gemini-1.5 pro</td>
<td valign="top" align="left">0.62 (0.52, 0.68)</td>
<td valign="top" align="left">0.25 (0.19, 0.30)</td>
<td valign="top" align="left">0.99 (0.92, 1.00)</td>
<td valign="top" align="left">0.40</td>
<td valign="top" align="left">0.95</td>
<td valign="top" align="left">0.55</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Kimi</td>
<td valign="top" align="left">0.87 (0.81, 0.92)</td>
<td valign="top" align="left">0.83 (0.77, 0.89)</td>
<td valign="top" align="left">0.91 (0.85, 0.96)</td>
<td valign="top" align="left">0.87</td>
<td valign="top" align="left">0.91</td>
<td valign="top" align="left">0.83</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;ERNIE-4</td>
<td valign="top" align="left">0.79 (0.71, 0.85)</td>
<td valign="top" align="left">0.66 (0.61, 0.72)</td>
<td valign="top" align="left">0.92 (0.88, 0.95)</td>
<td valign="top" align="left">0.76</td>
<td valign="top" align="left">0.90</td>
<td valign="top" align="left">0.71</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Llama-3.1</td>
<td valign="top" align="left">0.81 (0.73, 0.86)</td>
<td valign="top" align="left">0.65 (0.60, 0.72)</td>
<td valign="top" align="left">0.97 (0.91, 0.99)</td>
<td valign="top" align="left">0.78</td>
<td valign="top" align="left">0.96</td>
<td valign="top" align="left">0.72</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Qwen-2</td>
<td valign="top" align="left">0.76 (0.68, 0.82)</td>
<td valign="top" align="left">0.65 (0.60, 0.71)</td>
<td valign="top" align="left">0.87 (0.82, 0.92)</td>
<td valign="top" align="left">0.73</td>
<td valign="top" align="left">0.84</td>
<td valign="top" align="left">0.69</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;GLM-4</td>
<td valign="top" align="left">0.77 (0.69, 0.83)</td>
<td valign="top" align="left">0.69 (0.62, 0.75)</td>
<td valign="top" align="left">0.86 (0.82, 0.90)</td>
<td valign="top" align="left">0.75</td>
<td valign="top" align="left">0.84</td>
<td valign="top" align="left">0.71</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Deepseek-R1</td>
<td valign="top" align="left">0.83 (0.76, 0.89)</td>
<td valign="top" align="left">0.81 (0.77, 0.86)</td>
<td valign="top" align="left">0.86 (0.81, 0.91)</td>
<td valign="top" align="left">0.83</td>
<td valign="top" align="left">0.86</td>
<td valign="top" align="left">0.80</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<th valign="top" colspan="8" align="left">Experienced Physicians</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;EP1</td>
<td valign="top" align="left">0.84 (0.78, 0.90)</td>
<td valign="top" align="left">0.87 (0.82, 0.91)</td>
<td valign="top" align="left">0.82 (0.77, 0.88)</td>
<td valign="top" align="left">0.85</td>
<td valign="top" align="left">0.84</td>
<td valign="top" align="left">0.85</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;EP2</td>
<td valign="top" align="left">0.80 (0.73, 0.86)</td>
<td valign="top" align="left">0.80 (0.73, 0.85)</td>
<td valign="top" align="left">0.80 (0.75, 0.86)</td>
<td valign="top" align="left">0.80</td>
<td valign="top" align="left">0.81</td>
<td valign="top" align="left">0.78</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;EP3</td>
<td valign="top" align="left">0.79 (0.71, 0.84)</td>
<td valign="top" align="left">0.64 (0.59, 0.71)</td>
<td valign="top" align="left">0.93 (0.85, 0.96)</td>
<td valign="top" align="left">0.75</td>
<td valign="top" align="left">0.91</td>
<td valign="top" align="left">0.70</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;EP4</td>
<td valign="top" align="left">0.81 (0.74, 0.87)</td>
<td valign="top" align="left">0.84 (0.79, 0.88)</td>
<td valign="top" align="left">0.78 (0.72, 0.85)</td>
<td valign="top" align="left">0.82</td>
<td valign="top" align="left">0.80</td>
<td valign="top" align="left">0.82</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;EP5</td>
<td valign="top" align="left">0.81 (0.74, 0.87)</td>
<td valign="top" align="left">0.81 (0.77, 0.87)</td>
<td valign="top" align="left">0.82 (0.76, 0.88)</td>
<td valign="top" align="left">0.82</td>
<td valign="top" align="left">0.83</td>
<td valign="top" align="left">0.79</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;EP6</td>
<td valign="top" align="left">0.80 (0.73, 0.86)</td>
<td valign="top" align="left">0.76 (0.70, 0.82)</td>
<td valign="top" align="left">0.84 (0.79, 0.88)</td>
<td valign="top" align="left">0.80</td>
<td valign="top" align="left">0.84</td>
<td valign="top" align="left">0.76</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;EP7</td>
<td valign="top" align="left">0.79 (0.71, 0.85)</td>
<td valign="top" align="left">0.70 (0.65, 0.76)</td>
<td valign="top" align="left">0.88 (0.82, 0.93)</td>
<td valign="top" align="left">0.77</td>
<td valign="top" align="left">0.87</td>
<td valign="top" align="left">0.73</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;EP8</td>
<td valign="top" align="left">0.80 (0.72, 0.85)</td>
<td valign="top" align="left">0.66 (0.61, 0.72)</td>
<td valign="top" align="left">0.93 (0.88, 0.96)</td>
<td valign="top" align="left">0.77</td>
<td valign="top" align="left">0.92</td>
<td valign="top" align="left">0.72</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;EP9</td>
<td valign="top" align="left">0.78 (0.70, 0.84)</td>
<td valign="top" align="left">0.71 (0.66, 0.78)</td>
<td valign="top" align="left">0.84 (0.79, 0.89)</td>
<td valign="top" align="left">0.77</td>
<td valign="top" align="left">0.83</td>
<td valign="top" align="left">0.73</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;EP10</td>
<td valign="top" align="left">0.79 (0.72, 0.85)</td>
<td valign="top" align="left">0.77 (0.71, 0.83)</td>
<td valign="top" align="left">0.82 (0.78, 0.89)</td>
<td valign="top" align="left">0.80</td>
<td valign="top" align="left">0.82</td>
<td valign="top" align="left">0.77</td>
<td valign="top" align="left">&#x2013;</td>
</tr>
<tr>
<th valign="top" colspan="8" align="left">Clinical Model</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Clinical Model</td>
<td valign="top" align="left">0.83 (0.69, 0.92)</td>
<td valign="top" align="left">0.83 (0.77, 0.88)</td>
<td valign="top" align="left">0.82 (0.76, 0.89)</td>
<td valign="top" align="left">0.83</td>
<td valign="top" align="left">0.83</td>
<td valign="top" align="left">0.82</td>
<td valign="top" align="left">0.83, (0.70, 0.96)</td>
</tr>
<tr>
<th valign="top" colspan="8" align="left">Tumor markers</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;CA19-9</td>
<td valign="top" align="left">0.66 (0.59, 0.75)</td>
<td valign="top" align="left">0.93 (0.88, 0.97)</td>
<td valign="top" align="left">0.39 (0.35, 0.44)</td>
<td valign="top" align="left">0.75</td>
<td valign="top" align="left">0.63</td>
<td valign="top" align="left">0.83</td>
<td valign="top" align="left">0.77, (0.69, 0.84)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;CEA</td>
<td valign="top" align="left">0.71 (0.63, 0.77)</td>
<td valign="top" align="left">0.59 (0.55, 0.64)</td>
<td valign="top" align="left">0.83 (0.78, 0.59)</td>
<td valign="top" align="left">0.68</td>
<td valign="top" align="left">0.79</td>
<td valign="top" align="left">0.65</td>
<td valign="top" align="left">0.66, (0.58, 0.75)</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;P *</td>
<td valign="top" align="left">0.016</td>
<td valign="top" align="left">0.201</td>
<td valign="top" align="left">0.769</td>
<td valign="top" align="left">0.678</td>
<td valign="top" align="left">1.000</td>
<td valign="top" align="left">0.769</td>
<td valign="top" align="left">0.387</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;P &#x2020;</td>
<td valign="top" align="left">&lt;0.001</td>
<td valign="top" align="left">0.013</td>
<td valign="top" align="left">0.646</td>
<td valign="top" align="left">0.029</td>
<td valign="top" align="left">0.009</td>
<td valign="top" align="left">0.646</td>
<td valign="top" align="left">0.031</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>CA19-9, Carbohydrate Antigen 199; CEA, Carcinoembryonic Antigen; AUC, Area Under the Curve; Acc, Accuracy; Sens, Sensitivity; Spec, Specificity; F1, F1 score; PPV, Positive Predictive Value; NPV, Negative Predictive Value; CI, Confidence Interval; EP, Experienced Physician.</p>
</fn>
<fn>
<p>*p value of Logistic Prediction Model versus CA19-9.</p>
</fn>
<fn>
<p>&#x2020;p value of Logistic Prediction Model versus CEA.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>ROC curve analysis <bold>(A)</bold> and radar chart <bold>(B)</bold> of diagnostic model and tumor markers. CA19-9, Carbohydrate Antigen 199; CEA, Carcinoembryonic Antigen; AUC, Area Under the Curve; CI, Confidence Interval; ROC, Receiver Operating Characteristic.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1613818-g002.tif">
<alt-text content-type="machine-generated">Panel A shows a Receiver Operating Characteristic (ROC) curve comparing a clinical model, CA19-9, and CEA, with AUC values of 0.83, 0.77, and 0.66 respectively. Panel B displays a radar chart for the clinical model, CA19-9, and CEA, illustrating AUC, accuracy, F1 score, sensitivity, and specificity with various colored lines.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3_3">
<title>Comparative diagnostic performance of LLMs, tumor markers, clinical model, and physicians</title>
<p>Four LLMs (40%), one clinical model, and six physicians (60%) achieved accuracies &#x2265;80%. Kimi demonstrated the highest accuracy (87%), significantly outperforming 70% of physicians (7/10, p &lt; 0.01) (<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>, <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>). Top-performing LLMs including Deepseek-R1 (0.83), Claude-3.5S (0.82), Llama-3.1 (0.81), and GPT-4T (0.79) showed comparable accuracy to physicians (0.78&#x2013;0.84, p &gt; 0.05), while physicians outperformed lower-performing LLMs (GPT-4o, Gemini-1.5-pro). Eighty percent of LLMs exceeded CEA (0.66) and CA19-9 (0.71) in accuracy, with the clinical model (0.83) competing with top LLMs.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Comparative performance evaluation. <bold>(A)</bold> Accuracy among Different LLMs, Experienced Physicians, Clinical Model and Tumor Markers. <bold>(B)</bold> A Comparative Analysis of Diagnostic Accuracy and Significance Testing among models. The p-values (Holm-adjusted) were from the comparison of accuracy between the predictive groups listed along the horizontal axis and those on the vertical axis. A positive p-value indicates that the accuracy of the group on the horizontal axis is statistically greater than that of the group on the vertical axis, whereas a negative p-value signifies the opposite. CA19-9, Carbohydrate Antigen 199; CEA, Carcinoembryonic Antigen; EP, Experienced physician.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1613818-g003.tif">
<alt-text content-type="machine-generated">Chart A shows the accuracy of different models, categorized by color: EP (purple), Tumor marker (green), Clinical model (blue), and LLM (orange). Chart B is a matrix illustrating p-values for model comparisons, with varying hues representing different p-value ranges. Both charts contain numerous models like EP, Qwen, and GPT, alongside Tumor marker measures, highlighting statistical significance variations across pairs.</alt-text>
</graphic>
</fig>
<p>CA19&#x2013;9 exhibited the highest sensitivity (0.93) across 23 predictive groups, significant in 18 groups (p: 0&#x2013;0.044). GPT-4T, GPT-4o, and Gemini-1.5-pro showed the highest specificity (0.99), significant in 14 groups (p: 0&#x2013;0.048). Kimi led in F1-score (0.87, significant in 12 groups), GPT-4T in PPV (0.98, 13 groups), and EP1 in NPV (0.85, 12 groups). Top LLMs dominated four metrics (accuracy, specificity, F1-score, PPV), while tumor markers and physicians excelled in sensitivity and NPV, respectively.</p>
<p>Analysis of 360 incorrect diagnoses revealed that over 80% resulted from LLMs over-relying on single data sources (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S6</bold>
</xref>). Representative examples: Claude 3.5 misdiagnosed a benign stricture as malignant solely based on elevated CA19-9, while GPT-4T correctly integrated bilirubin trends, imaging, and histology (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>).</p>
<p>Analysis of the misdiagnoses revealed &gt;80% originated from LLMs over-relying on single data sources (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S3</bold>
</xref>). For instance, Claude 3.5 misdiagnosed a benign stricture as malignant based solely on elevated CA19-9, whereas GPT-4T integrated bilirubin trends, imaging, and histology for correct diagnosis (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Document</bold>
</xref>).</p>
</sec>
<sec id="s3_4">
<title>Subgroup analysis by stricture location</title>
<p>Performance metrics for hilar (n=29, 9 benign/20 malignant) and non-hilar subgroups are detailed in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref> and <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S4</bold>
</xref>. Given the limited sample size of hilar strictures (n=29, including 9 benign and 20 malignant cases), the subgroup analysis should be interpreted with caution due to potential overfitting risks. Despite this limitation, we observed that in Physicians showed higher accuracy in hilar versus non-hilar strictures (0.87 vs. 0.79, p &lt; 0.001), while LLMs had comparable accuracy in both subgroups (hilar: 0.64-0.95; non-hilar: 0.62-0.88, p &gt; 0.05). In the hilar subgroup, Deepseek-R1 showed highest hilar accuracy (0.95, 95% CI:0.77-0.99), followed by Kimi (0.87) and Claude-3.5S (0.84). Notably in this exploratory analysis, 7/10 physicians achieved superior accuracy over 8/10 LLMs (87-90% vs. 64-84%, p&lt;0.05). In the non-hilar subgroup, LLMs showed competitive/exceeding accuracy. Given the small sample size, these findings should be interpreted as preliminary and require validation in larger cohorts.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Accuracy and 95% CIs across subgroups for different prediction models. Error bars represent 95% confidence intervals. Hilar strictures (n=29: 9 benign, 20 malignant), non-hilar strictures (n=130: 74 benign, 56 malignant). CA19-9, Carbohydrate Antigen 199; CEA, Carcinoembryonic Antigen; EP, Experienced physician.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-15-1613818-g004.tif">
<alt-text content-type="machine-generated">Bar chart showing accuracy comparisons between Hilar and Non-hilar subgroups across various models. Each bar represents a model, with yellow for Hilar and turquoise for Non-hilar. Error bars indicate variability. Models include GPT-4T, Claude-3.5S, Qwen-2, and others, with accuracy ranging from 0.25 to 1.00.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3_5">
<title>LLM diagnostic concordance assessment</title>
<p>Deepseek-R1, GPT-4T, GPT-4o, Llama-3.1, Gemini-1.5-pro, and Claude-3.5S showed near-perfect internal concordance (&#x3ba;=0.81-0.97). Kimi, ERNIE-4, and GLM-4 demonstrated substantial concordance, while Qwen-2 showed fair concordance. When compared to true values, no model reached almost perfect concordance: Deepseek-R1, Llama-3.1, Claude-3.5S, and Kimi showed substantial concordance; ERNIE-4, GLM-4, Qwen-2 moderate; Gemini-1.5 pro and GPT-4o fair (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S6</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S3</bold>
</xref>).</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<title>Discussion</title>
<p>The differentiation between benign and malignant biliary strictures remains clinically challenging. Current endoscopic techniques show limited sensitivity: ERCP-based brushing/biopsy (0.45-0.67) (<xref ref-type="bibr" rid="B27">27</xref>&#x2013;<xref ref-type="bibr" rid="B29">29</xref>) and cholangioscopy biopsy (0.43-0.74) (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B30">30</xref>) often prove inadequate. Our study demonstrates that select large language models (LLMs) achieve diagnostic accuracy rivaling or exceeding human expertise. Kimi attained the highest accuracy (87%), significantly outperforming 70% of experienced physicians (7/10, p&lt;0.01). Three additional LLMs (Deepseek-R1:83%, Claude-3.5S:82%, Llama-3.1:81%) and our clinical prediction model (83%) performed comparably to physicians (78-84%, p&gt;0.05). Collectively, 80% of LLMs surpassed conventional tumor markers (CA19&#x2013;9 accuracy:66%; CEA:71%).</p>
<p>Notably, this is the first study to demonstrate that select large language models (LLMs) match or exceed the accuracy of a clinical model and experienced physicians. These findings suggest LLMs could serve as accessible, real-time diagnostic aids, particularly in resource-constrained settings where specialist expertise is limited. The variation among LLM performance reveals clinically meaningful insights. While GPT-4o and Gemini-1.5-Pro lead general benchmarking tasks (<xref ref-type="bibr" rid="B31">31</xref>), these models underperformed in our specific diagnostic application (accuracies 0.66 and 0.62 respectively). These underperforming models exhibited extreme specificity (0.99) and PPV (0.95-0.97) but critically low sensitivity (0.25-0.34), likely reflecting excessive safety prioritization during training protocols. This pattern necessitates caution when using such models for biliary stricture assessment. Conversely, Kimi&#x2019;s superior performance (87%) highlights how task-specific optimization can yield exceptional diagnostic capability irrespective of general benchmarking performance.</p>
<p>Our subgroup analysis revealed physicians outperformed LLMs in the evaluation of hilar strictures (n=29), with 7/10 physicians achieving significantly higher accuracy than 8/10 LLMs (87-90% vs. 64-84%, p&lt;0.05). This finding aligns with established clinical knowledge that &gt;90% of hilar strictures are malignant (<xref ref-type="bibr" rid="B1">1</xref>), suggesting experienced clinicians better integrate this epidemiological context. The performance gap may indicate incomplete learning of clinical nuances by current LLMs. However, targeted fine-tuning with medical knowledge or Retrieval-Augmented Generation (RAG) (<xref ref-type="bibr" rid="B32">32</xref>&#x2013;<xref ref-type="bibr" rid="B34">34</xref>) could potentially bridge this gap in future iterations. Clinically, this underscores the continued value of expert judgment in anatomically complex presentations, while suggesting LLMs may currently serve best as diagnostic aids for non-hilar strictures where they demonstrated parity with physicians. However, due to due to small sample size in the subgroup of hilar strictures, this analysis is exploratory and requires validation.</p>
<p>Our clinical prediction model, incorporating established risk factors (age, CA19-9, CEA, disease duration &lt;1 month, CRP, surgical history, lymphadenopathy), achieved an AUC of 0.83 (95% CI:0.70-0.96), aligning with prior reports (AUC 0.75-0.83) (<xref ref-type="bibr" rid="B35">35</xref>&#x2013;<xref ref-type="bibr" rid="B38">38</xref>) while maintaining practical clinical utility. CA19&#x2013;9 demonstrated an expected AUC (0.77) matching literature reports (0.759-0.783) (<xref ref-type="bibr" rid="B35">35</xref>, <xref ref-type="bibr" rid="B38">38</xref>), but presented a diagnostic paradox with high sensitivity (0.93) yet poor specificity (0.39). This suggests potential utility as a rule-out screening tool requiring subsequent confirmation, while CEA demonstrated weaker discriminative capacity (AUC 0.66) than some prior studies (<xref ref-type="bibr" rid="B39">39</xref>&#x2013;<xref ref-type="bibr" rid="B42">42</xref>), emphasizing context-dependent variability.</p>
<p>Despite promising diagnostic capabilities, clinical implementation of LLMs faces significant barriers requiring strategic resolution. Error analysis demonstrated that &gt;80% of misdiagnoses originated from LLMs over-relying on isolated data elements rather than multimodal integration. Representative examples included Claude 3.5 misclassifying a benign stricture as malignant based solely on elevated CA19-9, contrasting with GPT-4T&#x2019;s accurate diagnosis achieved through synthesizing bilirubin trends, imaging findings, and histology. This significant limitation persists despite recent demonstrations of LLMs outperforming physicians in controlled diagnostic settings (<xref ref-type="bibr" rid="B43">43</xref>), underscoring a critical challenge in translating artificial intelligence capabilities to clinical practice where multimodal reasoning is essential. Text-based implementation currently constrains LLMs, but emerging multimodal capabilities in radiology (<xref ref-type="bibr" rid="B44">44</xref>&#x2013;<xref ref-type="bibr" rid="B46">46</xref>) and dermatology (<xref ref-type="bibr" rid="B47">47</xref>) suggest promising diagnostic extensions. Future integration of CT/MRCP imaging could substantially enhance biliary stricture evaluation, though diagnosing rare conditions requires specialized training approaches. To address these constraints, we propose a structured workflow comprising: 1) electronic Medical Record-integrated real-time malignancy probability scoring, and 2) automatic referral to targeted multidisciplinary review for cases with LLM confidence scores below 80%. This structure preserves physician oversight while optimizing diagnostic efficiency, particularly valuable in resource-limited settings. However, clinical deployment of LLMs demands addressing several critical ethical considerations: 1) Accountability through legal frameworks addressing liability for diagnostic errors; 2) Hallucination mitigation requiring detection protocols (evidenced in 12% of erroneous outputs); 3) Patient acceptance considerations, with survey data showing 67% rejection of AI-exclusive diagnoses for cancer-related decisions; 4) Equity concerns including documented performance disparities in elderly populations; 5) Transparency requirements for interpretable decision pathways; and 6) Privacy mandates demanding robust data anonymization. Essential mitigation strategies include human-AI collaborative diagnostic models, algorithmic bias correction techniques, and targeted patient education initiatives clarifying LLMs&#x2019; assistive role.</p>
<p>Our study exhibits several limitations that warrant consideration. First, the retrospective single-center design introduces potential selection bias. Second, modest sample size limiting statistical power to address heterogeneity in biliary stricture presentations, potentially restricting generalizability across diverse healthcare settings; Third, the small hilar stricture subgroup (n = 29) limits statistical power for physician-LLM comparisons in this anatomically complex subset. Fourth, version-specific LLM evaluation restricts generalizability to updated iterations. Fifth, variability in physician experience levels may impact human performance benchmarks. Sixth, absence of external validation constrains generalizability.</p>
</sec>
<sec id="s5" sec-type="conclusions">
<title>Conclusion</title>
<p>In conclusion, this study demonstrates select LLMs (Kimi, Deepseek-R1, Claude-3.5S, Llama-3.1) achieve diagnostic accuracy comparable to or exceeding clinical models and physicians for biliary strictures, though hilar cases remain challenging. Their optimal implementation involves augmenting clinical judgment rather than replacing it, especially valuable for non-hilar strictures where performance matched physicians.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>. Further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s7" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The studies involving humans were approved by the Ethics Committee of Xijing Hospital. The studies were conducted in accordance with the local legislation and institutional requirements. Written informed consent for participation was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and institutional requirements. Written informed consent was obtained from the individual(s) for the publication of any potentially identifiable images or data included in this article.</p>
</sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>CK: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Software, Validation, Visualization, Writing &#x2013; original draft. JL: Data curation, Formal analysis, Visualization, Writing &#x2013; review &amp; editing, Resources. XY: Data curation, Formal analysis, Visualization, Writing &#x2013; review &amp; editing, Software. GR: Investigation, Validation, Writing &#x2013; review &amp; editing. LZ: Investigation, Validation, Writing &#x2013; review &amp; editing. WW: Investigation, Writing &#x2013; review &amp; editing. XL: Investigation, Writing &#x2013; review &amp; editing. LW: Investigation, Writing &#x2013; review &amp; editing. GS: Investigation, Writing &#x2013; review &amp; editing. JH: Investigation, Writing &#x2013; review &amp; editing. BW: Investigation, Writing &#x2013; review &amp; editing. YD: Investigation, Writing &#x2013; review &amp; editing. WZ: Investigation, Writing &#x2013; review &amp; editing. YLL: Methodology, Writing &#x2013; review &amp; editing. TL: Validation, Writing &#x2013; review &amp; editing. LL: Validation, Writing &#x2013; review &amp; editing. HL: Investigation, Writing &#x2013; review &amp; editing. SL: Investigation, Writing &#x2013; review &amp; editing. YL: Conceptualization, Methodology, Supervision, Writing &#x2013; review &amp; editing. YP: Conceptualization, Funding acquisition, Project administration, Resources, Supervision, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This study was supported by National Key R&amp;D Program of China (2022YFC2505100) and grants from the National Natural Science Foundation of China (82373117&amp;82370619).</p>
</sec>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="s12" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s13" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fonc.2025.1613818/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fonc.2025.1613818/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elmunzer</surname> <given-names>BJ</given-names>
</name>
<name>
<surname>Maranki</surname> <given-names>JL</given-names>
</name>
<name>
<surname>G&#xf3;mez</surname> <given-names>V</given-names>
</name>
<name>
<surname>Tavakkoli</surname> <given-names>A</given-names>
</name>
<name>
<surname>Sauer</surname> <given-names>BG</given-names>
</name>
<name>
<surname>Limketkai</surname> <given-names>BN</given-names>
</name>
<etal/>
</person-group>. <article-title>ACG clinical guideline: diagnosis and management of biliary strictures</article-title>. <source>Am J Gastroenterol</source>. (<year>2023</year>) <volume>118</volume>:<page-range>405&#x2013;26</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.14309/ajg.0000000000002190</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wen</surname> <given-names>L-J</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J-H</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>H-J</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>K</given-names>
</name>
</person-group>. <article-title>Efficacy and safety of digital single-operator cholangioscopy in the diagnosis of indeterminate biliary strictures by targeted biopsies: A systematic review and meta-analysis</article-title>. <source>Diagn Basel Switz</source>. (<year>2020</year>) <volume>10</volume>:<elocation-id>666</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/diagnostics10090666</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>de Moura</surname> <given-names>DTH</given-names>
</name>
<name>
<surname>Ryou</surname> <given-names>M</given-names>
</name>
<name>
<surname>de Moura</surname> <given-names>EGH</given-names>
</name>
<name>
<surname>Ribeiro</surname> <given-names>IB</given-names>
</name>
<name>
<surname>Bernardo</surname> <given-names>WM</given-names>
</name>
<name>
<surname>Thompson</surname> <given-names>CC</given-names>
</name>
</person-group>. <article-title>Endoscopic ultrasound-guided fine needle aspiration and endoscopic retrograde cholangiopancreatography-based tissue sampling in suspected Malignant biliary strictures: A meta-analysis of same-session procedures</article-title>. <source>Clin Endosc</source>. (<year>2020</year>) <volume>53</volume>:<page-range>417&#x2013;28</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.5946/ce.2019.053</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Moura</surname> <given-names>DTH</given-names>
</name>
<name>
<surname>Moura</surname> <given-names>EGHD</given-names>
</name>
<name>
<surname>Bernardo</surname> <given-names>WM</given-names>
</name>
<name>
<surname>De Moura</surname> <given-names>ETH</given-names>
</name>
<name>
<surname>Baraca</surname> <given-names>FI</given-names>
</name>
<name>
<surname>Kondo</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Endoscopic retrograde cholangiopancreatography versus endoscopic&#xa0;ultrasound for tissue diagnosis of Malignant biliary stricture: Systematic review and meta-analysis</article-title>. <source>Endosc Ultrasound</source>. (<year>2018</year>) <volume>7</volume>:<page-range>10&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4103/2303-9027.193597</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tummala</surname> <given-names>P</given-names>
</name>
<name>
<surname>Munigala</surname> <given-names>S</given-names>
</name>
<name>
<surname>Eloubeidi</surname> <given-names>MA</given-names>
</name>
<name>
<surname>Agarwal</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Patients with obstructive jaundice and biliary stricture &#xb1; mass lesion on imaging: prevalence of Malignancy and potential role of EUS-FNA</article-title>. <source>J Clin Gastroenterol</source>. (<year>2013</year>) <volume>47</volume>:<page-range>532&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1097/MCG.0b013e3182745d9f</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname> <given-names>B</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>B</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Wong Lau</surname> <given-names>JY</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>S</given-names>
</name>
<name>
<surname>Itoi</surname> <given-names>T</given-names>
</name>
<etal/>
</person-group>. <article-title>Asia-Pacific consensus guidelines for endoscopic management of benign biliary strictures</article-title>. <source>Gastrointest Endosc</source>. (<year>2017</year>) <volume>86</volume>:<fpage>44</fpage>&#x2013;<lpage>58</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.gie.2017.02.031</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ramchandani</surname> <given-names>M</given-names>
</name>
<name>
<surname>Lakhtakia</surname> <given-names>S</given-names>
</name>
<name>
<surname>Costamagna</surname> <given-names>G</given-names>
</name>
<name>
<surname>Tringali</surname> <given-names>A</given-names>
</name>
<name>
<surname>P&#xfc;sp&#xf6;ek</surname> <given-names>A</given-names>
</name>
<name>
<surname>Tribl</surname> <given-names>B</given-names>
</name>
<etal/>
</person-group>. <article-title>Fully covered self-expanding metal stent vs multiple plastic stents to treat benign biliary strictures secondary to chronic pancreatitis: A multicenter randomized trial</article-title>. <source>Gastroenterology</source>. (<year>2021</year>) <volume>161</volume>:<page-range>185&#x2013;95</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1053/j.gastro.2021.03.015</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sato</surname> <given-names>T</given-names>
</name>
<name>
<surname>Kogure</surname> <given-names>H</given-names>
</name>
<name>
<surname>Nakai</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Ishigaki</surname> <given-names>K</given-names>
</name>
<name>
<surname>Hakuta</surname> <given-names>R</given-names>
</name>
<name>
<surname>Saito</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>A prospective study of fully covered metal stents for different types of refractory benign biliary strictures</article-title>. <source>Endoscopy</source>. (<year>2020</year>) <volume>52</volume>:<page-range>368&#x2013;76</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1055/a-1111-8666</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ponsioen</surname> <given-names>CY</given-names>
</name>
<name>
<surname>Arnelo</surname> <given-names>U</given-names>
</name>
<name>
<surname>Bergquist</surname> <given-names>A</given-names>
</name>
<name>
<surname>Rauws</surname> <given-names>EA</given-names>
</name>
<name>
<surname>Paulsen</surname> <given-names>V</given-names>
</name>
<name>
<surname>Cant&#xfa;</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>No superiority of stents vs balloon dilatation for dominant strictures in patients with primary sclerosing cholangitis</article-title>. <source>Gastroenterology</source>. (<year>2018</year>) <volume>155</volume>:<fpage>752</fpage>&#x2013;<lpage>759.e5</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1053/j.gastro.2018.05.034</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Franssen</surname> <given-names>S</given-names>
</name>
<name>
<surname>Van Driel</surname> <given-names>LMJW</given-names>
</name>
<name>
<surname>Moelker</surname> <given-names>A</given-names>
</name>
<name>
<surname>Groot Koerkamp</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Primary percutaneous stenting above the ampulla for palliative biliary drainage of Malignant hilar biliary obstruction</article-title>. <source>J Clin Oncol</source>. (<year>2023</year>) <volume>41</volume>:<page-range>527&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1200/JCO.2023.41.4_suppl.527</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Stevens</surname> <given-names>T</given-names>
</name>
<name>
<surname>Parsi</surname> <given-names>MA</given-names>
</name>
<name>
<surname>Bhatt</surname> <given-names>A</given-names>
</name>
<name>
<surname>Kichler</surname> <given-names>A</given-names>
</name>
<name>
<surname>Vargo</surname> <given-names>JJ</given-names>
</name>
</person-group>. <article-title>Superiority of self-expandable metallic stents over plastic stents in treatment of Malignant distal biliary strictures</article-title>. <source>Clin Gastroenterol Hepatol</source>. (<year>2022</year>) <volume>20</volume>:<page-range>e182&#x2013;95</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cgh.2020.12.020</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>AY</given-names>
</name>
<name>
<surname>Yachimski</surname> <given-names>PS</given-names>
</name>
</person-group>. <article-title>Endoscopic management of pancreatobiliary neoplasms</article-title>. <source>Gastroenterology</source>. (<year>2018</year>) <volume>154</volume>:<page-range>1947&#x2013;63</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1053/j.gastro.2017.11.295</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wallace</surname> <given-names>MB</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>KK</given-names>
</name>
<name>
<surname>Adler</surname> <given-names>DG</given-names>
</name>
<name>
<surname>Rastogi</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Recent advances in endoscopy</article-title>. <source>Gastroenterology</source>. (<year>2017</year>) <volume>153</volume>:<page-range>364&#x2013;81</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1053/j.gastro.2017.06.014</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khamaysi</surname> <given-names>I</given-names>
</name>
<name>
<surname>Firman</surname> <given-names>R</given-names>
</name>
<name>
<surname>Martin</surname> <given-names>P</given-names>
</name>
<name>
<surname>Vasilyev</surname> <given-names>G</given-names>
</name>
<name>
<surname>Boyko</surname> <given-names>E</given-names>
</name>
<name>
<surname>Zussman</surname> <given-names>E</given-names>
</name>
</person-group>. <article-title>Mechanical perspective on increasing brush cytology yield</article-title>. <source>ACS Biomater Sci Eng</source>. (<year>2024</year>) <volume>10</volume>:<page-range>1743&#x2013;52</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/acsbiomaterials.3c00935</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>M</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>H</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Dai</surname> <given-names>W</given-names>
</name>
<etal/>
</person-group>. <article-title>More endoscopy-based brushing passes improve the detection of Malignant biliary strictures: A multicenter randomized controlled trial</article-title>. <source>Am J Gastroenterol</source>. (<year>2022</year>) <volume>117</volume>:<page-range>733&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.14309/ajg.0000000000001666</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bang</surname> <given-names>JY</given-names>
</name>
<name>
<surname>Navaneethan</surname> <given-names>U</given-names>
</name>
<name>
<surname>Hasan</surname> <given-names>M</given-names>
</name>
<name>
<surname>Sutton</surname> <given-names>B</given-names>
</name>
<name>
<surname>Hawes</surname> <given-names>R</given-names>
</name>
<name>
<surname>Varadarajulu</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Optimizing outcomes of single-operator cholangioscopy-guided biopsies based on a randomized trial</article-title>. <source>Clin Gastroenterol Hepatol Off Clin Pract J Am Gastroenterol Assoc</source>. (<year>2020</year>) <volume>18</volume>:<fpage>441</fpage>&#x2013;<lpage>448.e1</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cgh.2019.07.035</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>L</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Kulkarni</surname> <given-names>P</given-names>
</name>
</person-group>. <article-title>Tu1008 a meta-analysis of the value of Intraductal Ultrasound (IDUS) in differentiating Malignant from benign biliary strictures</article-title>. <source>Gastroenterology</source>. (<year>2020</year>) <volume>158</volume>:<fpage>S</fpage>&#x2013;<lpage>1003-S-1004</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S0016-5085(20)33183-8</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname> <given-names>S</given-names>
</name>
<name>
<surname>Kahaleh</surname> <given-names>M</given-names>
</name>
<name>
<surname>Sharaiha</surname> <given-names>RZ</given-names>
</name>
<name>
<surname>Tarnasky</surname> <given-names>PR</given-names>
</name>
<name>
<surname>Kedia</surname> <given-names>P</given-names>
</name>
<name>
<surname>Slivka</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Probe-based confocal laser endomicroscopy in the evaluation of dominant strictures in patients with primary sclerosing cholangitis: results of a U.S. multicenter prospective trial</article-title>. <source>Gastrointest Endosc</source>. (<year>2021</year>) <volume>94</volume>:<fpage>569</fpage>&#x2013;<lpage>576.e1</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.gie.2021.03.027</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Joshi</surname> <given-names>V</given-names>
</name>
<name>
<surname>Patel</surname> <given-names>SN</given-names>
</name>
<name>
<surname>Vanderveldt</surname> <given-names>H</given-names>
</name>
<name>
<surname>Oliva</surname> <given-names>I</given-names>
</name>
<name>
<surname>Raijman</surname> <given-names>I</given-names>
</name>
<name>
<surname>Molina</surname> <given-names>C</given-names>
</name>
<etal/>
</person-group>. <article-title>Mo1963 A pilot study of safety and efficacy of directed cannulation with a low profile catheter (LP) and imaging characteristics of bile duct wall using optical coherance tomography (OCT) for indeterminate biliary strictures initial report on <italic>in-vivo</italic> evaluation during ERCP</article-title>. <source>Gastrointest Endosc</source>. (<year>2017</year>) <volume>85</volume>:<page-range>AB496&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.gie.2017.03.1150</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>W</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>W</given-names>
</name>
<name>
<surname>Song</surname> <given-names>R</given-names>
</name>
<name>
<surname>Mei</surname> <given-names>T</given-names>
</name>
<etal/>
</person-group>. <article-title>Diagnostic value of carbohydrate antigen 50 in biliary tract cancer: a large-scale multicenter study</article-title>. <source>Cancer Med</source>. (<year>2024</year>) <volume>13</volume>:<fpage>e7388</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/cam4.7388</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Levine</surname> <given-names>DM</given-names>
</name>
<name>
<surname>Tuwani</surname> <given-names>R</given-names>
</name>
<name>
<surname>Kompa</surname> <given-names>B</given-names>
</name>
<name>
<surname>Varma</surname> <given-names>A</given-names>
</name>
<name>
<surname>Finlayson</surname> <given-names>SG</given-names>
</name>
<name>
<surname>Mehrotra</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>The diagnostic and triage accuracy of the GPT-3 artificial intelligence model: an observational study</article-title>. <source>Lancet Digit Health</source>. (<year>2024</year>) <volume>6</volume>:<page-range>e555&#x2013;61</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S2589-7500(24)00097-9</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname> <given-names>L</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>S</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Shang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>X</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X</given-names>
</name>
</person-group>. <article-title>MMGPL: multimodal medical data analysis with graph prompt learning</article-title>. <source>Med Image Anal</source>. (<year>2024</year>) <volume>97</volume>:<elocation-id>103225</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.media.2024.103225</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thirunavukarasu</surname> <given-names>AJ</given-names>
</name>
<name>
<surname>Ting</surname> <given-names>DSJ</given-names>
</name>
<name>
<surname>Elangovan</surname> <given-names>K</given-names>
</name>
<name>
<surname>Gutierrez</surname> <given-names>L</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>TF</given-names>
</name>
<name>
<surname>Ting</surname> <given-names>DSW</given-names>
</name>
</person-group>. <article-title>Large language models in medicine</article-title>. <source>Nat Med</source>. (<year>2023</year>) <volume>29</volume>:<page-range>1930&#x2013;40</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Santoki</surname> <given-names>A</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>R</given-names>
</name>
<name>
<surname>Peter</surname> <given-names>M</given-names>
</name>
<name>
<surname>Mathew</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Implementing large language model-based artificial intelligence (AI) technology in proposing effective treatment plans in patients with cancer</article-title>. <source>J Clin Oncol</source>. (<year>2024</year>) <volume>42</volume>:<page-range>e13660&#x2013;0</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1200/JCO.2024.42.16_suppl.e13660</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiu</surname> <given-names>P</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>C</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>W</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Towards building multilingual language model for medicine</article-title>. <source>Nat Commun</source>. (<year>2024</year>) <volume>15</volume>:<fpage>8384</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-024-52417-z</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>S</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>X</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>C</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Ran</surname> <given-names>G</given-names>
</name>
<etal/>
</person-group>. <article-title>The performance of large language model powered chatbots compared to oncology physicians on colorectal cancer queries</article-title>. <source>Int J Surg</source>. (<year>2024</year>) <volume>110</volume>(<issue>10</issue>):<page-range>6509&#x2013;17</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1097/JS9.0000000000001850</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yoon</surname> <given-names>SB</given-names>
</name>
<name>
<surname>Moon</surname> <given-names>S-H</given-names>
</name>
<name>
<surname>Ko</surname> <given-names>SW</given-names>
</name>
<name>
<surname>Lim</surname> <given-names>H</given-names>
</name>
<name>
<surname>Kang</surname> <given-names>HS</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>JH</given-names>
</name>
</person-group>. <article-title>Brush cytology, forceps biopsy, or endoscopic ultrasound-guided sampling for diagnosis of bile duct cancer: A meta-analysis</article-title>. <source>Dig Dis Sci</source>. (<year>2022</year>) <volume>67</volume>:<page-range>3284&#x2013;97</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10620-021-07138-4</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jeon</surname> <given-names>TY</given-names>
</name>
<name>
<surname>Choi</surname> <given-names>MH</given-names>
</name>
<name>
<surname>Yoon</surname> <given-names>SB</given-names>
</name>
<name>
<surname>Soh</surname> <given-names>JS</given-names>
</name>
<name>
<surname>Moon</surname> <given-names>S-H</given-names>
</name>
</person-group>. <article-title>Systematic review and meta-analysis of percutaneous transluminal forceps biopsy for diagnosing Malignant biliary strictures</article-title>. <source>Eur Radiol</source>. (<year>2022</year>) <volume>32</volume>:<page-range>1747&#x2013;56</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00330-021-08301-1</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Navaneethan</surname> <given-names>U</given-names>
</name>
<name>
<surname>Njei</surname> <given-names>B</given-names>
</name>
<name>
<surname>Lourdusamy</surname> <given-names>V</given-names>
</name>
<name>
<surname>Konjeti</surname> <given-names>R</given-names>
</name>
<name>
<surname>Vargo</surname> <given-names>JJ</given-names>
</name>
<name>
<surname>Parsi</surname> <given-names>MA</given-names>
</name>
</person-group>. <article-title>Comparative effectiveness of biliary brush cytology and intraductal biopsy for detection of Malignant biliary strictures: a systematic review and meta-analysis</article-title>. <source>Gastrointest Endosc</source>. (<year>2015</year>) <volume>81</volume>:<page-range>168&#x2013;76</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.gie.2014.09.017</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>J</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>Is single-operator peroral cholangioscopy a useful tool for the diagnosis of indeterminate biliary lesion? A systematic review and meta-analysis</article-title>. <source>Gastrointest Endosc</source>. (<year>2015</year>) <volume>82</volume>:<fpage>79</fpage>&#x2013;<lpage>87</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.gie.2014.12.021</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Chiang</surname> <given-names>W-L</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>L</given-names>
</name>
<name>
<surname>Sheng</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Angelopoulos</surname> <given-names>AN</given-names>
</name>
<name>
<surname>Li</surname> <given-names>T</given-names>
</name>
<name>
<surname>Li</surname> <given-names>D</given-names>
</name>
<etal/>
</person-group>. <source>Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference</source> (<year>2024</year>). Available online at: <uri xlink:href="http://arxiv.org/abs/2403.04132">http://arxiv.org/abs/2403.04132</uri> (Accessed <access-date>August 26, 2024</access-date>).</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tran</surname> <given-names>H</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Yao</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>H</given-names>
</name>
</person-group>. <article-title>BioInstruct: instruction tuning of large language models for biomedical natural language processing</article-title>. <source>J Am Med Inform Assoc</source>. (<year>2024</year>) <volume>31</volume>:<page-range>1821&#x2013;32</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/jamia/ocae122</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ge</surname> <given-names>J</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>S</given-names>
</name>
<name>
<surname>Owens</surname> <given-names>J</given-names>
</name>
<name>
<surname>Galvez</surname> <given-names>V</given-names>
</name>
<name>
<surname>Gologorskaya</surname> <given-names>O</given-names>
</name>
<name>
<surname>Lai</surname> <given-names>JC</given-names>
</name>
<etal/>
</person-group>. <article-title>Development of a liver disease-specific large language model chat interface using retrieval-augmented generation</article-title>. <source>Hepatol Baltim Md</source>. (<year>2024</year>) <volume>80</volume>(<issue>5</issue>):<page-range>1158&#x2013;68</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1097/HEP.0000000000000834</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Du</surname> <given-names>D</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>F</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L</given-names>
</name>
</person-group>. <article-title>Enhancing recognition and interpretation of functional phenotypic sequences through fine-tuning pre-trained genomic models</article-title>. <source>J Transl Med</source>. (<year>2024</year>) <volume>22</volume>:<fpage>756</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12967-024-05567-z</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wen</surname> <given-names>N</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>D</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>X</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>G</given-names>
</name>
<name>
<surname>Nie</surname> <given-names>G</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Cholangiocarcinoma combined with biliary obstruction: an exosomal circRNA signature for diagnosis and early recurrence monitoring</article-title>. <source>Signal Transduct Target Ther</source>. (<year>2024</year>) <volume>9</volume>:<fpage>107</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41392-024-01814-3</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname> <given-names>B</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>L</given-names>
</name>
<name>
<surname>He</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Pan</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>T</given-names>
</name>
<etal/>
</person-group>. <article-title>Diagnostic accuracy of serum CA19&#x2013;9 in patients with cholangiocarcinoma: A systematic review and meta-analysis</article-title>. <source>Med Sci Monit</source>. (<year>2015</year>) <volume>21</volume>:<page-range>3555&#x2013;63</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.12659/MSM.895040</pub-id>
</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>DSK</given-names>
</name>
<name>
<surname>Prado</surname> <given-names>MM</given-names>
</name>
<name>
<surname>Giovannetti</surname> <given-names>E</given-names>
</name>
<name>
<surname>Jiao</surname> <given-names>LR</given-names>
</name>
<name>
<surname>Krell</surname> <given-names>J</given-names>
</name>
<name>
<surname>Frampton</surname> <given-names>AE</given-names>
</name>
</person-group>. <article-title>MOY 2 microRNAs as bile based biomarkers for pancreaticobiliary cancers (MIRABILE)</article-title>. <source>Br J Surg</source>. (<year>2023</year>) <volume>110</volume>:<elocation-id>znad241.005</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bjs/znad241.005</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname> <given-names>L</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Yue</surname> <given-names>P</given-names>
</name>
<name>
<surname>Li</surname> <given-names>S</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Mi</surname> <given-names>N</given-names>
</name>
<etal/>
</person-group>. <article-title>Identification of a novel bile marker clusterin and a public online prediction platform based on deep learning for cholangiocarcinoma</article-title>. <source>BMC Med</source>. (<year>2023</year>) <volume>21</volume>:<fpage>294</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12916-023-02990-9</pub-id>
</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Loosen</surname> <given-names>SH</given-names>
</name>
<name>
<surname>Roderburg</surname> <given-names>C</given-names>
</name>
<name>
<surname>Kauertz</surname> <given-names>KL</given-names>
</name>
<name>
<surname>Koch</surname> <given-names>A</given-names>
</name>
<name>
<surname>Vucur</surname> <given-names>M</given-names>
</name>
<name>
<surname>Schneider</surname> <given-names>AT</given-names>
</name>
<etal/>
</person-group>. <article-title>CEA but not CA19&#x2013;9 is an independent prognostic factor in patients undergoing resection of cholangiocarcinoma</article-title>. <source>Sci Rep</source>. (<year>2017</year>) <volume>7</volume>:<fpage>16975</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-017-17175-7</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Li</surname> <given-names>D-J</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>W</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J-W</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>Application of joint detection&#xa0;of&#xa0;AFP, CA19-9, CA125 and CEA in identification and diagnosis of cholangiocarcinoma</article-title>. <source>Asian Pac J Cancer Prev</source>. (<year>2015</year>) <volume>16</volume>:<page-range>3451&#x2013;5</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.7314/APJCP.2015.16.8.3451</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tuzun Ince</surname> <given-names>A</given-names>
</name>
<name>
<surname>Yildiz</surname> <given-names>K</given-names>
</name>
<name>
<surname>Baysal</surname> <given-names>B</given-names>
</name>
<name>
<surname>Danalioglu</surname> <given-names>A</given-names>
</name>
<name>
<surname>Kocaman</surname> <given-names>O</given-names>
</name>
<name>
<surname>Tozlu</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Roles of serum and biliary CEA, CA19-9, VEGFR3, and TAC in differentiating between Malignant and benign biliary obstructions</article-title>. <source>Turk J Gastroenterol</source>. (<year>2014</year>) <volume>25</volume>:<page-range>162&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.5152/tjg.2014.6056</pub-id>
</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Macias</surname> <given-names>RIR</given-names>
</name>
<name>
<surname>Kornek</surname> <given-names>M</given-names>
</name>
<name>
<surname>Rodrigues</surname> <given-names>PM</given-names>
</name>
<name>
<surname>Paiva</surname> <given-names>NA</given-names>
</name>
<name>
<surname>Castro</surname> <given-names>RE</given-names>
</name>
<name>
<surname>Urban</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Diagnostic and prognostic biomarkers in cholangiocarcinoma</article-title>. <source>Liver Int</source>. (<year>2019</year>) <volume>39</volume>:<page-range>108&#x2013;22</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/liv.14090</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brodeur</surname> <given-names>PG</given-names>
</name>
<name>
<surname>Buckley</surname> <given-names>TA</given-names>
</name>
<name>
<surname>Kanjee</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Goh</surname> <given-names>E</given-names>
</name>
<name>
<surname>Ling</surname> <given-names>EB</given-names>
</name>
<name>
<surname>Jain</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>Superhuman performance of a large language model on the reasoning tasks of a physician</article-title>. <source>arXiv</source> (<year>2024</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.48550/ARXIV.2412.10849</pub-id>
</citation>
</ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brin</surname> <given-names>D</given-names>
</name>
<name>
<surname>Sorin</surname> <given-names>V</given-names>
</name>
<name>
<surname>Barash</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Konen</surname> <given-names>E</given-names>
</name>
<name>
<surname>Glicksberg</surname> <given-names>BS</given-names>
</name>
<name>
<surname>Nadkarni</surname> <given-names>GN</given-names>
</name>
<etal/>
</person-group>. <article-title>Assessing GPT-4 multimodal performance in radiological image analysis</article-title>. <source>Eur Radiol</source>. (<year>2025</year>) <volume>35</volume>(<issue>4</issue>):<page-range>1959&#x2013;65</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00330-024-11035-5</pub-id>
</citation>
</ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaczmarczyk</surname> <given-names>R</given-names>
</name>
<name>
<surname>Wilhelm</surname> <given-names>TI</given-names>
</name>
<name>
<surname>Martin</surname> <given-names>R</given-names>
</name>
<name>
<surname>Roos</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Evaluating multimodal AI in medical diagnostics</article-title>. <source>NPJ Digit Med</source>. (<year>2024</year>) <volume>7</volume>:<fpage>205</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41746-024-01208-3</pub-id>
</citation>
</ref>
<ref id="B46">
<label>46</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Be&#x15f;ler</surname> <given-names>MS</given-names>
</name>
</person-group>. <article-title>The accuracy of the multimodal large language model GPT-4 on sample questions from the interventional radiology board examination</article-title>. <source>Acad Radiol</source>. (<year>2024</year>) <volume>31</volume>:<fpage>3476</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.acra.2024.03.023</pub-id>
</citation>
</ref>
<ref id="B47">
<label>47</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>J</given-names>
</name>
<name>
<surname>He</surname> <given-names>X</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>L</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>X</given-names>
</name>
<name>
<surname>Chu</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Pre-trained multimodal large language model enhances dermatological diagnosis using SkinGPT-4</article-title>. <source>Nat Commun</source>. (<year>2024</year>) <volume>15</volume>:<fpage>5649</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-024-50043-3</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>