<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1629149</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Evaluation of the accuracy and repeatability of Deepseek V3, Doubao, and Kimi1.5 in answering knowledge-related queries about chronic non-bacterial osteitis</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author"><name><surname>Zhu</surname> <given-names>Zhenxing</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Xie</surname> <given-names>Jun</given-names></name><xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Zhou</surname> <given-names>Longxin</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Yang</surname> <given-names>Chaoran</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes"><name><surname>Li</surname> <given-names>Feng</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3068216/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Rehabilitation Medicine, Ganzhou People's Hospital</institution>, <addr-line>Ganzhou</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Obstetrics and Gynecology, Dayu County Maternal and Child Health Hospital, Ganzhou University</institution>, <addr-line>Ganzhou</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0001">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/588118/overview">Tim Hulsen</ext-link>, Rotterdam University of Applied Sciences, Netherlands</p>
</fn>
<fn fn-type="edited-by" id="fn0002">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1902269/overview">Ivan &#x0160;o&#x0161;a</ext-link>, University of Rijeka, Croatia</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1788763/overview">Abdur Rasool</ext-link>, University of Hawaii at Manoa, United States</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2795007/overview">Hossein Motahari-Nezhad</ext-link>, University of Isfahan, Iran</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3030377/overview">Jun-hee Kim</ext-link>, Yonsei University, Republic of Korea</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3101269/overview">David J. Bunnell</ext-link>, University of Maryland, United States</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Feng Li, <email>li15297779272@163.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>29</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1629149</elocation-id>
<history>
<date date-type="received">
<day>18</day>
<month>07</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>15</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Zhu, Xie, Zhou, Yang and Li.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Zhu, Xie, Zhou, Yang and Li</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec id="sec1">
<title>Background</title>
<p>There are significant differences in the diagnosis and treatment of chronic non-bacterial osteitis (CNO), and there is an urgent need for health education efforts to enhance awareness of this condition. Deepseek V3, Doubao, and Kimi1.5 are highly popular language models in China that can provide knowledge related to diseases. This article aims to investigate the accuracy and reproducibility of the responses provided by these three artificial intelligence (AI) language models in answering questions about CNO.</p>
</sec>
<sec id="sec2">
<title>Methods</title>
<p>According to the latest expert consensus, 16 questions related to CNO were collected. The three AI language models were separately asked these questions at three different times. The answers were independently evaluated by two orthopedic experts.</p>
</sec>
<sec id="sec3">
<title>Results</title>
<p>Among the responses of the three AI models to 16 CNO-related questions across three rounds of testing, only Doubao received &#x201C;Completely incorrect&#x201D; ratings (accounting for 6.25%) in the third round of scoring by Reviewer 2. During the answering process, Doubao had the shortest response time and provided the most words in its answers. In the first and third rounds of scoring by the first expert, Kimi scored the highest (3.938&#x202F;&#x00B1;&#x202F;0.342, 3.875&#x202F;&#x00B1;&#x202F;0.873), while in the second round, Doubao scored the highest (3.875&#x202F;&#x00B1;&#x202F;0.5). In the second round of scoring by the second expert, Doubao received the highest score (3.812&#x202F;&#x00B1;&#x202F;0.403). In the first and third rounds, Kimi1.5 received the highest score (3.812&#x202F;&#x00B1;&#x202F;0.602, 3.812&#x202F;&#x00B1;&#x202F;0.704).</p>
</sec>
<sec id="sec4">
<title>Conclusion</title>
<p>Deepseek V3, Doubao, and Kimi1.5 are capable of answering most questions related to CNO with good accuracy and reproducibility, showing no significant differences.</p>
</sec>
</abstract>
<kwd-group>
<kwd>chronic non-bacterial osteitis</kwd>
<kwd>Chinese AI chatbots</kwd>
<kwd>knowledge retrieval</kwd>
<kwd>Deepseek V3</kwd>
<kwd>Doubao</kwd>
<kwd>Kimi1.5</kwd>
</kwd-group>
<counts>
<fig-count count="4"/>
<table-count count="3"/>
<equation-count count="0"/>
<ref-count count="26"/>
<page-count count="10"/>
<word-count count="5510"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Medicine and Public Health</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec5">
<label>1</label>
<title>Introduction</title>
<p>Chronic non-bacterial osteitis (CNO) is an autoinflammatory bone disease that commonly affects children and adolescents (<xref ref-type="bibr" rid="ref25">Zhao et al., 2021</xref>). CNO can lead to severe complications, including bone pain and bone damage. Its pathophysiological characteristics are marked by increased inflammasome assembly and imbalanced cytokine expression. Treatment medications include nonsteroidal anti-inflammatory drugs (NSAIDs), corticosteroids, etc. (<xref ref-type="bibr" rid="ref7">Hedrich et al., 2023</xref>). Though reliable epidemiological data are lacking, CNO is likely to be one of the most common autoinflammatory syndromes (<xref ref-type="bibr" rid="ref19">Schnabel et al., 2016</xref>). In Germany, incidence rates were estimated at 0.4 per 100.000 children/year (<xref ref-type="bibr" rid="ref10">Jansson et al., 2011</xref>). However, despite recent advances in our understanding of CNO, the number of detected cases has increased in recent years (<xref ref-type="bibr" rid="ref13">Lenert and Ferguson, 2020</xref>). Nevertheless, there are considerable practice variations in the marking, diagnosis, and treatment of CNO currently (<xref ref-type="bibr" rid="ref23">Winter et al., 2025</xref>). Therefore, there is an urgent need for health education efforts to enhance awareness of CNO.</p>
<p>Artificial intelligence (AI) is a system&#x2019;s capacity to analyze data. It uses computers and machines to boost humans&#x2019; decision - making, problem - solving, and tech innovation capabilities (<xref ref-type="bibr" rid="ref20">Tang et al., 2022</xref>). Currently, AI is becoming increasingly popular in China and even worldwide, and it is being applied in various fields (<xref ref-type="bibr" rid="ref22">Wei et al., 2022</xref>). With the aging of the population, the global demand for high-quality healthcare has increased, while artificial intelligence (AI) has been widely applied in modern medicine&#x2014;for instance, in diagnosis and treatment (<xref ref-type="bibr" rid="ref8">Higuchi et al., 2024</xref>; <xref ref-type="bibr" rid="ref4">Chadebecq et al., 2023</xref>). These AI tools can assist clinicians in different fields to make more informed decisions (<xref ref-type="bibr" rid="ref11">Kulkarni and Singh, 2023</xref>). In the field of oncology, AI can be used for cancer screening and diagnosis (<xref ref-type="bibr" rid="ref1">Abbasi, 2020</xref>). In cardiovascular medicine, it can integrate various forms of patient data to aid doctors in treatment (<xref ref-type="bibr" rid="ref14">L&#x00FC;scher et al., 2024</xref>). Even in surgical procedures, AI can improve multiple aspects such as preoperative, intraoperative, and postoperative care (<xref ref-type="bibr" rid="ref21">Varghese et al., 2024</xref>). The rapid development of AI has provided convenience for the entire medical industry. In daily life, it offers both patients and doctors channels to access disease-related knowledge. However, people are easily misled due to inaccurate and outdated information.</p>
<p>In clinical practice, it is common to encounter patients using these artificial intelligence language models to inquire about disease-related knowledge. Previous studies have evaluated the accuracy and reproducibility of ChatGPT in answering questions related to <italic>Helicobacter pylori</italic> (<xref ref-type="bibr" rid="ref12">Lai et al., 2024</xref>). The results showed that ChatGPT could provide correct answers to most queries related to <italic>Helicobacter pylori</italic>, demonstrating good accuracy and reproducibility.</p>
<p>Deepseek V3, Doubao, and Kimi1.5 are the three most popular and freely accessible artificial intelligence language models in China. They have been widely applied in the country and are used by a large number of people, so it is necessary to assess the accuracy and reproducibility of these three language models in answering disease-related questions.</p>
</sec>
<sec sec-type="methods" id="sec6">
<label>2</label>
<title>Methods</title>
<sec id="sec7">
<label>2.1</label>
<title>Data source</title>
<p>In order to evaluate the accuracy of Deepseek V3 (<ext-link xlink:href="https://chat.deepseek.com" ext-link-type="uri">https://chat.deepseek.com</ext-link>), Doubao (<ext-link xlink:href="https://www.doubao.com" ext-link-type="uri">https://www.doubao.com</ext-link>), and Kimi1.5 (<ext-link xlink:href="https://kimi.moonshot.cn" ext-link-type="uri">https://kimi.moonshot.cn</ext-link>) in answering questions related to CNO, we selected highly scored statements from the latest expert consensus on CNO diseases. This guideline was jointly released in 2025 by over 40 medical experts and served as the source for the questionnaire in this article (<xref ref-type="bibr" rid="ref23">Winter et al., 2025</xref>). To ensure that the selection process not only guarantees the clinical representativeness of the questions (aligning with patients&#x2019; actual needs) but also highlights professional guidance (addressing complex diagnosis and treatment challenges), a core question list that combines practicality and professionalism was finally developed.</p>
<p>We have proposed a total of 16 questions, covering the definition (Question 1), clinical manifestations (Questions 3 and 4), diagnosis (Questions 2, 5, and 6&#x2013;8), differential diagnosis (Questions 9&#x2013;11), treatment (Questions 13&#x2013;16), and treatment of the disease. These questions include those frequently asked by patients, such as <italic>&#x201C;What are the most common manifestations of adult chronic nonbacterial osteomyelitis?,&#x201D;</italic> as well as highly complex ones, such as <italic>&#x201C;What is the first-line treatment option for adult chronic nonbacterial osteomyelitis?.&#x201D;</italic> Furthermore, only questions based on highly recommended or consensus-reached outcomes were selected; those involving controversial recommendations were excluded.</p>
<p>The language of the questions affects both model performance and the applicability of results to other linguistic or regional contexts (<xref ref-type="bibr" rid="ref16">Nezhad et al., 2024</xref>). Therefore, all questions in this study were posed in Chinese. <xref ref-type="fig" rid="fig1">Figure 1</xref> provides a comprehensive overview of the screening process.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>The flow chart of this study.</p>
</caption>
<graphic xlink:href="frai-08-1629149-g001.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Flowchart illustrating the evaluation process of language models in responding to chronic non-bacterial osteitis questions. The process includes selection of questions, collection of responses from three Chinese language models (Deepseek, Doubao, and Kimi) at various time points, evaluation by clinicians, and comparison based on criteria such as comprehensive, correct but incomplete, mixed data, and completely incorrect.</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec8">
<label>2.2</label>
<title>Response generation</title>
<p>Deepseek V3, Doubao, and Kimi1.5 are the three most popular AI language large models in China, all of which can be used for free. In order to evaluate the usefulness of these AI models in answering questions related to CNO, each question was independently inputted, and each model was allowed to provide only one response. We directly input the organized questions for inquiry without using any additional prompts. Nor did we standardize the AI &#x201C;system messages&#x201D; or context windows.</p>
<p>The response time taken to answer the question and the number of characters in the answer were recorded. To assess their reliability in responding to CNO-related queries, independent tests were conducted at three different time points within a 28-day period. All data collection time is April 2025. The counting object of this article is Chinese characters, and the counting tool is Microsoft Word document.</p>
</sec>
<sec id="sec9">
<label>2.3</label>
<title>Assessment and grading</title>
<p>Each question was independently reviewed and scored by two orthopedic surgeons, and the reviewers were unaware of the source of the answer corresponding to each question. For questions where there were discrepancies in scoring, they were referred back to a more experienced expert, who was asked to judge which reviewer&#x2019;s scoring for these questions was more reasonable. To assess accuracy, each question was evaluated individually based on statements recommended in the latest expert consensus, using the following scoring system (<xref ref-type="bibr" rid="ref9">Hu et al., 2024</xref>; <xref ref-type="bibr" rid="ref12">Lai et al., 2024</xref>): (1) Comprehensive (4 points); (2) correct but incomplete (3 points); (3) mixed with correct and incorrect/outdated data (2 points); and (4) completely incorrect (1 point).</p>
</sec>
<sec id="sec10">
<label>2.4</label>
<title>Term definitions</title>
<p>
<list list-type="order">
<list-item>
<p>Accuracy: Evaluated based on the scores of the three AI models across three rounds of testing for the 16 CNO-related questions.</p>
</list-item>
<list-item>
<p>Reproducibility: Evaluated based on the score variations of the three AI models across three rounds of testing when answering the 16 CNO-related questions.</p>
</list-item>
<list-item>
<p>Comprehensiveness: Evaluated based on the degree of consistency between the responses of the three AI models to the 16 CNO-related questions and the recommendations in the expert consensus.</p>
</list-item>
</list>
</p>
</sec>
<sec id="sec11">
<label>2.5</label>
<title>Statistical analysis</title>
<p>The evaluators&#x2019; assessments of the AI&#x2019;s responses related to CNO were expressed as median and interquartile range (IQR), and non-parametric tests were used to evaluate the differences between groups. Bonferroni correction was applied to the comparison of the three groups.<italic>p</italic>&#x202F;&#x003C;&#x202F;0.05 was considered statistically significant. All analyses were conducted statistically using R (version 4.4.3) and visualized through GraphPad Prism (version 9.3.0) and OriginPro 2024. Cohen&#x2019;s kappa was used to assess the level of inter-rater agreement between the two reviewers, with calculations performed using IBM SPSS Statistics 27. The results showed that the Kappa value was 0.336, and <italic>p</italic>&#x202F;&#x003C;&#x202F;0.01.</p>
</sec>
</sec>
<sec sec-type="results" id="sec12">
<label>3</label>
<title>Results</title>
<sec id="sec13">
<label>3.1</label>
<title>The basic characteristics of AI&#x2019;S responses to CNO-related questions</title>
<p>Based on the latest expert consensus, we proposed 16 questions related to CNO, covering basic knowledge, diagnosis, differential diagnosis, examination, and treatment. As can be seen from <xref ref-type="fig" rid="fig2">Figures 2A</xref>&#x2013;<xref ref-type="fig" rid="fig2">C</xref> and <xref ref-type="table" rid="tab1">Table 1</xref>, Doubao took the shortest time to answer questions in all three rounds of Q&#x0026;A, with response times of 5.09 (4.18, 5.84)s, 4.73 (3.94, 5.36)s, and 4.78 (4.26, 5.01)s, respectively. Kimi came next, with response times of 7.28 (5.99, 8.19)s, 7.05 (5.65, 8.04)s, and 7.15 (6.17, 8.02)s for the three rounds. Deepseek V3 took the longest time to answer these questions, with response times of 15.01 (12.39, 17.36)s, 15.93 (14.40, 17.57)s, and 15.84 (13.67, 17.24)s, respectively. From <xref ref-type="fig" rid="fig2">Figures 2D</xref>,<xref ref-type="fig" rid="fig2">E</xref> and <xref ref-type="table" rid="tab2">Table 2</xref>, it can be seen that in the three rounds of answers to CNO-related questions, Doubao provided the longest responses, with word counts of 804.5 (703.2, 1044.5), 778.5 (670.8,946.0), and 958.5 (694.5, 1329.2), respectively. Deepseek V3 ranked second, with word counts of 624.0 (432.0, 824.0), 675.5 (459.0, 785.2), and 679.0 (548.2, 874.8). Kimi1.5 had the shortest responses, with word counts of 439.0 (374.2, 683.5), 526.5 (441.8, 668.2), and 473.0 (268.5, 638.0).</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>The basic characteristics of AI&#x2019;s responses to CNO-related questions. <bold>(A&#x2013;C)</bold> Three sets of running results regarding the response time of different types of artificial intelligence to CNO-related questions. <bold>(D&#x2013;F)</bold> Three sets of running results regarding the word counts of different types of artificial intelligence to CNO-related questions. <bold>(G&#x2013;I)</bold> The results of three trials conducted by Reviewer 1 in inquiring about CNO-related questions to these three AI models. <bold>(J&#x2013;L)</bold> The results of three trials conducted by Reviewer 2 in inquiring about CNO-related questions to these three AI models. Statistical analysis was performed using non-parametric tests. ns, nonsignificant; &#x002A;<italic>p</italic>&#x202F;&#x003C;&#x202F;0.05; &#x002A;&#x002A;<italic>p</italic>&#x202F;&#x003C;&#x202F;0.01; &#x002A;&#x002A;&#x002A;<italic>p</italic>&#x202F;&#x003C;&#x202F;0.001; &#x002A;&#x002A;&#x002A;&#x002A;<italic>p</italic>&#x202F;&#x003C;&#x202F;0.0001.</p>
</caption>
<graphic xlink:href="frai-08-1629149-g002.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Box plots and bar graphs comparing reflection time and word count across three runs (first, second, third) for Deepset, Doubao, and Kimi. Panels A-C show reflection time, with significant differences indicated by p-values. Panels D-F display the number of words, with varying significance levels. Panels G-I and J-L present proportions of responses labeled as comprehensive, correct but inadequate, mixed, or completely incorrect, evaluated by two reviewers across three runs.</alt-text>
</graphic>
</fig>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Comparison of the response times of three AIs to CNO-related questions.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th>Response times</th>
<th align="center" valign="top">Deepseek V3</th>
<th align="center" valign="top">Doubao</th>
<th align="center" valign="top">Kimi 1.5</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">First run, median (IQR)</td>
<td align="center" valign="top">15.01 (12.39,17.36)</td>
<td align="center" valign="top">5.09 (4.18,5.84)</td>
<td align="center" valign="top">7.28 (5.99,8.19)</td>
</tr>
<tr>
<td align="left" valign="top">Second run, median (IQR)</td>
<td align="center" valign="top">15.93 (14.40,17.57)</td>
<td align="center" valign="top">4.73 (3.94,5.36)</td>
<td align="center" valign="top">7.05 (5.65,8.04)</td>
</tr>
<tr>
<td align="left" valign="top">Third run, median (IQR)</td>
<td align="center" valign="top">15.84 (13.67,17.24)</td>
<td align="center" valign="top">4.78 (4.26,5.01)</td>
<td align="center" valign="top">7.15 (6.17,8.02)</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Comparison of the word counts of the answers given by three AIs to CNO-related questions.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th>Word counts</th>
<th align="center" valign="top">Deepseek V3</th>
<th align="center" valign="top">Doubao</th>
<th align="center" valign="top">Kimi 1.5</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">First run, median (IQR)</td>
<td align="center" valign="top">624.0 (432.0,824.0)</td>
<td align="center" valign="top">804.5 (703.2,1044.5)</td>
<td align="center" valign="top">439.0 (374.2,683.5)</td>
</tr>
<tr>
<td align="left" valign="top">Second run, Median (IQR)</td>
<td align="center" valign="top">675.5 (459.0,785.2)</td>
<td align="center" valign="top">778.5 (670.8,946.0)</td>
<td align="center" valign="top">526.5 (441.8,668.2)</td>
</tr>
<tr>
<td align="left" valign="top">Third run, median (IQR)</td>
<td align="center" valign="top">679.0 (548.2,874.8)</td>
<td align="center" valign="top">958.5 (694.5,1329.2)</td>
<td align="center" valign="top">473.0 (268.5,638.0)</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In the first round of scoring by Reviewer 1 (<xref ref-type="fig" rid="fig2">Figure 2G</xref>): For Deepseek V3&#x2019;s scores, Correct but Inadequate and Comprehensive account for 18.75 and 81.25% respectively; For Doubao&#x2019;s scores, Mixed with Correct and Incorrect/Outdated Data and Comprehensive account for 12.5 and 87.5% respectively; For Kimi1.5&#x2019;s scores, Correct but Inadequate and Comprehensive account for 6.25 and 93.75%, respectively. In the second round of scoring by Reviewer 1 (<xref ref-type="fig" rid="fig2">Figure 2H</xref>): For Deepseek V3&#x2019;s scores, Mixed with Correct and Incorrect/Outdated Data, Correct but Inadequate, and Comprehensive account for 6.25, 12.5, and 81.25% respectively; For Doubao&#x2019;s scores, Correct but Inadequate and Comprehensive account for 6.25 and 93.75% respectively; For Kimi1.5&#x2019;s scores, Mixed with Correct and Incorrect/Outdated Data and Comprehensive account for 6.25 and 93.75%, respectively. In the third round of scoring by Reviewer 1 (<xref ref-type="fig" rid="fig2">Figure 2I</xref>): For Deepseek V3&#x2019;s scores, Correct but Inadequate and Comprehensive account for 25 and 75% respectively; For Doubao&#x2019;s scores, Mixed with Correct and Incorrect/Outdated Data, Correct but Inadequate, and Comprehensive account for 6.25, 6.25, and 87.5% respectively; For Kimi1.5&#x2019;s scores, Mixed with Correct and Incorrect/Outdated Data and Comprehensive account for 6.25 and 93.75%, respectively.</p>
<p>In the first round of scoring by Reviewer 2: For Deepseek V3&#x2019;s scores, Mixed with Correct and Incorrect/Outdated Data, Correct but Inadequate, and Comprehensive account for 6.25, 25, and 68.75% respectively; For Doubao&#x2019;s scores, Correct but Inadequate and Comprehensive account for 25 and 75% respectively; For Kimi1.5&#x2019;s scores, Correct but Inadequate and Comprehensive account for 18.75 and 81.25%, respectively. In the second round of scoring by Reviewer 2: For Deepseek V3&#x2019;s scores, Mixed with Correct and Incorrect/Outdated Data, Correct but Inadequate, and Comprehensive account for 6.25, 31.25, and 62.5% respectively; For Doubao&#x2019;s scores, Correct but Inadequate and Comprehensive account for 18.75 and 81.25% respectively; For Kimi1.5&#x2019;s scores, Mixed with Correct and Incorrect/Outdated Data, Correct but Inadequate, and Comprehensive account for 6.25, 12.5, and 81.25%, respectively. In the third round of scoring by Reviewer 2: For Deepseek V3&#x2019;s scores, Mixed with Correct and Incorrect/Outdated Data, Correct but Inadequate, and Comprehensive account for 6.25, 6.25, and 87.5% respectively; For Doubao&#x2019;s scores, Completely Incorrect, Correct but Inadequate, and Comprehensive account for 6.25, 6.25, and 75% respectively; For Kimi1.5&#x2019;s scores, Mixed with Correct and Incorrect/Outdated Data, Correct but Inadequate, and Comprehensive account for 6.25, 6.25, and 87.5%, respectively.</p>
</sec>
<sec id="sec14">
<label>3.2</label>
<title>Comparison of ratings between two reviewers</title>
<p>The questions answered by the three AI models were evaluated by two orthopedic doctors based on the expert consensus. When there were disagreements between the two doctors, a more experienced expert conducted a re-evaluation. As can be seen in <xref ref-type="fig" rid="fig3">Figure 3A&#x2013;C</xref>, in the first two rounds of Deepseek V3&#x2019;s operation, Reviewer 1 gave slightly higher scores than Reviewer 2, while in the third round, Reviewer 2 gave higher scores. However, there were no significant differences in the scores across the three rounds. From <xref ref-type="fig" rid="fig3">Figure 3D&#x2013;F</xref>, it can be seen that Reviewer 1 and Reviewer 2 gave similar scores to Doubao&#x2019;s answers, with no significant differences. In <xref ref-type="fig" rid="fig3">Figure 3G&#x2013;I</xref>, it can be observed that in the first round, Reviewer 1 gave a higher score than Reviewer 2, while in the subsequent two rounds, the scores were similar, with no significant differences across the three rounds.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Comparison of ratings between two reviewers by the three AI language models to questions related to CNO. <bold>(A&#x2013;C)</bold> The results of three trials conducted by two reviewers in asking CNO-related questions to Deepseek V3. (D-F) The results of three trials conducted by two reviewers in asking CNO-related questions to Doubao. <bold>(G&#x2013;I)</bold> The results of three trials conducted by two reviewers in asking CNO-related questions to Kimi 1.5. Statistical analysis was performed using <italic>t</italic> test. ns, not significant.</p>
</caption>
<graphic xlink:href="frai-08-1629149-g003.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Nine bar charts compare grades given by Reviewer 1 and Reviewer 2 across three runs for three different systems: Deepseek (A, B, C), Doubao (D, E, F), and Kimi (G, H, I). Each panel shows no significant differences with p-values above 0.3, indicating consistency between reviewers.</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec15">
<label>3.3</label>
<title>The score distribution of Deepseek V3, Doubao, and Kimi1.5 in answering CNO-related questions</title>
<p><xref ref-type="supplementary-material" rid="SM1">Supplementary Tables S1&#x2013;S3</xref> present the scores given by the two reviewers. From <xref ref-type="supplementary-material" rid="SM1">Supplementary Table S4</xref> and <xref ref-type="fig" rid="fig4">Figure 4</xref>, it can be seen that Reviewer 1&#x2019;s score distributions for Deepseek V3, Doubao, and Kimi1.5 in the first and second rounds were all 4.00 (4.00, 4.00) (<xref ref-type="fig" rid="fig4">Figures 4A</xref>,<xref ref-type="fig" rid="fig4">B</xref>). In the third round of scoring, the scores for Deepseek V3, Doubao, and Kimi1.5 were 4.00 (3.75, 4.00), 4.00 (4.00, 4.00), and 4.00 (4.00, 4.00), respectively (<xref ref-type="fig" rid="fig4">Figure 4C</xref>).</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Comparison among Deepseek V3, Doubao, and Kimi 1.5 in answering CNO-related questions. <bold>(A&#x2013;C)</bold> The results of three trials conducted by Reviewer 1 in inquiring about CNO-related questions to these three AI models. <bold>(D&#x2013;F)</bold> The results of three trials conducted by Reviewer 2 in inquiring about CNO-related questions to these three AI models. Statistical analysis was performed using Kruskal-Wallis. ns, nonsignificant.</p>
</caption>
<graphic xlink:href="frai-08-1629149-g004.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Bar graphs comparing grades from three runs for two reviewers using three systems: Deepseek, Doubao, and Kimi. All graphs show similar grades with statistics marked "ns" for non-significance, and p-values displayed above bars. Each graph shows individual data points with error bars.</alt-text>
</graphic>
</fig>
<p>For Reviewer 2, the score distributions for Deepseek V3, Doubao, and Kimi1.5 in the first round were 4.00 (3.00, 4.00), 4.00 (3.75, 4.00), and 4.00 (4.00, 4.00), respectively (<xref ref-type="fig" rid="fig4">Figure 4D</xref>). In the second round, their score distributions were 4.00 (3.00, 4.00), 4.00 (4.00, 4.00), and 4.00 (4.00, 4.00), respectively (<xref ref-type="fig" rid="fig4">Figure 4E</xref>). In the third round, all their score distributions were 4.00 (4.00, 4.00) (<xref ref-type="fig" rid="fig4">Figure 4F</xref>).</p>
<p>From <xref ref-type="table" rid="tab3">Table 3</xref>, it can be observed that in the first round of scoring by Reviewer 1, Kimi1.5 received the highest average score of 3.938&#x202F;&#x00B1;&#x202F;0.342, followed by Deepseek V3 (3.812&#x202F;&#x00B1;&#x202F;0.403), with Doubao (3.75&#x202F;&#x00B1;&#x202F;0.683) receiving the lowest score. In the second round of scoring, Doubao, Kimi1.5, and Deepseek V3 achieved scores of 3.875&#x202F;&#x00B1;&#x202F;0.5, 3.875&#x202F;&#x00B1;&#x202F;0.683, and 3.75&#x202F;&#x00B1;&#x202F;0.557, respectively. In the third round of scoring, Doubao, Kimi1.5, and Deepseek V3 obtained scores of 3.812&#x202F;&#x00B1;&#x202F;0.544, 3.875&#x202F;&#x00B1;&#x202F;0.873, and 3.75&#x202F;&#x00B1;&#x202F;0.447, respectively. It can be seen that in the first rounds of scoring by Reviewer 2, Kimi1.5 received the highest scores (3.812&#x202F;&#x00B1;&#x202F;0.602), but in the second round, Doubao received the highest scores (3.812&#x202F;&#x00B1;&#x202F;0.403). In the third round of scoring, Kimi1.5 received the highest score of 3.812&#x202F;&#x00B1;&#x202F;0.704.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>The results of three trials conducted by two reviewers in asking CNO-related questions to the three AI language models.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th>Reviewers&#x2019; grades</th>
<th/>
<th align="center" valign="top">DeepseekV3</th>
<th align="center" valign="top">Doubao</th>
<th align="center" valign="top">Kimi 1.5</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top" rowspan="3">Reviewer 1 grades</td>
<td align="center" valign="top">First run, mean&#x00B1;SD</td>
<td align="center" valign="top">3.812&#x202F;&#x00B1;&#x202F;0.403</td>
<td align="center" valign="top">3.75&#x202F;&#x00B1;&#x202F;0.683</td>
<td align="center" valign="top">3.938&#x202F;&#x00B1;&#x202F;0.342</td>
</tr>
<tr>
<td align="center" valign="top">Second run, mean&#x00B1;SD</td>
<td align="center" valign="top">3.75&#x202F;&#x00B1;&#x202F;0.557</td>
<td align="center" valign="top">3.875&#x202F;&#x00B1;&#x202F;0.5</td>
<td align="center" valign="top">3.875&#x202F;&#x00B1;&#x202F;0.683</td>
</tr>
<tr>
<td align="center" valign="top">Third run, mean&#x00B1;SD</td>
<td align="center" valign="top">3.75&#x202F;&#x00B1;&#x202F;0.447</td>
<td align="center" valign="top">3.812&#x202F;&#x00B1;&#x202F;0.544</td>
<td align="center" valign="top">3.875&#x202F;&#x00B1;&#x202F;0.873</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="3">Reviewer 2 grades</td>
<td align="center" valign="top">First run, mean&#x00B1;SD</td>
<td align="center" valign="top">3.625&#x202F;&#x00B1;&#x202F;0.619</td>
<td align="center" valign="top">3.75&#x202F;&#x00B1;&#x202F;0.447</td>
<td align="center" valign="top">3.812&#x202F;&#x00B1;&#x202F;0.602</td>
</tr>
<tr>
<td align="center" valign="top">Second run, mean&#x00B1;SD</td>
<td align="center" valign="top">3.562&#x202F;&#x00B1;&#x202F;0.629</td>
<td align="center" valign="top">3.812&#x202F;&#x00B1;&#x202F;0.403</td>
<td align="center" valign="top">3.75&#x202F;&#x00B1;&#x202F;0.602</td>
</tr>
<tr>
<td align="center" valign="top">Third run, mean&#x00B1;SD</td>
<td align="center" valign="top">3.812&#x202F;&#x00B1;&#x202F;0.544</td>
<td align="center" valign="top">3.75&#x202F;&#x00B1;&#x202F;0.775</td>
<td align="center" valign="top">3.812&#x202F;&#x00B1;&#x202F;0.704</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="sec16">
<label>4</label>
<title>Discussion</title>
<p>In the field of patient care, AI can serve as a &#x201C;highly efficient auxiliary tool&#x201D; for clinicians: On one hand, when doctors in primary hospitals treat patients with suspected CNO, AI can quickly generate an initial diagnosis differential checklist and examination recommendations based on the patients&#x2019; symptoms (<xref ref-type="bibr" rid="ref2">Apornvirat et al., 2024</xref>). This helps shorten the time for diagnostic decision-making and reduce missed diagnoses. On the other hand, when patients report side effects during follow-ups, AI can immediately provide &#x201C;directions for preliminary treatment plan adjustments&#x201D; to assist doctors in responding quickly (<xref ref-type="bibr" rid="ref6">Hartman-Kenzler et al., 2024</xref>). This avoids risks caused by long intervals between follow-up visits and indirectly improves the efficiency and safety of patient care. In the field of patient education, AI can provide &#x201C;personalized and accessible health guidance&#x201D;: Traditional patient education mostly relies on general manuals, which are difficult to adapt to individual conditions (<xref ref-type="bibr" rid="ref2">Apornvirat et al., 2024</xref>). In contrast, AI can integrate a patient&#x2019;s medical history (e.g., a history of diabetes) (<xref ref-type="bibr" rid="ref5">Davis et al., 2023</xref>) and treatment plan (e.g., hormone medication) (<xref ref-type="bibr" rid="ref15">Natchagande et al., 2024</xref>), while also issuing warnings about &#x201C;signals requiring emergency medical attention&#x201D; (e.g., sudden vision loss, worsening eye pain). This helps patients understand the logic of diagnosis and treatment, improve treatment adherence, and reduce treatment interruptions or neglect of risks caused by cognitive biases.</p>
<p>In summary, in this study, among Deepseek V3, Doubao, and Kimi1.5, Deepseek V3 took the longest time to think when answering questions, while Doubao responded the fastest. Based on the recommendations outlined in clinical expert guidelines, both reviewers confirmed that an answer to each question could be found in the expert consensus. Since each expert conducted independent evaluations, the majority of the scores were consistent. However, individual cognitive characteristics of experts&#x2014;such as personal clinical experience and attention allocation&#x2014;may also lead to discrepancies in scoring. This is an inherent attribute of human subjective judgment and constitutes a bias that is controllable but cannot be completely eliminated. Additionally, among all the answers, Doubao provided the most words in its responses, whereas Kimi1.5 provided the fewest. There were no significant differences in the scores given by the two reviewers for the answers provided by these three AIs, and the scores all fell between 3 and 4, indicating the accuracy of the AI&#x2019;s responses is highly satisfactory. More importantly, the high scores were maintained across all three rounds, demonstrating good repeatability in the AI&#x2019;s responses. However, for the fourth question (&#x201C;Which parts of the body are most frequently affected by chronic non-bacterial osteitis in adults?&#x201D;), none of the three AIs received a score of 4 and Doubao provided an incorrect answer to the fourth question in the third round. Regarding this issue, the latest guidelines suggest that in adults, CNO predominantly manifests in the anterior chest wall, including the clavicle, upper ribs, and sternum, which are the most commonly affected sites (<xref ref-type="bibr" rid="ref23">Winter et al., 2025</xref>).</p>
<p>Currently, there have been numerous studies on the application of ChatGPT in medicine, as well as comparative analyses of its research performance in the medical field against different AI models (<xref ref-type="bibr" rid="ref3">Bradshaw, 2023</xref>). Ozan Yaz&#x0131;c&#x0131; et al. compared ChatGPT and Perplexity in terms of treatment response and reliability assessment for rectal cancer, finding that Perplexity had higher accuracy than ChatGPT (<xref ref-type="bibr" rid="ref24">Yazici et al., 2023</xref>). <xref ref-type="bibr" rid="ref26">Zhou et al. (2025</xref>) evaluated the performance of ChatGPT and DeepSeek in generating educational materials for patients undergoing spinal surgery, discovering that DeepSeek-R1 produced the most credible answers, although the AI models generally received moderate DISCERN scores. Even within the same AI model, differences can exist between versions. Hu et al. examined two versions of ChatGPT in relation to questions about <italic>Helicobacter pylori</italic>, finding that both versions had high accuracy, with ChatGPT 3.5 performing well in the areas of indications, treatment, and gut microbiota, and ChatGPT 4 excelling in diagnosis, gastric cancer, and prevention (<xref ref-type="bibr" rid="ref9">Hu et al., 2024</xref>). Additionally, researchers have found that the fusion of emotion-aware embeddings in large language models (including Flan-T5, Llama 2, DeepSeek-R1, and ChatGPT 4) is applied to intelligent response generation (<xref ref-type="bibr" rid="ref17">Rasool et al., 2025</xref>). In this study, when answering relevant questions, we can observe that each AI model has its own characteristics. Deepseek V3 reminds users after each answer that the response is for reference only. Doubao expands on other related questions after answering the initial query. Kimi1.5 not only expands on related questions but also displays the answer sources on the right-hand side.</p>
<p>Although these three AIs scored relatively high, they all have limitations. (1) their sources are not based on clinical evidence but rather on internet sources, many of which are not professional literature guidelines. (2) When new clinical guidelines are updated, AIs do not provide answers according to the latest clinical guidelines or expert consensus, which can easily lead to outdated information and incorrect answers. Furthermore, when AI models are updated, their accuracy also changes. In future analyses, we will also include re-testing after major model updates. (3) The conclusions of this study are only applicable to Chinese Q&#x0026;A scenarios. Due to the variations in treatment guidelines across different specialties and regions worldwide, giving priority to the guidelines of those regions poses a significant challenge for AI. (4) This study only based its inquiries on 16 questions. Though these questions cover such aspects as disease definition, diagnosis, and treatment, the number is relatively limited. Moreover, the grading system is based on a 4-point scale. Though it has been used in relevant literature, the non-refined scale has the drawback of missing minor errors. (5) Semantic similarity-based evaluation indicators such as BERTScore have not been applied in this study, yet these indicators are an important factor for conducting objective comparisons between models and driving model performance improvement.</p>
<p>Furthermore, patient inquiries in the real world may differ from expert-developed questions in terms of complexity and wording. In future studies, the number of questions should be further increased, and the wording used by patients when they ask questions via AI should be collected. Nowadays, the use frequency of AI in clinical work is increasing. Future research can shift from static Q&#x0026;A to dynamic clinical reasoning by designing case-based scenarios&#x2014;requiring models to interpret symptoms, order tests, and propose treatment plans, thereby moving from factual retrieval to diagnostic reasoning (<xref ref-type="bibr" rid="ref18">Rasool et al., 2020</xref>). Therefore, we suggest that in the future, AI-generated medical answers should undergo appropriate certification and regular review. This approach can provide patients with more accurate disease information, enhance their understanding of the disease, and improve their management of the condition.</p>
</sec>
<sec sec-type="conclusions" id="sec17">
<label>5</label>
<title>Conclusion</title>
<p>Overall, through this study, we found that these three AI models demonstrate relatively high accuracy and reproducibility when answering CNO-related static questions based on their training data. With no significant differences observed. We believe that with the continuous development of AI prediction models, Deepseek V3, Doubao, and Kimi1.5 have the potential to serve as supplementary tools in addition to expert consensus and guidelines.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec18">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>, further inquiries can be directed to the corresponding author/s.</p>
</sec>
<sec sec-type="ethics-statement" id="sec19">
<title>Ethics statement</title>
<p>Ethical review and approval was not required for the study on human participants in accordance with the local legislation and institutional requirements. Written informed consent from the participants was not required to participate in this study in accordance with the national legislation and the institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="sec20">
<title>Author contributions</title>
<p>ZZ: Visualization, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. JX: Methodology, Visualization, Writing &#x2013; review &#x0026; editing. LZ: Methodology, Visualization, Writing &#x2013; review &#x0026; editing. CY: Methodology, Visualization, Writing &#x2013; review &#x0026; editing. FL: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing.</p>
</sec>
<sec sec-type="funding-information" id="sec21">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research and/or publication of this article.</p>
</sec>
<sec sec-type="COI-statement" id="sec22">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec23">
<title>Generative AI statement</title>
<p>The authors declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="sec24">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec25">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/frai.2025.1629149/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/frai.2025.1629149/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.zip" id="SM1" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_2.zip" id="SM2" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abbasi</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>Artificial intelligence improves breast cancer screening in study</article-title>. <source>JAMA</source> <volume>323</volume>:<fpage>499</fpage>. doi: <pub-id pub-id-type="doi">10.1001/jama.2020.0370</pub-id>, PMID: <pub-id pub-id-type="pmid">32044919</pub-id></citation></ref>
<ref id="ref2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Apornvirat</surname> <given-names>S.</given-names></name> <name><surname>Thinpanja</surname> <given-names>W.</given-names></name> <name><surname>Damrongkiet</surname> <given-names>K.</given-names></name> <name><surname>Benjakul</surname> <given-names>N.</given-names></name> <name><surname>Laohawetwanit</surname> <given-names>T.</given-names></name></person-group> (<year>2024</year>). <article-title>ChatGPT for histopathologic diagnosis</article-title>. <source>Ann. Diagn. Pathol.</source> <volume>73</volume>:<fpage>152365</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.anndiagpath.2024.152365</pub-id>, PMID: <pub-id pub-id-type="pmid">39098307</pub-id></citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bradshaw</surname> <given-names>J. C.</given-names></name></person-group> (<year>2023</year>). <article-title>The ChatGPT era: artificial intelligence in emergency medicine</article-title>. <source>Ann. Emerg. Med.</source> <volume>81</volume>, <fpage>764</fpage>&#x2013;<lpage>765</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.annemergmed.2023.01.022</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chadebecq</surname> <given-names>F.</given-names></name> <name><surname>Lovat</surname> <given-names>L. B.</given-names></name> <name><surname>Stoyanov</surname> <given-names>D.</given-names></name></person-group> (<year>2023</year>). <article-title>Artificial intelligence and automation in endoscopy and surgery</article-title>. <source>Nat. Rev. Gastroenterol. Hepatol.</source> <volume>20</volume>, <fpage>171</fpage>&#x2013;<lpage>182</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41575-022-00701-y</pub-id>, PMID: <pub-id pub-id-type="pmid">36352158</pub-id></citation></ref>
<ref id="ref5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Davis</surname> <given-names>G. M.</given-names></name> <name><surname>Shao</surname> <given-names>H.</given-names></name> <name><surname>Pasquel</surname> <given-names>F. J.</given-names></name></person-group> (<year>2023</year>). <article-title>AI-supported insulin dosing for type 2 diabetes</article-title>. <source>Nat. Med.</source> <volume>29</volume>, <fpage>2414</fpage>&#x2013;<lpage>2415</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41591-023-02573-4</pub-id>, PMID: <pub-id pub-id-type="pmid">37821684</pub-id></citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hartman-Kenzler</surname> <given-names>J.</given-names></name> <name><surname>Torres</surname> <given-names>J.</given-names></name> <name><surname>Alami-Harandi</surname> <given-names>A.</given-names></name> <name><surname>Miller</surname> <given-names>C.</given-names></name> <name><surname>Park</surname> <given-names>J.</given-names></name> <name><surname>Berg</surname> <given-names>W.</given-names></name></person-group> (<year>2024</year>). <article-title>MP47-13 ChatGPT explains testosterone therapy: accurate answers with questionable references</article-title>. <source>J. Urol.</source> <volume>211</volume>:<fpage>e768</fpage>. doi: <pub-id pub-id-type="doi">10.1097/01.JU.0001008880.11564.10.13</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hedrich</surname> <given-names>C. M.</given-names></name> <name><surname>Beresford</surname> <given-names>M. W.</given-names></name> <name><surname>Dedeoglu</surname> <given-names>F.</given-names></name> <name><surname>Hahn</surname> <given-names>G.</given-names></name> <name><surname>Hofmann</surname> <given-names>S. R.</given-names></name> <name><surname>Jansson</surname> <given-names>A. F.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Gathering expert consensus to inform a proposed trial in chronic nonbacterial osteomyelitis (CNO)</article-title>. <source>Clin. Immunol.</source> <volume>251</volume>:<fpage>109344</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.clim.2023.109344</pub-id>, PMID: <pub-id pub-id-type="pmid">37098355</pub-id></citation></ref>
<ref id="ref8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Higuchi</surname> <given-names>M.</given-names></name> <name><surname>Nagata</surname> <given-names>T.</given-names></name> <name><surname>Suzuki</surname> <given-names>J.</given-names></name> <name><surname>Matsumura</surname> <given-names>Y.</given-names></name> <name><surname>Suzuki</surname> <given-names>H.</given-names></name></person-group> (<year>2024</year>). <article-title>1194P development of a novel artificial intelligence (AI) algorithm to detect pulmonary nodules on chest radiography</article-title>. <source>Ann. Oncol.</source> <volume>35</volume>:<fpage>S770</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.annonc.2024.08.1254</pub-id>, PMID: <pub-id pub-id-type="pmid">40955376</pub-id></citation></ref>
<ref id="ref9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>Y.</given-names></name> <name><surname>Lai</surname> <given-names>Y.</given-names></name> <name><surname>Liao</surname> <given-names>F.</given-names></name> <name><surname>Shu</surname> <given-names>X.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Du</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Assessing accuracy of ChatGPT on addressing <italic>Helicobacter pylori</italic> infection-related questions: a national survey and comparative study</article-title>. <source>Helicobacter</source> <volume>29</volume>:<fpage>e13116</fpage>. doi: <pub-id pub-id-type="doi">10.1111/hel.13116</pub-id>, PMID: <pub-id pub-id-type="pmid">39080910</pub-id></citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jansson</surname> <given-names>A. F.</given-names></name> <name><surname>Grote</surname> <given-names>V.</given-names></name><collab id="coll1">ESPED Study Group</collab></person-group> (<year>2011</year>). <article-title>Nonbacterial osteitis in children: data of a German incidence surveillance study</article-title>. <source>Acta Paediatr.</source> <volume>100</volume>, <fpage>1150</fpage>&#x2013;<lpage>1157</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1651-2227.2011.02205.x</pub-id>, PMID: <pub-id pub-id-type="pmid">21352353</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kulkarni</surname> <given-names>P. A.</given-names></name> <name><surname>Singh</surname> <given-names>H.</given-names></name></person-group> (<year>2023</year>). <article-title>Artificial intelligence in clinical diagnosis: opportunities, challenges, and hype</article-title>. <source>JAMA</source> <volume>330</volume>, <fpage>317</fpage>&#x2013;<lpage>318</lpage>. doi: <pub-id pub-id-type="doi">10.1001/jama.2023.11440</pub-id>, PMID: <pub-id pub-id-type="pmid">37410477</pub-id></citation></ref>
<ref id="ref12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lai</surname> <given-names>Y.</given-names></name> <name><surname>Liao</surname> <given-names>F.</given-names></name> <name><surname>Zhao</surname> <given-names>J.</given-names></name> <name><surname>Zhu</surname> <given-names>C.</given-names></name> <name><surname>Hu</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name></person-group> (<year>2024</year>). <article-title>Exploring the capacities of ChatGPT: a comprehensive evaluation of its accuracy and repeatability in addressing <italic>Helicobacter pylori</italic>-related queries</article-title>. <source>Helicobacter</source> <volume>29</volume>:<fpage>e13078</fpage>. doi: <pub-id pub-id-type="doi">10.1111/hel.13078</pub-id>, PMID: <pub-id pub-id-type="pmid">38867649</pub-id></citation></ref>
<ref id="ref13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lenert</surname> <given-names>A.</given-names></name> <name><surname>Ferguson</surname> <given-names>P. J.</given-names></name></person-group> (<year>2020</year>). <article-title>Comparing children and adults with chronic nonbacterial osteomyelitis</article-title>. <source>Curr. Opin. Rheumatol.</source> <volume>32</volume>, <fpage>421</fpage>&#x2013;<lpage>426</lpage>. doi: <pub-id pub-id-type="doi">10.1097/BOR.0000000000000734</pub-id>, PMID: <pub-id pub-id-type="pmid">32744822</pub-id></citation></ref>
<ref id="ref14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>L&#x00FC;scher</surname> <given-names>T. F.</given-names></name> <name><surname>Wenzl</surname> <given-names>F. A.</given-names></name> <name><surname>D&#x2019;Ascenzo</surname> <given-names>F.</given-names></name> <name><surname>Friedman</surname> <given-names>P. A.</given-names></name> <name><surname>Antoniades</surname> <given-names>C.</given-names></name></person-group> (<year>2024</year>). <article-title>Artificial intelligence in cardiovascular medicine: clinical applications</article-title>. <source>Eur. Heart J.</source> <volume>45</volume>, <fpage>4291</fpage>&#x2013;<lpage>4304</lpage>. doi: <pub-id pub-id-type="doi">10.1093/eurheartj/ehae465</pub-id>, PMID: <pub-id pub-id-type="pmid">39158472</pub-id></citation></ref>
<ref id="ref15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Natchagande</surname> <given-names>G.</given-names></name> <name><surname>Vinh-Hung</surname> <given-names>V.</given-names></name> <name><surname>Verschraegen</surname> <given-names>C.</given-names></name></person-group> (<year>2024</year>). <article-title>Testosterone hallucination by artificial intelligence? ENZAMET trial, in the spring of 2023</article-title>. <source>Cancer Research Statistics Treatment</source> <volume>7</volume>, <fpage>268</fpage>&#x2013;<lpage>269</lpage>. doi: <pub-id pub-id-type="doi">10.4103/crst.crst_83_24</pub-id></citation></ref>
<ref id="ref16"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Nezhad</surname> <given-names>S. B.</given-names></name> <name><surname>Agrawal</surname> <given-names>A.</given-names></name> <name><surname>Pokharel</surname> <given-names>R.</given-names></name></person-group> (<year>2024</year>). <article-title>Beyond data quantity: key factors driving performance in multilingual language models</article-title>. <volume>323</volume>, <fpage>499</fpage>&#x2013;<lpage>500</lpage>. doi: <pub-id pub-id-type="doi">10.48550/ARXIV.2412.12500</pub-id></citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rasool</surname> <given-names>A.</given-names></name> <name><surname>Shahzad</surname> <given-names>M. I.</given-names></name> <name><surname>Aslam</surname> <given-names>H.</given-names></name> <name><surname>Chan</surname> <given-names>V.</given-names></name> <name><surname>Arshad</surname> <given-names>M. A.</given-names></name></person-group> (<year>2025</year>). <article-title>Emotion-aware embedding fusion in large language models (flan-T5, llama 2, DeepSeek-R1, and ChatGPT 4) for intelligent response generation</article-title>. <source>AI</source> <volume>6</volume>:<fpage>56</fpage>. doi: <pub-id pub-id-type="doi">10.3390/ai6030056</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Rasool</surname> <given-names>A.</given-names></name> <name><surname>Tao</surname> <given-names>R.</given-names></name> <name><surname>Kashif</surname> <given-names>K.</given-names></name> <name><surname>Khan</surname> <given-names>W.</given-names></name> <name><surname>Agbedanu</surname> <given-names>P.</given-names></name> <name><surname>Choudhry</surname> <given-names>N.</given-names></name></person-group> (<year>2020</year>). &#x201C;<article-title>Statistic solution for machine learning to analyze heart disease data</article-title>&#x201D; in <source>Proceedings of the 2020 12th international conference on machine learning and computing</source> (<publisher-loc>Shenzhen China</publisher-loc>: <publisher-name>ACM</publisher-name>), <fpage>134</fpage>&#x2013;<lpage>139</lpage>.</citation></ref>
<ref id="ref19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schnabel</surname> <given-names>A.</given-names></name> <name><surname>Range</surname> <given-names>U.</given-names></name> <name><surname>Hahn</surname> <given-names>G.</given-names></name> <name><surname>Siepmann</surname> <given-names>T.</given-names></name> <name><surname>Berner</surname> <given-names>R.</given-names></name> <name><surname>Hedrich</surname> <given-names>C. M.</given-names></name></person-group> (<year>2016</year>). <article-title>Unexpectedly high incidences of chronic non-bacterial as compared to bacterial osteomyelitis in children</article-title>. <source>Rheumatol. Int.</source> <volume>36</volume>, <fpage>1737</fpage>&#x2013;<lpage>1745</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s00296-016-3572-6</pub-id>, PMID: <pub-id pub-id-type="pmid">27730289</pub-id></citation></ref>
<ref id="ref20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>R.</given-names></name> <name><surname>De Donato</surname> <given-names>L.</given-names></name> <name><surname>Bes&#x0306;inovi&#x0107;</surname> <given-names>N.</given-names></name> <name><surname>Flammini</surname> <given-names>F.</given-names></name> <name><surname>Goverde</surname> <given-names>R. M. P.</given-names></name> <name><surname>Lin</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>A literature review of artificial intelligence applications in railway systems</article-title>. <source>Transport. Res. Part C</source> <volume>140</volume>:<fpage>103679</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.trc.2022.103679</pub-id></citation></ref>
<ref id="ref21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Varghese</surname> <given-names>C.</given-names></name> <name><surname>Harrison</surname> <given-names>E. M.</given-names></name> <name><surname>O&#x2019;Grady</surname> <given-names>G.</given-names></name> <name><surname>Topol</surname> <given-names>E. J.</given-names></name></person-group> (<year>2024</year>). <article-title>Artificial intelligence in surgery</article-title>. <source>Nat. Med.</source> <volume>30</volume>, <fpage>1257</fpage>&#x2013;<lpage>1268</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41591-024-02970-3</pub-id>, PMID: <pub-id pub-id-type="pmid">38740998</pub-id></citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wei</surname> <given-names>W.</given-names></name> <name><surname>Piuri</surname> <given-names>V.</given-names></name> <name><surname>Pedrycz</surname> <given-names>W.</given-names></name> <name><surname>Ahmed</surname> <given-names>S. H.</given-names></name></person-group> (<year>2022</year>). <article-title>Special issue on artificial intelligence-of-things (AIoT): opportunities, challenges, and solutions-part I: artificial intelligence applications in various fields</article-title>. <source>Futur. Gener. Comput. Syst.</source> <volume>137</volume>, <fpage>216</fpage>&#x2013;<lpage>218</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.future.2022.07.018</pub-id></citation></ref>
<ref id="ref23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Winter</surname> <given-names>E.</given-names></name> <name><surname>Dekkers</surname> <given-names>O.</given-names></name> <name><surname>Andreasen</surname> <given-names>C.</given-names></name> <name><surname>D&#x2019;Angelo</surname> <given-names>S.</given-names></name> <name><surname>Appelman-Dijkstra</surname> <given-names>N.</given-names></name> <name><surname>Appenzeller</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Expert consensus recommendations for the diagnosis and treatment of chronic non-bacterial osteitis (CNO) in adults</article-title>. <source>Ann. Rheum. Dis.</source> <volume>84</volume>, <fpage>169</fpage>&#x2013;<lpage>187</lpage>. doi: <pub-id pub-id-type="doi">10.1136/ard-2024-226446</pub-id></citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yazici</surname> <given-names>O.</given-names></name> <name><surname>Yucel</surname> <given-names>K. B.</given-names></name> <name><surname>Sutcuoglu</surname> <given-names>O.</given-names></name></person-group> (<year>2023</year>). <article-title>2066P evaluation of the quality and reliability of ChatGPT and perplexity&#x2019;s responses about rectal cancer</article-title>. <source>Ann. Oncol.</source> <volume>34</volume>:<fpage>S1090</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.annonc.2023.09.848</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>D. Y.</given-names></name> <name><surname>McCann</surname> <given-names>L.</given-names></name> <name><surname>Hahn</surname> <given-names>G.</given-names></name> <name><surname>Hedrich</surname> <given-names>C. M.</given-names></name></person-group> (<year>2021</year>). <article-title>Chronic nonbacterial osteomyelitis (CNO) and chronic recurrent multifocal osteomyelitis (CRMO)</article-title>. <source>J. Transl. Autoimmun.</source> <volume>4</volume>:<fpage>100095</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jtauto.2021.100095</pub-id>, PMID: <pub-id pub-id-type="pmid">33870159</pub-id></citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>M.</given-names></name> <name><surname>Pan</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Song</surname> <given-names>X.</given-names></name> <name><surname>Zhou</surname> <given-names>Y.</given-names></name></person-group> (<year>2025</year>). <article-title>Evaluating AI-generated patient education materials for spinal surgeries: comparative analysis of readability and DISCERN quality across ChatGPT and deepseek models</article-title>. <source>Int. J. Med. Inform.</source> <volume>198</volume>:<fpage>105871</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2025.105871</pub-id>, PMID: <pub-id pub-id-type="pmid">40107040</pub-id></citation></ref>
</ref-list>
</back>
</article>