<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Cell Dev. Biol.</journal-id>
<journal-title>Frontiers in Cell and Developmental Biology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Cell Dev. Biol.</abbrev-journal-title>
<issn pub-type="epub">2296-634X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1642539</article-id>
<article-id pub-id-type="doi">10.3389/fcell.2025.1642539</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Cell and Developmental Biology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Multimodal reasoning agent for enhanced ophthalmic decision-making: a preliminary real-world clinical validation</article-title>
<alt-title alt-title-type="left-running-head">Zhuang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fcell.2025.1642539">10.3389/fcell.2025.1642539</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Zhuang</surname>
<given-names>Yijing</given-names>
</name>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1082260/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Fang</surname>
<given-names>Dong</given-names>
</name>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1340811/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Li</surname>
<given-names>Pengfeng</given-names>
</name>
<xref ref-type="author-notes" rid="fn001">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3136190/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bai</surname>
<given-names>Bingyu</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/3137671/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hei</surname>
<given-names>Xiangqing</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/1664825/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Feng</surname>
<given-names>Lujia</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/1605966/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Wangting</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/1991011/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhang</surname>
<given-names>Shaochong</given-names>
</name>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2911335/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
</contrib-group>
<aff>Shenzhen Eye Hospital, Shenzhen Eye Institute, <institution>Jinan University</institution>, <addr-line>Shenzhen</addr-line>, <addr-line>Guangdong</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2152748/overview">Huihui Fang</ext-link>, Nanyang Technological University, Singapore</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1076968/overview">Gilbert Yong San Lim</ext-link>, SingHealth, Singapore</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2362953/overview">Hanyi Yu</ext-link>, South China University of Technology, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Shaochong Zhang, <email>shaochongzhang@outlook.com</email>
</corresp>
<fn fn-type="equal" id="fn001">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>23</day>
<month>07</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>13</volume>
<elocation-id>1642539</elocation-id>
<history>
<date date-type="received">
<day>06</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>10</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Zhuang, Fang, Li, Bai, Hei, Feng, Li and Zhang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Zhuang, Fang, Li, Bai, Hei, Feng, Li and Zhang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Although large language models (LLMs) show significant potential in clinical practice, accurate diagnosis and treatment planning in ophthalmology require multimodal integration of imaging, clinical history, and guideline-based knowledge. Current LLMs predominantly focus on unimodal language tasks and face limitations in specialized ophthalmic diagnosis due to domain knowledge gaps, hallucination risks, and inadequate alignment with clinical workflows. This study introduces a structured reasoning agent (ReasonAgent) that integrates a multimodal visual analysis module, a knowledge retrieval module, and a diagnostic reasoning module to address the limitations of current AI systems in ophthalmic decision-making. Validated on 30 real-world ophthalmic cases (27 common and 3 rare diseases), ReasonAgent demonstrated diagnostic accuracy comparable to ophthalmology residents (<italic>&#x3b2;</italic> &#x3d; &#x2212;0.07, <italic>p</italic> &#x3d; 0.65). However, in treatment planning, it significantly outperformed both GPT-4o (<italic>&#x3b2;</italic> &#x3d; 0.49, <italic>p</italic> &#x3d; 0.01) and residents (<italic>&#x3b2;</italic> &#x3d; 1.71, <italic>p</italic> &#x3c; 0.001), particularly excelling in rare disease scenarios (all <italic>p</italic> &#x3c; 0.05). While GPT-4o showed vulnerabilities in rare cases (90.48% low diagnostic scores), ReasonAgent&#x2019;s hybrid design mitigated errors through structured reasoning. Statistical analysis identified significant case-level heterogeneity (diagnosis ICC &#x3d; 0.28), highlighting the need for domain-specific AI solutions in complex clinical contexts. This framework establishes a novel paradigm for domain-specific AI in real-world clinical practice, demonstrating the potential of modularized architectures to advance decision fidelity through human-aligned reasoning pathways.</p>
</abstract>
<kwd-group>
<kwd>artificial intelligence</kwd>
<kwd>large language models</kwd>
<kwd>reasoning agent</kwd>
<kwd>GPT-4o</kwd>
<kwd>ocular diseases</kwd>
</kwd-group>
<contract-num rid="cn001">KCXFZ20211020163813019</contract-num>
<contract-num rid="cn002">2022A1515111155</contract-num>
<contract-sponsor id="cn001">Shenzhen Science and Technology Innovation Program<named-content content-type="fundref-id">10.13039/501100017610</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">Basic and Applied Basic Research Foundation of Guangdong Province<named-content content-type="fundref-id">10.13039/501100021171</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Molecular and Cellular Pathology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>The integration of artificial intelligence (AI) into ophthalmology has demonstrated transformative potential in automating image analysis (<xref ref-type="bibr" rid="B8">Feng et al., 2025</xref>; <xref ref-type="bibr" rid="B24">Rao et al., 2023</xref>), streamlining diagnostic workflows (<xref ref-type="bibr" rid="B5">Choi et al., 2024</xref>; <xref ref-type="bibr" rid="B34">Waisberg et al., 2024</xref>), and enhancing clinical decision-making (<xref ref-type="bibr" rid="B7">Delsoz et al., 2023</xref>; <xref ref-type="bibr" rid="B30">Tan et al., 2024</xref>). Numerous AI systems have been developed to tackle different tasks such as interpreting ophthalmic images, including fundus photography (<xref ref-type="bibr" rid="B12">Gulshan et al., 2016</xref>; <xref ref-type="bibr" rid="B19">Li et al., 2018</xref>), optical coherence tomography (OCT) (<xref ref-type="bibr" rid="B26">Schlegl et al., 2018</xref>), and scanning laser ophthalmoscopy (SLO) (<xref ref-type="bibr" rid="B21">Meyer et al., 2017</xref>; <xref ref-type="bibr" rid="B31">Tang et al., 2021</xref>), whose performance benchmarks often rival human experts in controlled settings. On the other hand, the advent of large language models (LLMs), particularly generative AI systems like ChatGPT, has rapidly expanded public access to AI technologies. Such models generate human-like responses from text prompts, offering applications ranging from facilitating physician-patient communication to synthesizing clinical data (<xref ref-type="bibr" rid="B6">Dave et al., 2023</xref>; <xref ref-type="bibr" rid="B33">Thirunavukarasu et al., 2023</xref>; <xref ref-type="bibr" rid="B36">Wu et al., 2023</xref>; <xref ref-type="bibr" rid="B10">Goh et al., 2024</xref>; <xref ref-type="bibr" rid="B32">Tangsrivimol et al., 2025</xref>; <xref ref-type="bibr" rid="B39">Yang X. et al., 2025</xref>). However, the diagnosis and management of many ophthalmic diseases require a complex integration of multimodal imaging interpretation and contextual clinical information (<xref ref-type="bibr" rid="B38">Yang et al., 2023</xref>; <xref ref-type="bibr" rid="B11">Gong et al., 2024</xref>). Current AI tools are predominantly designed for singular tasks and lack dynamic reasoning capabilities to emulate clinicians&#x2019; integrative decision-making processes (<xref ref-type="bibr" rid="B15">Homolak, 2023</xref>; <xref ref-type="bibr" rid="B20">Li et al., 2025</xref>; <xref ref-type="bibr" rid="B35">Wang et al., 2025</xref>). This critical gap limits their utility in real-world scenarios where diagnostic accuracy hinges on correlating heterogeneous data sources.</p>
<p>GPT-4o, OpenAI&#x2019;s multimodal large language model, demonstrates enhanced capability in processing hybrid inputs (text, imaging, and audio) through cross-modal alignment (<xref ref-type="bibr" rid="B27">Shea et al., 2023</xref>). While this capability offers distinct advantages for analyzing medical data, the model&#x2019;s multimodal architecture inadvertently amplifies hallucination risks, which may generate descriptions of pathological features absent from actual imaging findings (<xref ref-type="bibr" rid="B28">Shea and Ma, 2023</xref>; <xref ref-type="bibr" rid="B3">Chen D. et al., 2024</xref>; <xref ref-type="bibr" rid="B13">G&#xfc;nay et al., 2024</xref>; <xref ref-type="bibr" rid="B18">Li and Li, 2024</xref>). Furthermore, its black-box reasoning process fails to provide traceable diagnostic rationales anchored in medical literature, posing significant concerns for real-world clinical applications (<xref ref-type="bibr" rid="B16">Keles et al., 2025</xref>). Moreover, GPT-4o&#x2032;s reliance on general-domain training data limits its mastery of specialized ophthalmic knowledge, particularly rare disease patterns and region-specific diagnostic criteria (<xref ref-type="bibr" rid="B2">Cai et al., 2024</xref>). Although retrieval-augmented generation (RAG) enhances LLM responses by retrieving relevant information from external sources before generating answers, improving accuracy and reducing hallucinations (<xref ref-type="bibr" rid="B17">Lewis et al., 2020</xref>; <xref ref-type="bibr" rid="B23">Nguyen et al., 2025</xref>; <xref ref-type="bibr" rid="B29">Song et al., 2025</xref>), conventional RAG systems exhibit critical shortcomings in ophthalmology applications: they frequently retrieve contextually irrelevant guidelines due to inadequate understanding of imaging biomarkers while mechanically concatenating retrieved evidence without synthesizing pathophysiological logic, resulting in clinically incoherent recommendations (<xref ref-type="bibr" rid="B9">Gargari and Habibi, 2025</xref>; <xref ref-type="bibr" rid="B37">Yang R. et al., 2025</xref>). This dual challenge of multimodal hallucination control and context-aware knowledge integration necessitates an architectural paradigm that synergistically combines the perceptual strengths of multimodal LLMs with rigorous evidence-based reasoning. To bridge these gaps, a structured framework that seamlessly integrates multimodal image analysis, real-time knowledge retrieval, and clinical reasoning is urgently needed.</p>
<p>In January 2025, DeepSeek introduced DeepSeek-R1, an innovative open-source reasoning LLM rapidly gaining worldwide prominence (<xref ref-type="bibr" rid="B14">Guo et al., 2025</xref>). Differing from opaque models, DeepSeek-R1 enables transparent, hierarchical reasoning through probabilistic causal graphs, dynamically resolving conflicting clinical evidence to produce auditable diagnostic pathways (<xref ref-type="bibr" rid="B22">Mo&#xeb;ll et al., 2025</xref>; <xref ref-type="bibr" rid="B25">Sandmann et al., 2025</xref>). Additionally, its offline deployment capability allows healthcare institutions to locally operate and adapt the model without internet dependency, ensuring compliance with stringent data privacy regulations by eliminating sensitive data transmission, thereby fortifying security and confidentiality in clinical workflows (<xref ref-type="bibr" rid="B25">Sandmann et al., 2025</xref>).</p>
<p>Here, we proposed a structured reasoning agent (ReasonAgent) integrating three specialized modules: (1) a vision understanding module leveraging GPT-4o to analyze multimodal ophthalmic images and flag abnormalities; (2) a RAG module that retrieves diagnostic criteria from a curated knowledge base of ophthalmic guidelines based on patient history and exam findings; and (3) a diagnostic reasoning module (DeepSeek-R1) that synthesizes image interpretation, retrieved evidence, and clinical narratives to generate final diagnoses and treatment plans. To evaluate its clinical applicability, we compared the ReasonAgent&#x2019;s performance against standalone GPT-4o outputs and answers from three ophthalmology residents across 30 real-world cases. This study aim to investigate whether a structured ReasonAgent can surpass general-purpose LLMs in ophthalmic diagnosis and evaluate how AI-assisted decision-making compares to human resident physicians in complex real-world scenarios.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Study design and participants</title>
<p>This comparative, single-center, cross-sectional study adheres to the Strengthening the Reporting of Observational Studies in Epidemiology (STROBE) reporting guideline. A total of 30 deidentified ophthalmic cases (collected from January to March 2025) were included, with all protected health information rigorously encrypted. These cases were randomly selected from a database of Jinan University-affiliated Shenzhen Eye Hospital clinic, ensuring diversity in disease severity and presentation.</p>
</sec>
<sec id="s2-2">
<title>2.2 ReasonAgent implementation</title>
<p>We developed a hierarchical ReasonAgent (<xref ref-type="fig" rid="F1">Figure 1</xref>) through localized deployment of Dify. AI (Beijing, China, v1.0.0) workflow orchestration platform, where GPT-4o, RAG architecture, and DeepSeek-R1 were programmatically chained as core processing nodes. We configured all system components through dedicated API interfaces, establishing automated data pipelines between modules. The agent was programmed to emulate clinical ophthalmologists&#x2019; diagnostic workflow through the following technical implementation.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Flowchart of the Reasoning Agent Design and the Evaluation of Different Methods&#x2019; Responses in Clinical Ophthalmology Scenarios. Ophthalmic imaging (e.g., OCT, B scan, SLO, FFA) and clinical history serve as input sources. The Vision Understanding Module (GPT-4o) analyzes ophthalmic images for abnormalities and descriptions. The Evidence Retrieval Module (RAG) extracts diagnostic knowledge from guidelines based on clinical history and ocular examination. These outputs, combined with clinical history text, are input into the Diagnostic Reasoning Module (DeepSeek-R1) within the reasoning agent for diagnostic analysis and treatment planning. Comparison groups included standalone GPT-4o and three residents. Responses were evaluated using Likert scales by 7 attending physicians.</p>
</caption>
<graphic xlink:href="fcell-13-1642539-g001.tif">
<alt-text content-type="machine-generated">Flowchart depicting a system of ophthalmic diagnosis. Inputs are Ophthalmic Imaging and Clinical History leading to ReasonAgent, consisting of Vision Understanding Module (GPT-4o) and Evidence Retrieval Module (RAG), which connects to Diagnostic Reasoning Module (DeepSeek-R1). Outputs involve Ophthalmology Residents reviewing, followed by Rating Questionnaires, evaluation by 7 attendings, and Data Analysis with a Cumulative Link Mixed Model.</alt-text>
</graphic>
</fig>
<sec id="s2-2-1">
<title>2.2.1 Vision understanding module</title>
<p>GPT-4o (OpenAI, USA; version 2024&#x2013;11&#x2013;20, temperature &#x3d; 0.7) was adopted as a visual analysis module. This module received multimodal ophthalmic imaging inputs (e.g., OCT, B-scan, SLO, fluorescein fundus angiography/indocyanine green angiography (FFA/ICGA)) with the prompt provided in <xref ref-type="sec" rid="s11">Supplementary Appendix 1</xref>.</p>
</sec>
<sec id="s2-2-2">
<title>2.2.2 Evidence retrieval module</title>
<p>We employed a RAG architecture BGE-M3 (<xref ref-type="bibr" rid="B4">Chen J. et al., 2024</xref>) embeddings (designed by BAAI, China; provided by SiliconFlow, China) for multilingual knowledge retrieval. The knowledge base integrated two principal corpora: 1) Kanski&#x2019;s Clinical Ophthalmology (ninth Edition) as foundational textbook knowledge, and 2) annually updated clinical guidelines (January 2024 to February 2025) from the American Academy of Ophthalmology (AAO) and the Chinese Medical Association (CMA). A unified prompting framework was adopted across retrieval components, synchronizing with the DeepSeek-R1 model through the shared instruction.</p>
</sec>
<sec id="s2-2-3">
<title>2.2.3 Diagnostic reasoning module</title>
<p>We applied DeepSeek-R1 (DeepSeek, China; 671B version, temperature &#x3d; 0.6) for comprehensive reasoning analysis. The model received formatted inputs: [Imaging Analysis] &#x2b; [Retrieved Evidence] &#x2b; [Clinical History], generating a reasoning process, preliminary diagnosis, and treatment plans with explicit citations, with the detailed prompt provided in <xref ref-type="sec" rid="s11">Supplementary Appendix 1</xref>.</p>
</sec>
</sec>
<sec id="s2-3">
<title>2.3 Case selection</title>
<p>To evaluate the models&#x2019; performance versus clinicians across diseases of varying complexity, we curated 30 clinical cases spanning corneal diseases, cataracts, glaucoma, and fundus disorders. The cohort included 27 common ophthalmic conditions (e.g., age-related cataracts, retinal detachment) and 3 rare diseases (Coats disease, malignant glaucoma, Vogt-Koyanagi-Harada syndrome).</p>
</sec>
<sec id="s2-4">
<title>2.4 Comparison and scoring criteria</title>
<p>To establish comparative benchmarks, clinical cases in Chinese with associated imaging data were independently analyzed using three methods: 1) the output of ReasonAgent pipeline, 2) GPT-4o analysis with explicit instructions shown in <xref ref-type="sec" rid="s11">Supplementary Appendix 1</xref>, 3) three residents producing comprehensive diagnoses and prioritized treatment plans.</p>
<p>The diagnoses and treatment plans generated by ReasonAgent, GPT4o, and residents were anonymized and randomly presented to the panel of 7 senior attending physicians for evaluation using a 5-point Likert scale:<list list-type="simple">
<list-item>
<p>1: Unacceptably poor or containing critical errors</p>
</list-item>
<list-item>
<p>2: Poor accuracy with potentially harmful errors/omissions</p>
</list-item>
<list-item>
<p>3: Neutral (moderate quality with ambiguous/minor issues)</p>
</list-item>
<list-item>
<p>4: Good quality with non-critical errors/omissions</p>
</list-item>
<list-item>
<p>5: Excellent quality with no errors/omissions</p>
</list-item>
</list>
</p>
</sec>
<sec id="s2-5">
<title>2.5 Statistical analysis</title>
<p>Descriptive statistics were reported as means, standard deviations (SD), along with medians and interquartile ranges (IQR) of the Likert scores. A Cumulative Link Mixed Model (CLMM) fitted with the Laplace approximation was implemented to evaluate decision-making performance differences between methods (ReasonAgent, GPT-4o, and residents). Two separate model analyses were conducted for diagnoses and treatment plans, with each model preserving identical random effects structures. Fixed effects were modeled to assess the accuracy of diagnoses and treatment plans, while random effects accounted for variability across individual cases and between different raters. Main analytical indices included estimated marginal means (EMMs) with 95% confidence interval (CI), and interpretation of intraclass correlation coefficients (ICCs) derived from variance components of the logistic distribution to quantify proportional variance contributions of case-level and rater-level heterogeneity, with lower ICC values indicating higher consistency. Post hoc pairwise comparisons with Tukey adjustment for multiple testing were conducted to identify specific group differences. For subgroup comparisons between common and rare diseases, Kruskal&#x2013;Wallis tests were performed to detect overall differences in scores across groups within each disease category. Dunn&#x2019;s <italic>post hoc</italic> tests with Bonferroni adjustment were applied for pairwise group comparisons. Additionally, low-score proportions (scores&#x2264;2) were analyzed as a secondary metric to evaluate performance across methods. The level of significance was set at <italic>p</italic> &#x3c; 0.05. All analyses were conducted in R (version 4.4.3) with the ordinal package (<italic>clmm</italic> function) for model fitting, <italic>lme4</italic> for mixed-effects infrastructure, <italic>emmeans</italic> for marginal mean estimation and comparisons, and <italic>FSA</italic> for non-parametric subgroup analysis.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Fixed effects analysis of reasoning performance across methods</title>
<p>The CLMM analysis revealed no statistically significant differences in diagnostic reasoning performance between the ReasonAgent (median &#x3d; 4, IQR &#x3d; 3&#x2013;5), GPT-4o (median &#x3d; 4, IQR &#x3d; 3&#x2013;4), and residents (median &#x3d; 4, IQR &#x3d; 3&#x2013;4, <xref ref-type="fig" rid="F2">Figure 2A</xref>). Fixed effects comparisons showed non-significant deviations for GPT-4o (<italic>&#x3b2;</italic> &#x3d; 0.04, 95% CI: &#x2212;0.31 to 0.40, <italic>p</italic> &#x3d; 0.81) and physicians (<italic>&#x3b2;</italic> &#x3d; &#x2212;0.07, 95% CI: &#x2212;0.36 to 0.22, <italic>p</italic> &#x3d; 0.65, <xref ref-type="table" rid="T1">Table 1</xref>) relative to the ReasonAgent. Post hoc pairwise contrasts revealed consistently non-significant differences across all groups (<xref ref-type="table" rid="T1">Table 1</xref>). These findings demonstrate concordant diagnostic reasoning across methods, with algorithmic approaches (ReasonAgent and GPT-4o) achieving performance comparable to human physicians.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Distribution of Likert Scores for Different Methods in Diagnostic Tasks and Treatment Planning Tasks. <bold>(A)</bold> Violin plot of Likert scores for diagnostic tasks; <bold>(B)</bold> Violin plot of Likert scores for treatment planning tasks. Embedded boxplots illustrate the interquartile range (25th to 75th percentile), the median (black horizontal line), and the whiskers represent the range of scores excluding outliers. Statistical analysis revealed no significant differences in diagnostic task scores between ReasonAgent, GPT-4o, and residents. In contrast, treatment planning tasks showed significantly higher scores for ReasonAgent than GPT-4o and residents. &#x2a;<italic>p</italic> &#x3c; 0.05, &#x2a;&#x2a;<italic>p</italic> &#x3c; 0.01, &#x2a;&#x2a;&#x2a;<italic>p</italic> &#x3c; 0.001.</p>
</caption>
<graphic xlink:href="fcell-13-1642539-g002.tif">
<alt-text content-type="machine-generated">Violin plots with box plots inside show Likert scores for diagnostic and treatment planning tasks across three groups: ReasonAgent, GPT, and Residents. Panel A shows diagnostic tasks, and Panel B shows treatment planning tasks. Higher scores are shown for Residents across both tasks, with significant differences indicated by asterisks in Panel B.</alt-text>
</graphic>
</fig>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Fixed effects analysis and post-hoc pairwise comparisons of different methods.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Group</th>
<th align="center">EMMs</th>
<th align="center">OR<break/>(95% CI)</th>
<th align="center">
<italic>z</italic>-value</th>
<th align="center">
<italic>p</italic>-value<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</th>
<th align="center">Adjusted <italic>p</italic>-value<xref ref-type="table-fn" rid="Tfn2">
<sup>b</sup>
</xref>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td colspan="6" align="left">Diagnostic task</td>
</tr>
<tr>
<td align="left">ReasonAgent (reference)</td>
<td align="center">0</td>
<td align="center">1.00</td>
<td align="center">&#x2014;</td>
<td align="left">&#x2014;</td>
<td align="left">vs. GPT-4o (0.97)</td>
</tr>
<tr>
<td align="left">GPT-4o</td>
<td align="center">0.04</td>
<td align="center">1.05<break/>(0.73&#x2013;1.49)</td>
<td align="center">0.24</td>
<td align="left">0.81</td>
<td align="left">vs. Resident (0.73)</td>
</tr>
<tr>
<td align="left">Resident</td>
<td align="center">&#x2212;0.07</td>
<td align="center">0.94<break/>(0.70&#x2013;1.25)</td>
<td align="center">&#x2212;0.45</td>
<td align="left">0.65</td>
<td align="left">vs. ReasonAgent (0.90)</td>
</tr>
<tr>
<td colspan="6" align="left">Treatment planning task</td>
</tr>
<tr>
<td align="left">ReasonAgent (reference)</td>
<td align="center">0</td>
<td align="center">1.00</td>
<td align="center">&#x2014;</td>
<td align="left">&#x2014;</td>
<td align="left">vs. GPT-4o (0.03)&#x2a;</td>
</tr>
<tr>
<td align="left">GPT-4o</td>
<td align="center">0.05</td>
<td align="center">0.62<break/>(0.42&#x2013;0.89)</td>
<td align="center">&#x2212;2.54</td>
<td align="left">0.011&#x2a;</td>
<td align="left">vs. Resident (&#x3c;0.001)&#x2a;&#x2a;&#x2a;</td>
</tr>
<tr>
<td align="left">Resident</td>
<td align="center">&#x2212;1.71</td>
<td align="center">0.94<break/>(0.13&#x2013;0.25)</td>
<td align="center">&#x2212;10.69</td>
<td align="left">&#x3c;0.001&#x2a;&#x2a;&#x2a;</td>
<td align="left">vs. ReasonAgent (&#x3c;0.001)&#x2a;&#x2a;&#x2a;</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>EMMs: Estimated Marginal Means, OR: odds ratio, CI: confidence interval.</p>
</fn>
<fn id="Tfn1">
<label>
<sup>a</sup>
</label>
<p>Cumulative Link Mixed Model (CLMM) fitted with the Laplace approximation.</p>
</fn>
<fn id="Tfn2">
<label>
<sup>b</sup>
</label>
<p>Post-hoc pairwise comparisons with Tukey adjustment.</p>
</fn>
<fn>
<p>&#x2a;<italic>p</italic> &#x3c; 0.05, &#x2a;&#x2a;<italic>p</italic> &#x3c; 0.01, &#x2a;&#x2a;&#x2a;<italic>p</italic> &#x3c; 0.001.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>In terms of the treatment planning task, the statistical analysis identified significant between-group differences across methods. Fixed effects analysis revealed that ReasonAgent (median &#x3d; 4, IQR &#x3d; 4&#x2013;5) significantly outperformed both GPT-4o (median &#x3d; 4, IQR &#x3d; 3-5, <italic>&#x3b2;</italic> &#x3d; 0.49, 95% CI: &#x2212;0.86 to &#x2212;0.11, <italic>p</italic> &#x3d; 0.01) and residents (median &#x3d; 3, IQR &#x3d; 2-4, <italic>&#x3b2;</italic> &#x3d; 1.71, 95% CI: &#x2212;2.03 to &#x2212;1.40, <italic>p</italic> &#x3c; 0.001, <xref ref-type="table" rid="T1">Table 1</xref>; <xref ref-type="fig" rid="F2">Figure 2B</xref>). Post-hoc comparisons with Tukey adjustment demonstrated significantly superior performance of Reasoning Agent over both GPT-4o (<italic>&#x3b2;</italic> &#x3d; 0.486, <italic>p</italic> &#x3d; 0.030) and residents (<italic>&#x3b2;</italic> &#x3d; 1.714, <italic>p</italic> &#x3c; 0.001, <xref ref-type="table" rid="T1">Table 1</xref>). GPT-4o also demonstrated significant advantages over human physicians (<italic>&#x3b2;</italic> &#x3d; 1.228, <italic>p</italic> &#x3c; 0.001). These results indicate that algorithmic approaches (Reasoning Agent and GPT-4o) consistently outperformed human physicians in treatment plan formulation, with ReasonAgent achieving particularly enhanced efficacy relative to GPT-4o.</p>
</sec>
<sec id="s3-2">
<title>3.2 Rater-level variability analysis</title>
<p>Low variability in rating stringency was observed across raters, with the random intercept variance between raters in the diagnostic reasoning task estimated at &#x3c3;<sup>2</sup> &#x3d; 0.40 (SD &#x3d; 0.63, <xref ref-type="table" rid="T2">Table 2</xref>). This between-rater heterogeneity corresponded to an ICC of 0.08 (95% CI: 0&#x2013;0.16), indicating that approximately 8% of the total variance originated from systematic differences in scoring severity across raters. This indicates that low inter-rater disagreement in Likert scores for diagnostic tasks, reflecting relatively consistent clinical expertise or interpretive standards among raters.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Random effects analysis in diagnostic and treatment planning tasks.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Group</th>
<th align="left">Variance</th>
<th align="left">SD</th>
<th align="left">Groups</th>
<th align="left">ICC (95% CI)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td colspan="5" align="left">Diagnostic task</td>
</tr>
<tr>
<td align="left">Case</td>
<td align="left">1.45</td>
<td align="left">1.20</td>
<td align="left">30</td>
<td align="left">0.28 (0.18&#x2013;0.39)</td>
</tr>
<tr>
<td align="left">Rater</td>
<td align="left">0.40</td>
<td align="left">0.63</td>
<td align="left">7</td>
<td align="left">0.08 (0&#x2013;0.16)</td>
</tr>
<tr>
<td colspan="5" align="left">Treatment planning task</td>
</tr>
<tr>
<td align="left">Case</td>
<td align="left">1.01</td>
<td align="left">1.01</td>
<td align="left">30</td>
<td align="left">0.21 (0.15&#x2013;0.31)</td>
</tr>
<tr>
<td align="left">Rater</td>
<td align="left">0.47</td>
<td align="left">0.68</td>
<td align="left">7</td>
<td align="left">0.10 (0.04&#x2013;0.18)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>SD, standard deviation, ICC, intraclass correlation coefficient.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>While in the treatment planning task, the variability of rating is slightly increased compared to the diagnostic task (&#x3c3;<sup>2</sup> &#x3d; 0.47, SD &#x3d; 0.68, <xref ref-type="table" rid="T2">Table 2</xref>), and the corresponding ICC is 0.10 (95% CI: 0.04&#x2013;0.18), suggesting greater inconsistency in treatment planning assessment ratings.</p>
</sec>
<sec id="s3-3">
<title>3.3 Case-level variability analysis</title>
<p>In the diagnostic reasoning task, considerable case-level heterogeneity was observed, with case-level random intercepts accounting for significant variance in diagnostic performance ratings (&#x3c3;<sup>2</sup> &#x3d; 1.44, SD &#x3d; 1.20, <xref ref-type="table" rid="T2">Table 2</xref>). The ICC confirmed that 28% (ICC &#x3d; 0.28, 95% CI: 0.18&#x2013;0.39)of total variance stemmed from systematic differences between clinical cases. This indicates that clinical characteristics or case complexity exerted a notable influence on diagnostic assessments. For the treatment planning task, while case-level variability remained significant (&#x3c3;<sup>2</sup> &#x3d; 1.01, SD &#x3d; 1.01, <xref ref-type="table" rid="T2">Table 2</xref>), its absolute contribution decreased (ICC &#x3d; 0.21,95% CI: 0.15&#x2013;0.31), with the relative contribution to total variance components also reducing to 68.5% compared to diagnostic tasks.</p>
<p>Pronounced performance variations existed between rare and common cases across methods. While all three approaches achieved comparable ratings for common cases (ReasonAgent: median &#x3d; 4, IQR &#x3d; 3-5, GPT-4o: median &#x3d; 4, IQR &#x3d; 3-5, residents: median &#x3d; 4, IQR &#x3d; 3&#x2013;4; all <italic>p</italic> &#x3e; 0.05, <xref ref-type="table" rid="T3">Table 3</xref>), rare cases revealed substantial divergence in diagnostic performance. GPT-4o demonstrated significantly lower performance (median &#x3d; 1, IQR &#x3d; 1&#x2013;2) compared to ReasonAgent (median &#x3d; 3, IQR &#x3d; 1-4, <italic>p</italic> &#x3d; 0.02) and residents (median &#x3d; 3, IQR &#x3d; 1-4, <italic>p</italic> &#x3d; 0.03, <xref ref-type="table" rid="T3">Table 3</xref>). This substantial gap persisted despite limited rare-case samples (n &#x3d; 3), reflecting methodological vulnerabilities. In addition, GPT-4o yielded low scores (&#x2264;2) in 90.48% of rare-case diagnostic tasks, a proportion substantially surpassing the rates recorded for ReasonAgent (38.10%) and physicians (49.21%), suggesting greater vulnerability to the diagnostic complexity of rare ophthalmic conditions.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Diagnostic and treatment planning performance by method and case rarity.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Group</th>
<th align="center">Method</th>
<th align="center">Mean score<break/>&#xb1;SD</th>
<th align="center">Median score (IQR)</th>
<th align="center">% low scores (&#x2264;2)</th>
<th align="left">Pairwise<break/>Comparisons (<italic>p</italic>-values<xref ref-type="table-fn" rid="Tfn3">
<sup>a</sup>
</xref>)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td colspan="6" align="left">Diagnostic task</td>
</tr>
<tr>
<td rowspan="3" align="left">Common diseases</td>
<td align="left">ReasonAgent</td>
<td align="center">3.59 &#xb1; 1.27</td>
<td align="center">4 (3&#x2013;5)</td>
<td align="left">39/189 (15.34%)</td>
<td align="left">vs. GPT-4o (&#x3e;0.99)</td>
</tr>
<tr>
<td align="left">GPT-4o</td>
<td align="center">3.68 &#xb1; 1.18</td>
<td align="center">4 (3&#x2013;5)</td>
<td align="left">32/189 (16.93%)</td>
<td align="left">vs. Resident (0.47)</td>
</tr>
<tr>
<td align="left">Resident</td>
<td align="center">3.57 &#xb1; 1.16</td>
<td align="center">4 (3&#x2013;4)</td>
<td align="left">100/567 (17.64%)</td>
<td align="left">vs. ReasonAgent (&#x3e;0.99)</td>
</tr>
<tr>
<td rowspan="3" align="left">Rare diseases</td>
<td align="left">ReasonAgent</td>
<td align="center">2.91 &#xb1; 1.58</td>
<td align="center">3 (1&#x2013;4)</td>
<td align="left">8/21 (38.10%)</td>
<td align="left">vs. GPT-4o (0.02)&#x2a;</td>
</tr>
<tr>
<td align="left">GPT-4o</td>
<td align="center">1.67 &#xb1; 0.91</td>
<td align="center">1 (1&#x2013;2)</td>
<td align="left">19/21 (90.48%)</td>
<td align="left">vs. Resident (0.03)&#x2a;</td>
</tr>
<tr>
<td align="left">Resident</td>
<td align="center">2.60 &#xb1; 1.40</td>
<td align="center">3 (1&#x2013;4)</td>
<td align="left">31/63 (49.21%)</td>
<td align="left">vs. ReasonAgent (&#x3e;0.99)</td>
</tr>
<tr>
<td colspan="6" align="left">Treatment planning task</td>
</tr>
<tr>
<td rowspan="3" align="left">Common diseases</td>
<td align="left">ReasonAgent</td>
<td align="center">4.12 &#xb1; 1.10</td>
<td align="center">4 (4&#x2013;5)</td>
<td align="left">21/189 (11.11%)</td>
<td align="left">vs. GPT-4o (0.67)</td>
</tr>
<tr>
<td align="left">GPT-4o</td>
<td align="center">3.96 &#xb1; 1.23</td>
<td align="center">4 (3&#x2013;5)</td>
<td align="left">29/189 (15.34%)</td>
<td align="left">vs. Resident (&#x3c;0.001)&#x2a;&#x2a;&#x2a;</td>
</tr>
<tr>
<td align="left">Resident</td>
<td align="center">3.27 &#xb1; 1.11</td>
<td align="center">3 (3&#x2013;4)</td>
<td align="left">139/567 (24.51%)</td>
<td align="left">vs. ReasonAgent (&#x3c;0.001)&#x2a;&#x2a;&#x2a;</td>
</tr>
<tr>
<td rowspan="3" align="left">Rare diseases</td>
<td align="left">ReasonAgent</td>
<td align="center">3.10 &#xb1; 1.30</td>
<td align="center">4 (2&#x2013;4)</td>
<td align="left">8/21 (38.10%)</td>
<td align="left">vs. GPT-4o (&#x3c;0.001)&#x2a;&#x2a;&#x2a;</td>
</tr>
<tr>
<td align="left">GPT-4o</td>
<td align="center">1.67 &#xb1; 1.11</td>
<td align="center">1 (1&#x2013;2)</td>
<td align="left">17/21 (80.95%)</td>
<td align="left">vs. Resident (0.14)</td>
</tr>
<tr>
<td align="left">Resident</td>
<td align="center">2.25 &#xb1; 1.23</td>
<td align="center">2 (1&#x2013;3)</td>
<td align="left">39/63 (61.90%)</td>
<td align="left">vs. ReasonAgent (0.04)&#x2a;</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>SD, standard deviation, IQR, interquartile range.</p>
</fn>
<fn id="Tfn3">
<label>
<sup>a</sup>
</label>
<p>Dunn&#x2019;s post hoc tests with Bonferroni adjustment.</p>
</fn>
<fn>
<p>&#x2a;<italic>p</italic> &#x3c; 0.05, &#x2a;&#x2a;<italic>p</italic> &#x3c; 0.01, &#x2a;&#x2a;&#x2a;<italic>p</italic> &#x3c; 0.001.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>In the treatment planning task, we also observed distinct performance variations among the three methods in both common and rare diseases. For common cases, the algorithmic methods (ReasonAgent (median &#x3d; 4, IQR &#x3d; 4-5, <italic>p</italic> &#x3c; 0.001) and GPT4o (median &#x3d; 4, IQR &#x3d; 3-5, <italic>p</italic> &#x3c; 0.001)) demonstrated statistically significant superior scores compared to residents (median &#x3d; 3, IQR &#x3d; 3&#x2013;4, <xref ref-type="table" rid="T3">Table 3</xref>), while showing no significant performance difference between the two algorithms methods (<italic>p</italic> &#x3d; 0.67). In rare disease scenarios, ReasonAgent (median &#x3d; 4, IQR &#x3d; 2&#x2013;4) significantly outperformed both GPT4o (median &#x3d; 1, IQR &#x3d; 1-2, <italic>p</italic> &#x3c; 0.001, <xref ref-type="table" rid="T3">Table 3</xref>) and residents (median &#x3d; 2, IQR &#x3d; 1-3, <italic>p</italic> &#x3d; 0.04), with GPT4o showing no advantage over human physicians (<italic>p</italic> &#x3d; 0.14). Additionally, statistical analysis of low-performance probabilities (scores &#x2264;2) corroborated these trends: For common disease treatment planning, both GPT4o (15.34%) and ReasonAgent (11.11%) exhibited significantly lower rates of low scores compared to human physicians (24.51%). And for rare diseases, ReasonAgent (38.10%) maintained a substantially lower low-score rate than both residents (61.90%) and GPT4o (80.95%), demonstrating its dual advantage in treatment planning for both common and rare clinical conditions.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>With the rapid development of large language models, artificial intelligence has demonstrated tremendous potential for application in the medical field. However, the diagnosis and treatment decision-making of ophthalmic diseases possess unique complexity: it not only relies on the meticulous interpretation of multimodal images but also requires the integration of heterogeneous data, including the current medical history and systemic comorbidities, to formulate personalized treatment plans. This study constructs an ophthalmic reasoning agent by integrating the modules of visual understanding, knowledge retrieval, and causal reasoning, and evaluates its performance in real ophthalmic cases.</p>
<p>Using 30 ophthalmic cases, we conducted a comprehensive performance evaluation of diagnostic reasoning and treatment planning across three methods: ReasonAgent, GPT-4o, and human physicians. The results demonstrated that in diagnostic tasks, the algorithmic methods (Reasoning Agent and GPT-4o) achieved performance comparable to ophthalmology residents. In treatment planning tasks, both algorithmic approaches significantly outperformed human doctors, with Reasoning Agent showing a notably superior performance compared to GPT-4o alone. One possible explanation is that diagnostic classification critically relies on quantifiable biomarkers, such as the foveal thickness in OCT. Both AI systems and human physicians can achieve this through pattern recognition. However, in treatment planning, the Reasoning Agent mitigates inexperience-driven cognitive biases among ophthalmology residents through evidence integration via RAG. Additionally, it can correlate influential features with the latest guidelines, thus circumventing the generalization errors of GPT-4o. This indicates that through the structured reasoning involving multimodal data integration and evidence anchoring, AI has the potential to transcend the limitations of a single model. Specifically, the visual module (GPT-4o) of the Reasoning Agent accurately captures imaging abnormalities, the RAG module retrieves the latest guidelines in real time, and the DeepSeek-R1 reasoning module strings together clinical information based on causal logic to form a traceable decision-making path. For instance, in the case of Coats&#x2019; disease, GPT-4o misdiagnosed it as retinoblastoma, and some residents confused it with persistent hyperplastic primary vitreous (PHPV). However, Reasoning Agent accurately identified the typical vascular abnormalities through the cross-validation of imaging features and guideline criteria, and through the clinical characteristics of the medical history, it gave the reasoning process of the diagnosis and avoided serious misdiagnosis. In a case of diabetic retinopathy, GPT-4o had a conflict in the identification of OCTA and B-scan images (OCTA detected tractional retinal detachment, while B-scan indicated &#x201c;no characteristic strong echo signals of retinal detachment, suggesting that the retina is attached&#x201d;). However, the reasoning module (DeepseekR1) resolved this conflict, presented a detailed thought process, and obtained the correct diagnosis and treatment plan. On the other hand, DeepSeek-R1 demonstrates superior capabilities in processing clinical documentation in Chinese. Since this study is based on medical records in Chinese, its language architecture can more accurately capture the key clinical features in the Chinese context and reduce the ambiguity caused by direct Chinese-English translation in term mapping. These findings highlight the necessity of domain-specific architectures to constrain general model limitations while augmenting human expertise. These results are consistent with the theory proposed by Bommasani et al. that &#x201c;clinical AI needs to integrate perception and reasoning&#x201d;, suggesting that future system designs should give priority to cross-modal alignment and evidence-based constraint mechanisms (<xref ref-type="bibr" rid="B1">Bommasani et al., 2021</xref>).</p>
<p>The observed case-level heterogeneity (intraclass correlation coefficient for diagnosis, ICC &#x3d; 0.28, P &#x3c; 0.01) may reflect limitations of general large language models in processing multimodal inputs with varying pathological complexities. While GPT-4o&#x2032;s unstructured reasoning suffices for common conditions with typical patterns, its failure in rare cases (90.48% low diagnostic scores) suggests insufficient domain-specific knowledge acquisition and overreliance on probabilistic associations rather than pathophysiological logic. ReasonAgent&#x2019;s superior performance in rare disease diagnosis likely stems from its hybrid architecture: a vision-language model enables lesion localization, while RAG constrains reasoning to evidence-based diagnostic pathways, compensating for the scarcity of low-prevalence disease patterns in general training data. In treatment planning, the ReasonAgent&#x2019;s advantage over both GPT-4o and resident physicians indicates that structured knowledge retrieval mitigates knowledge gaps or clinical experience deficits in junior doctors. GPT-4o&#x2032;s poorer therapeutic performance compared to diagnosis aligns with its lack of hierarchical treatment action structures&#x2014;a critical gap addressed by RAG, which prioritizes guideline-recommended interventions. These findings highlight the utility of hybrid AI systems integrating deep visual understanding with evidence-based reasoning in overcoming the limitations of general-purpose models in complex medical scenarios.</p>
<p>ReasonAgent&#x2019;s hierarchical architecture distinguishes it from a single black-box large model like GPT-4o. By modularizing clinical reasoning into discrete stages&#x2014;imaging analysis, knowledge retrieval, biometric validation, and evidence-based conclusion generation&#x2014;it is analogous to the systematic logic of human clinicians while preserving traceable reasoning paths. Throughout this process, a reasoning path visualization module preserves the full decision-making trajectory, enabling clinicians to retrace critical nodes such as biological parameter calculations and literature evidence citations. This feature distinctly differentiates ReasonAgent from general models, whose untraceable probabilistic output paradigms prevent the attribution of diagnostic or therapeutic suggestions to specific evidential sources. Therefore, in clinical applications, the Reasoning Agent has two major application potentials. Firstly, as a decision-support tool for junior doctors and primary care doctors, its traceable reasoning process facilitates rapid and accurate identification of evidence-based rationales for therapeutic plans, addressing knowledge gaps in less experienced clinicians. Secondly, as a quality control tool, it can serve as an auxiliary reference for junior doctors in medical record writing, reducing the risks of missed or misdiagnosis. Its evidence-traceable reasoning framework can also act as an assistant in the medical record systems to assist in systematic validation of medical records.</p>
<p>In this study, several critical observations concerning large language models (LLMs) merit attention. Firstly, RAG cannot fully resolve hallucinations (<xref ref-type="bibr" rid="B9">Gargari and Habibi, 2025</xref>; <xref ref-type="bibr" rid="B37">Yang R. et al., 2025</xref>), such as mis-defining normative biometric thresholds (e.g., foveal thickness ranges) and conflating diagnostic criteria (e.g., high myopia). These errors emphasize the need for real-time biometric verification to counter parameter fabrication tendencies. Secondly, the GPT-4o demonstrated elevated misjudgment rates for highly specialized imaging data, such as anterior segment photography or dynamically interpreted datasets like B-scan, errors included misidentifying corneal reflection points as corneal leukomas and failing to determine posterior movement positivity in B-scans. Such misinterpretations propagate downstream analytical errors in Deepseek-R1 (e.g., the epiretinal membrane case unrecognized by GPT-4o in this study). These observations accentuate the imperative for the development of ophthalmology-specific visual processing modules, which transcend the limitations of generic image recognition algorithms. Futhermore, deploying AI-driven reasoning agents in clinical settings requires careful attention to data privacy, infrastructure, human oversight, and ethics. The proposed method is built upon closed-source services for preliminary clinical validation. To address patient privacy concerns in clinical practice, these components can be replaced with other locally deployed open-source models. However, local deployment incurs substantially higher costs and operational overhead. For example, deploying a large language model of DeepSeek R1&#x2019;s scale (671 billion parameters) poses significant challenges for hospital infrastructure stability and maintenance. Besides, although our method has demonstrated diagnostic capabilities on par with those of junior human clinicians and even superior performance in formulating treatment plans, it is intended solely as a decision-support tool and cannot replace human clinicians. In certain scenarios, the performance may also reflect biases originating from the training data and the models themselves. Beyond model-specific limitations, the current study still has several methodological limitations. The relatively modest sample size of 30 cases, with rare diseases constituting only 10% (n &#x3d; 3), may compromise the statistical power required for comprehensive subgroup analyses, and the inter-case heterogeneity inherent in small cohorts may obscure statistically significant differences in diagnostic performance across methods. Future investigations should endeavor to expand the sample cohort to robustly validate the stability of ReasonAgent. Additionally, the small sample of three resident evaluators may introduce observer bias; future studies should include senior ophthalmologists to enhance validation rigor.</p>
<p>This study developed an ophthalmic ReasonAgent integrating visual understanding (GPT-4o), evidence retrieval (RAG), and diagnostic reasoning (DeepSeek-R1) modules to enable interpretable decision-making in multimodal clinical scenarios. Testing on 30 real-world cases demonstrated that the ReasonAgent exhibited diagnostic accuracy comparable to that of resident ophthalmologists, while significantly outperforming both human physicians and the general-purpose large language model GPT-4o in treatment planning. Its core advantage lies in a hierarchical reasoning mechanism: dynamic knowledge retrieval is triggered by imaging feature analysis, combined with causal logic to generate traceable diagnostic and therapeutic decision trees. This approach mitigates GPT-4o&#x2032;s cross-modal misalignment and reduces empirical biases inherent to resident physicians in complex cases. The ReasonAgent addresses the cross-modal alignment limitations of general LLMs and offers evidence-based reasoning outcomes, establishing a novel framework for the application of medical artificial intelligence in real-world clinical practice.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s11">Supplementary Material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>YZ: Conceptualization, Data curation, Formal Analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review and editing. DF: Validation, Investigation, Writing &#x2013; review and editing, Conceptualization, Supervision, Writing &#x2013; original draft, Formal Analysis, Methodology, Software, Project administration, Visualization, Data curation. PL: Investigation, Visualization, Validation, Writing &#x2013; review and editing, Data curation, Methodology, Supervision, Formal Analysis, Conceptualization, Writing &#x2013; original draft. BB: Data curation, Investigation, Validation, Writing &#x2013; review and editing. XH: Validation, Data curation, Writing &#x2013; review and editing, Investigation. LF: Writing &#x2013; review and editing, Investigation, Data curation, Validation. WL: Data curation, Investigation, Validation, Writing &#x2013; review and editing. SZ: Conceptualization, Resources, Funding acquisition, Formal Analysis, Supervision, Writing &#x2013; review and editing, Project administration, Methodology.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported by grants from the Guangdong Basic and Applied Basic Research Foundation (2022A1515111155) and the Shenzhen Science and Technology Program (KCXFZ20211020163813019). The funding sources had no role in the design and conduct of the study; collection, management, analysis, and interpretation of the data; preparation, review, or approval of the manuscript; and decision to submit the manuscript for publication.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fcell.2025.1642539/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fcell.2025.1642539/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bommasani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hudson</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Adeli</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Altman</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Arora</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Arx</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>On the opportunities and risks of foundation models</article-title>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Assessing the accuracy and clinical utility of GPT-4O in abnormal blood cell morphology recognition</article-title>. <source>Digit. Health</source> <volume>10</volume>, <fpage>20552076241298503</fpage>. <pub-id pub-id-type="doi">10.1177/20552076241298503</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Jomy</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Croke</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2024a</year>). <article-title>Performance of multimodal artificial intelligence chatbots evaluated on clinical oncology cases</article-title>. <source>JAMA Netw. Open</source> <volume>7</volume> (<issue>10</issue>), <fpage>e2437711</fpage>. <pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.37711</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lian</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z. J. a.p.a.</given-names>
</name>
</person-group> (<year>2024b</year>). <article-title>Bge m3-embedding: multi-Lingual, multi-functionality, multi-granularity text embeddings through self-knowledge distillation</article-title>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Choi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Oh</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Yoo</surname>
<given-names>S. Y.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>D. J.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Evaluation of the quality and quantity of artificial intelligence-generated responses about anesthesia and surgery: using ChatGPT 3.5 and 4.0</article-title>. <source>Front. Med.</source> <volume>11</volume>, <fpage>1400153</fpage>. <pub-id pub-id-type="doi">10.3389/fmed.2024.1400153</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dave</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Athaluri</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>ChatGPT in medicine: an overview of its applications, advantages, limitations, future prospects, and ethical considerations</article-title>. <source>Front. Artif. Intell.</source> <volume>6</volume>, <fpage>1169595</fpage>. <pub-id pub-id-type="doi">10.3389/frai.2023.1169595</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Delsoz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Raja</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Madadi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Wirostko</surname>
<given-names>B. M.</given-names>
</name>
<name>
<surname>Kahook</surname>
<given-names>M. Y.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>The use of ChatGPT to assist in diagnosing glaucoma based on clinical case reports</article-title>. <source>Ophthalmol. Ther.</source> <volume>12</volume> (<issue>6</issue>), <fpage>3121</fpage>&#x2013;<lpage>3132</lpage>. <pub-id pub-id-type="doi">10.1007/s40123-023-00805-x</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Assessment of synthetic post-therapeutic OCT images using the generative adversarial network in patients with macular edema secondary to retinal vein occlusion</article-title>. <source>Front. Cell Dev. Biol.</source> <volume>13</volume>, <fpage>1609567</fpage>. <pub-id pub-id-type="doi">10.3389/fcell.2025.1609567</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gargari</surname>
<given-names>O. K.</given-names>
</name>
<name>
<surname>Habibi</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Enhancing medical AI with retrieval-augmented generation: a mini narrative review</article-title>. <source>Digit. Health</source> <volume>11</volume>, <fpage>20552076251337177</fpage>. <pub-id pub-id-type="doi">10.1177/20552076251337177</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goh</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Gallo</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hom</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Strong</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Weng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kerman</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Large language model influence on diagnostic reasoning: a randomized clinical trial</article-title>. <source>JAMA Netw. Open</source> <volume>7</volume> (<issue>10</issue>), <fpage>e2440969</fpage>. <pub-id pub-id-type="doi">10.1001/jamanetworkopen.2024.40969</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gong</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.-T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.-M.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.-J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.-J.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Development and research status of intelligent ophthalmology in China</article-title>. <source>Int. J. Ophthalmol.</source> <volume>17</volume> (<issue>12</issue>), <fpage>2308</fpage>&#x2013;<lpage>2315</lpage>. <pub-id pub-id-type="doi">10.18240/ijo.2024.12.20</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gulshan</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Coram</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Stumpe</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Narayanaswamy</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Development and validation of a deep learning algorithm for detection of diabetic retinopathy in retinal fundus photographs</article-title>. <source>jama</source> <volume>316</volume> (<issue>22</issue>), <fpage>2402</fpage>&#x2013;<lpage>2410</lpage>. <pub-id pub-id-type="doi">10.1001/jama.2016.17216</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>G&#xfc;nay</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>&#xd6;zt&#xfc;rk</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Yi&#x11f;it</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>The accuracy of gemini, GPT-4, and GPT-4o in ECG analysis: a comparison with cardiologists and emergency medicine specialists</article-title>. <source>Am. J. Emerg. Med.</source> <volume>84</volume>, <fpage>68</fpage>&#x2013;<lpage>73</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajem.2024.07.043</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Deepseek-r1: incentivizing reasoning capability in llms <italic>via</italic> reinforcement learning</article-title>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Homolak</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Opportunities and risks of ChatGPT in medicine, science, and academic publishing: a modern Promethean dilemma</article-title>. <source>Croat. Med. J.</source> <volume>64</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>3</lpage>. <pub-id pub-id-type="doi">10.3325/cmj.2023.64.1</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Keles</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Illeez</surname>
<given-names>O. G.</given-names>
</name>
<name>
<surname>Erbagci</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Giray</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Artificial intelligence-generated responses to frequently asked questions on coccydynia: evaluating the accuracy and consistency of GPT-4o&#x27;s performance</article-title>. <source>Arch. Rheumatol.</source> <volume>40</volume> (<issue>1</issue>), <fpage>63</fpage>&#x2013;<lpage>71</lpage>. <pub-id pub-id-type="doi">10.46497/ArchRheumatol.2025.10966</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lewis</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Perez</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Piktus</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Petroni</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Karpukhin</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Goyal</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Retrieval-augmented generation for knowledge-intensive nlp tasks</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>33</volume>, <fpage>9459</fpage>&#x2013;<lpage>9474</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2005.11401</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>P. H.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Transformative potential of GPT-4o in clinical immunology and allergy: opportunities and challenges of real-time voice interaction</article-title>. <source>Asia Pac Allergy</source> <volume>14</volume> (<issue>4</issue>), <fpage>232</fpage>&#x2013;<lpage>233</lpage>. <pub-id pub-id-type="doi">10.5415/apallergy.0000000000000152</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Keel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>R. T.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Efficacy of a deep learning system for detecting glaucomatous optic neuropathy based on color fundus photographs</article-title>. <source>Ophthalmology</source> <volume>125</volume> (<issue>8</issue>), <fpage>1199</fpage>&#x2013;<lpage>1206</lpage>. <pub-id pub-id-type="doi">10.1016/j.ophtha.2018.01.023</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xiu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Large language model-based multimodal system for detecting and grading ocular surface diseases from smartphone images</article-title>. <source>Front. Cell Dev. Biol.</source> <volume>13</volume>, <fpage>1600202</fpage>. <pub-id pub-id-type="doi">10.3389/fcell.2025.1600202</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Meyer</surname>
<given-names>M. I.</given-names>
</name>
<name>
<surname>Costa</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Galdran</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mendon&#xe7;a</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Campilho</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>A deep neural network for vessel segmentation of scanning laser ophthalmoscopy images</article-title>,&#x201d; in <source>Lecture notes in computer science</source> (<publisher-name>Springer International Publishing</publisher-name>).</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mo&#xeb;ll</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sand Aronsson</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Akbar</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Medical reasoning in LLMs: an in-depth analysis of DeepSeek R1</article-title>. <source>Front. Artif. Intell.</source> <volume>8</volume>, <fpage>1616145</fpage>&#x2013;<lpage>2025</lpage>. <pub-id pub-id-type="doi">10.3389/frai.2025.1616145</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>Q. N.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Dang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S. Y.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Advancing question-answering in ophthalmology with retrieval-augmented generation (RAG): benchmarking open-source and proprietary large language models</article-title>. <source>Investigative Ophthalmol. and Vis. Sci.</source> <volume>66</volume> (<issue>8</issue>), <fpage>4638</fpage>. <pub-id pub-id-type="doi">10.1101/2024.11.18.24317510</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rao</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kamineni</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lie</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Dreyer</surname>
<given-names>K. J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Evaluating GPT as an adjunct for radiologic decision making: GPT-4 <italic>versus</italic> GPT-3.5 in a breast imaging pilot</article-title>. <source>J. Am. Coll. Radiol.</source> <volume>20</volume> (<issue>10</issue>), <fpage>990</fpage>&#x2013;<lpage>997</lpage>. <pub-id pub-id-type="doi">10.1016/j.jacr.2023.05.003</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sandmann</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hegselmann</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fujarski</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bickmann</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wild</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Eils</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Benchmark evaluation of DeepSeek large language models in clinical decision-making</article-title>. <source>Nat. Med.</source> <pub-id pub-id-type="doi">10.1038/s41591-025-03727-2</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schlegl</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Waldstein</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Bogunovic</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Endstra&#xdf;er</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Sadeghipour</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Philip</surname>
<given-names>A.-M.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Fully automated detection and quantification of macular fluid in OCT using deep learning</article-title>. <source>Ophthalmology</source> <volume>125</volume> (<issue>4</issue>), <fpage>549</fpage>&#x2013;<lpage>558</lpage>. <pub-id pub-id-type="doi">10.1016/j.ophtha.2017.10.031</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shea</surname>
<given-names>Y. F.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>C. M. Y.</given-names>
</name>
<name>
<surname>Ip</surname>
<given-names>W. C. T.</given-names>
</name>
<name>
<surname>Luk</surname>
<given-names>D. W. A.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>S. S. W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Use of GPT-4 to analyze medical records of patients with extensive investigations and delayed diagnosis</article-title>. <source>JAMA Netw. Open</source> <volume>6</volume> (<issue>8</issue>), <fpage>e2325000</fpage>. <pub-id pub-id-type="doi">10.1001/jamanetworkopen.2023.25000</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shea</surname>
<given-names>Y. F.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>N. C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Limitations of GPT-4 in analyzing real-life medical notes related to cognitive impairment</article-title>. <source>Psychogeriatrics</source> <volume>23</volume> (<issue>5</issue>), <fpage>885</fpage>&#x2013;<lpage>887</lpage>. <pub-id pub-id-type="doi">10.1111/psyg.13002</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>T. Y. A.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Enhancing large language model performance on ophthalmology board-style questions with retrieval-augmented generation</article-title>. <source>Investigative Ophthalmol. and Vis. Sci.</source> <volume>66</volume> (<issue>8</issue>), <fpage>3930</fpage>.</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tan</surname>
<given-names>D. N. H.</given-names>
</name>
<name>
<surname>Tham</surname>
<given-names>Y.-C.</given-names>
</name>
<name>
<surname>Koh</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Loon</surname>
<given-names>S. C.</given-names>
</name>
<name>
<surname>Aquino</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Lun</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Evaluating chatbot responses to patient questions in the field of glaucoma</article-title>. <source>Front. Med.</source> <volume>11</volume>, <fpage>1359073</fpage>. <pub-id pub-id-type="doi">10.3389/fmed.2024.1359073</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Luenam</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ran</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Quadeer</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Raman</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sen</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Detection of diabetic retinopathy from ultra-widefield scanning laser ophthalmoscope images: a multicenter deep learning analysis</article-title>. <source>Ophthalmol. Retina</source> <volume>5</volume> (<issue>11</issue>), <fpage>1097</fpage>&#x2013;<lpage>1106</lpage>. <pub-id pub-id-type="doi">10.1016/j.oret.2021.01.013</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tangsrivimol</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Darzidehkalani</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Virk</surname>
<given-names>H. U. H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Egger</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Benefits, limits, and risks of ChatGPT in medicine</article-title>. <source>Front. Artif. Intell.</source> <volume>8</volume>, <fpage>1518049</fpage>&#x2013;<lpage>2025</lpage>. <pub-id pub-id-type="doi">10.3389/frai.2025.1518049</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thirunavukarasu</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Ting</surname>
<given-names>D. S. J.</given-names>
</name>
<name>
<surname>Elangovan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Gutierrez</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>T. F.</given-names>
</name>
<name>
<surname>Ting</surname>
<given-names>D. S. W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Large language models in medicine</article-title>. <source>Nat. Med.</source> <volume>29</volume> (<issue>8</issue>), <fpage>1930</fpage>&#x2013;<lpage>1940</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Waisberg</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Masalkhi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zaman</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Sarker</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>A. G.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>GPT-4 to document ophthalmic post-operative complications</article-title>. <source>Eye (Lond)</source> <volume>38</volume> (<issue>3</issue>), <fpage>414</fpage>&#x2013;<lpage>415</lpage>. <pub-id pub-id-type="doi">10.1038/s41433-023-02731-5</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Evaluating the performance of ChatGPT in patient consultation and image-based preliminary diagnosis in thyroid eye disease</article-title>. <source>Front. Med. (Lausanne)</source> <volume>12</volume>, <fpage>1546706</fpage>. <pub-id pub-id-type="doi">10.3389/fmed.2025.1546706</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sidhu</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>ChatGPT: is it good for our glaucoma patients?</article-title> <source>Front. Ophthalmol. (Lausanne)</source> <volume>3</volume>, <fpage>1260415</fpage>. <pub-id pub-id-type="doi">10.3389/fopht.2023.1260415</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ning</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Keppo</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bitterman</surname>
<given-names>D. S.</given-names>
</name>
<etal/>
</person-group> (<year>2025a</year>). <article-title>Retrieval-augmented generation for generative artificial intelligence in health care</article-title>. <source>npj Health Syst.</source> <volume>2</volume> (<issue>1</issue>), <fpage>2</fpage>. <pub-id pub-id-type="doi">10.1038/s44401-024-00004-1</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>W.-H.</given-names>
</name>
<name>
<surname>Shao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.-W.</given-names>
</name>
</person-group>
<collab>Expert Workgroup of Guidelines on Clinical Research Evaluation of Artificial Intelligence in Ophthalmology 2023, Ophthalmic Imaging and Intelligent Medicine Branch of Chinese Medicine Education Association, Intelligent Medicine Committee of Chinese Medicin</collab>
<collab>e Education Association</collab> (<year>2023</year>). <article-title>Guidelines on clinical research evaluation of artificial intelligence in ophthalmology (2023)</article-title>. <source>Int. J. Ophthalmol.</source> <volume>16</volume> (<issue>9</issue>), <fpage>1361</fpage>&#x2013;<lpage>1372</lpage>. <pub-id pub-id-type="doi">10.18240/ijo.2023.09.02</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L. J. F. i.O.</given-names>
</name>
</person-group> (<year>2025b</year>). <article-title>Early prediction of colorectal adenoma risk: leveraging large-language model for clinical electronic medical record data</article-title>. <source>Front. Oncol.</source> <volume>15</volume>, <fpage>1508455</fpage>. <pub-id pub-id-type="doi">10.3389/fonc.2025.1508455</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>