<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" dtd-version="1.3" article-type="other">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Med.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Medicine</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Med.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2296-858X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmed.2025.1744657</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Policy and Practice Reviews</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Ten tips for utilizing AI to generate high quality OSCE stations in medical education</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Zafar</surname> <given-names>Imran</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<uri xlink:href="https://loop.frontiersin.org/people/3314454"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Aboueisha</surname> <given-names>Hadeel</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<uri xlink:href="https://loop.frontiersin.org/people/3309274"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Elhassan</surname> <given-names>Ibrahim</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Shersad</surname> <given-names>Fouzia</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2641769"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Caliskan</surname> <given-names>Suleyman Ayhan</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<uri xlink:href="https://loop.frontiersin.org/people/3317076"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Syeda</surname> <given-names>Asma Fatima</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Al-Houqani</surname> <given-names>Mohammed</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<uri xlink:href="https://loop.frontiersin.org/people/533033"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Magzoub</surname> <given-names>Mohi Eldin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<uri xlink:href="https://loop.frontiersin.org/people/3276145"/>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>College of Medicine and Health Sciences, United Arab Emirates University</institution>, <city>Al Ain</city>, <country country="ae">United Arab Emirates</country></aff>
<aff id="aff2"><label>2</label><institution>National Institute of Health Specialties</institution>, <city>Al Ain</city>, <country country="ae">United Arab Emirates</country></aff>
<author-notes>
<corresp id="c001"><label>&#x0002A;</label>Correspondence: Mohammed Al-Houqani, <email xlink:href="mailto:alhouqani@uaeu.ac.ae">alhouqani@uaeu.ac.ae</email>; Mohi Eldin Magzoub, <email xlink:href="mailto:mmagzoub@uaeu.ac.ae">mmagzoub@uaeu.ac.ae</email></corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2026-01-16">
<day>16</day>
<month>01</month>
<year>2026</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1744657</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>11</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>15</day>
<month>12</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>22</day>
<month>12</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2026 Zafar, Aboueisha, Elhassan, Shersad, Caliskan, Syeda, Al-Houqani and Magzoub.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>Zafar, Aboueisha, Elhassan, Shersad, Caliskan, Syeda, Al-Houqani and Magzoub</copyright-holder>
<license>
<ali:license_ref start_date="2026-01-16">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<p>Artificial intelligence (AI) is reshaping medical education, offering novel solutions to long-standing challenges in clinical assessment design. One of the most resource-intensive components of assessment is the development of high-quality Objective Structured Clinical Examination (OSCE) stations that are valid, reliable, and aligned with curricular outcomes. Traditional approaches to OSCE case creation are time-consuming, vulnerable to inconsistency, and difficult to scale. AI, particularly large language models, has emerged as a powerful tool for generating realistic, diverse, and customizable clinical scenarios. However, its safe and effective use in high-stakes examinations requires structured guidance and faculty oversight. This manuscript presents 10 evidence-informed, practical tips for leveraging AI to generate OSCE stations that are pedagogically sound and clinically authentic. These tips draw on focused literature review, experiential insights from AI-assisted case development, and consensus from medical education experts. Together, they provide a framework for integrating AI into assessment workflows while ensuring quality, fairness, security, and ethical compliance. By following these recommendations, educators and examination committees can enhance efficiency, maintain validity, and prepare learners for the complexities of modern clinical practice.</p></abstract>
<kwd-group>
<kwd>AI generated</kwd>
<kwd>assessment</kwd>
<kwd>guidelines</kwd>
<kwd>medical education</kwd>
<kwd>OSCE</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declared that financial support was not received for this work and/or its publication.</funding-statement>
</funding-group>
<counts>
<fig-count count="2"/>
<table-count count="7"/>
<equation-count count="0"/>
<ref-count count="27"/>
<page-count count="14"/>
<word-count count="9569"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Healthcare Professions Education</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>The landscape of medical education is rapidly evolving, driven by advances in technology, increasing expectations of healthcare systems, and the need to graduate highly competent clinicians (<xref ref-type="bibr" rid="B1">1</xref>). Clinical assessments, particularly the Objective Structured Clinical Examination (OSCE) and similar structured formats such as PACES (UK), SOE (Canada), and the Comprehensive Clinical Examination (CCE) used in the United Arab Emirates, remain the gold standard for evaluating a range of competencies, including history taking, physical examination, procedural skills, clinical reasoning, and communication (<xref ref-type="bibr" rid="B2">2</xref>).</p>
<p>Formats such as PACES, SOE, and CCE highlight the vast global variation in how clinical examinations are planned, scheduled, and finally graded. Though all of these formats share essential aims, there are significant variances in station length, assessment focus, examiner deployment, and standard-setting techniques (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B3">3</xref>). This wide diversity presents a huge challenge for educational leaders, requiring them to regularly develop examples that are culturally appropriate, reproducible, and precisely linked to the assessment plan, often across various examination systems. To address this complexity, case development methodologies must be scalable and adaptive. AI-assisted technologies become particularly relevant in this context, with the ability to develop, alter, and standardize OSCE-type stations quickly across an array of different examination requirements.</p>
<p>It is precisely this growing complexity, driven by the diversity of OSCE formats, the expanding need for large examination banks across multiple specialties and assessment types, that highlights the need for more scalable and adaptive approaches to case creation. Under these conditions, the process of creating OSCE stations is labor-intensive and difficult to manage (<xref ref-type="bibr" rid="B3">3</xref>). Key challenges reported in the literature include ensuring blueprint coverage, achieving reproducibility, and maintaining fairness across stations and exam cycles (<xref ref-type="bibr" rid="B4">4</xref>). Inconsistencies in case quality, variability in examiner interpretation, and heterogeneous standardized patient (SP) portrayal can threaten validity and reliability (<xref ref-type="bibr" rid="B2">2</xref>). Moreover, traditional case-writing processes struggle to keep pace with rapidly evolving medical knowledge, emerging diseases, and updated clinical guidelines (<xref ref-type="bibr" rid="B5">5</xref>).</p>
<p>In this landscape, AI-assisted tools become particularly relevant, given their potential to generate, tailor, and standardize OSCE-type stations efficiently across a range of assessment formats. Artificial intelligence (AI) offers a transformative opportunity to address these challenges. Leveraging natural language processing, data synthesis, and pattern recognition, AI tools can generate realistic and diverse clinical scenarios, produce structured checklists, and rapidly adapt cases to specified learning outcomes (<xref ref-type="bibr" rid="B6">6</xref>). By integrating AI into the OSCE design process, educators can enhance scalability, reduce administrative burden, and support consistent quality across examination cycles (<xref ref-type="bibr" rid="B7">7</xref>).</p>
<p>Pilot studies report that AI-generated OSCE cases are feasible, useful, and time-saving, though expert oversight is required to ensure clinical accuracy (<xref ref-type="bibr" rid="B8">8</xref>). A randomized controlled trial confirmed that AI-generated practice stations are comparable to faculty-created ones in terms of student performance outcomes, suggesting their potential utility in formative and summative assessment preparation (<xref ref-type="bibr" rid="B9">9</xref>).</p>
<p>AI-powered virtual standardized patients (vSPs) further expand the potential of technology-enhanced assessment, allowing educators to simulate complex patient interactions and create highly realistic training encounters (<xref ref-type="bibr" rid="B10">10</xref>). These innovations collectively highlight the ability of AI to streamline case generation, enhance diversity and novelty in assessment content, and support scalability in high-stakes examinations.</p>
<p>Despite growing interest in using AI in assessment, existing frameworks remain broad and not tailored to the specific requirements of OSCE case development. Current recommendations is broadly focused on AI usage in education or assessment, but doesn&#x00027;t address critical OSCE-related requirements such as case authenticity, standardization, psychometric rigor, or safety safeguards. This gap underscores the need for a structured, evidence-informed framework to guide the responsible integration of AI into OSCE case creation. Moreover, uncritical use of generative AI can result in factual inaccuracies, algorithmic biases, and ethical risks. Human validation and structured oversight are necessary to ensure fairness and alignment with learning outcomes (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B12">12</xref>).</p>
<p>To address this gap, we employed an exploratory qualitative methodology integrating expert consultation, iterative analysis and consensus-building to develop practical, evidence-informed recommendations.</p>
<p>This manuscript introduces 10 actionable tips for medical educators, program directors, and examination committees seeking to incorporate AI into OSCE station development. These recommendations are grounded in current literature (<xref ref-type="bibr" rid="B13">13</xref>), experiential evidence from implementation in a national clinical examination setting, and expert consensus. Collectively, they aim to help educators harness AI to create clinically authentic, psychometrically sound, and secure assessments that prepare learners for the realities of modern healthcare practice.</p></sec>
<sec id="s2">
<title>Methods, setting, and context</title>
<sec>
<title>Study design</title>
<p>This work employed a qualitative, exploratory design to capture expert perspectives on the use of artificial intelligence (AI) for OSCE station generation. The approach focused on gathering experiential insights from educators who had directly implemented AI-assisted scenario development within high-stakes clinical examinations.</p>
</sec>
<sec>
<title>Participants</title>
<p>Six medical education experts were recruited through purposive sampling, selected intentionally to ensure representation across undergraduate and postgraduate programs, clinical specialties, and assessment leadership roles. All participants had prior experience in OSCE design, blueprinting, and faculty development, and had also been involved in pilot projects that used AI tools for scenario generation.</p>
</sec>
<sec>
<title>Data collection</title>
<p>Two members of the research team led an organized, in-person focus group meeting to gather data. In order to ensure that all participants responded to the same prompts and that fundamental topics were covered methodically, the session adhered to a set of pre-planned questions.</p>
<p>Due to the lack of audio recording, both facilitators took thorough notes in real time, recording participant reactions, areas of agreement or disagreement, and providing clarification on examples discussed. The facilitators checked and combined their notes right away to make sure they were accurate and comprehensive. Using standardized file-management practices, all data was safely kept on an encrypted institutional server.</p>
<p>Four predetermined domains were covered by the structured question set:</p>
<list list-type="bullet">
<list-item><p>Opportunities and challenges of using AI in OSCE case generation.</p></list-item>
<list-item><p>Strategies for integrating AI into assessment workflows.</p></list-item>
<list-item><p>Lessons learned from implementation during high-stakes examinations.</p></list-item>
<list-item><p>Recommendations for faculty training, validation, and ethical safeguards.</p></list-item>
</list>
<p>The discussions were supplemented by the authors&#x00027; experiential evidence from direct implementation of AI-assisted case generation during national OSCE clinical examinations.</p>
</sec>
<sec>
<title>Setting and context</title>
<p>The study was conducted within the context of the National Institute of Health Specialties (NIHS), United Arab Emirates. NIHS is responsible for administering final written and clinical examinations for residency programs across multiple specialties. Since its establishment, NIHS has conducted more than 45 clinical examinations, engaging over 200 examiners from diverse disciplines and training backgrounds. The NIHS experience provided a robust, real-world perspective on integrating AI into high-stakes assessment design at scale.</p>
</sec>
<sec>
<title>Data sources</title>
<p>Three distinct sources of information contributed to this work, each serving a different purpose and treated separately throughout the analysis:</p>
<list list-type="order">
<list-item><p>Experiential evidence&#x02014;This consisted of observational insights and reflective accounts from the authors&#x00027; direct involvement in AI-assisted OSCE case development during national examinations. These data were not collected through the focus group but emerged from real-world implementation experiences.</p></list-item>
<list-item><p>Focus group notes&#x02014;These notes were systematically documented during a structured, in-person expert focus group. They captured participant responses to predetermined questions, areas of agreement or divergence, and illustrative examples shared during the discussion. This dataset represents the formal qualitative evidence generated specifically for this study.</p></list-item>
<list-item><p>The focus group notes and experiential evidence were kept separate for analysis because they came from different places, had different goals, and had different levels of methodological control. Only the notes from the focus group were formally coded for themes. Experiential evidence was only used to put emerging themes in context and help us understand them, so that practice-based insights didn&#x00027;t change the way the coded dataset was set up.</p></list-item>
<list-item><p>Literature review&#x02014;A focused review of peer-reviewed studies, conference proceedings, and scoping reviews published over the past 5 years on AI applications in medical education, clinical scenario generation, and assessment validity (<xref ref-type="bibr" rid="B14">14</xref>, <xref ref-type="bibr" rid="B15">15</xref>).</p></list-item>
</list>
</sec>
<sec>
<title>Data analysis</title>
<p>Braun and Clarke&#x00027;s reflexive thematic analysis framework, following the stages of familiarization, coding, theme development, and iterative refinement, was applied to synthesize findings from focus group discussions. The structured focus group notes were coded using an inductive method to make sure that themes were based on participant viewpoints rather than presumptions. Initial coding recommendations were produced by AI-assisted tools, but the research team manually reviewed, validated, and improved each code to guarantee accuracy, contextual sensitivity, and compliance with reflexive thematic analysis principles. Any disagreements were reviewed cooperatively and settled by team consensus. Emergent themes were organized into 10 concrete, evidence-informed recommendations. These recommendations were iteratively refined through author consensus meetings, incorporating illustrative examples from real examination scenarios to maximize relevance and applicability for educators.</p>
</sec>
</sec>
<sec id="s3">
<title>Results and literature review</title>
<sec>
<title>Findings from the literature</title>
<p>Findings from the focus group discussions, together with experiential evidence, highlighted key opportunities and challenges in the use of AI for OSCE development. These insights were complemented by a focused literature review, which revealed a rapidly expanding body of evidence on AI in medical education and assessment. Together, these sources formed the foundation of a broader developmental workflow in which evidence was synthesized and iteratively refined to generate the 10 practical tips presented in the Results, as illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>Developmental workflow for generating the 10 tips.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1744657-g0001.tif">
<alt-text content-type="machine-generated">Pyramid diagram illustrating the process of developing ten tips for AI-generated clinical scenarios. From bottom to top: Experiential Evidence, Focused Literature Review and Assessment, Synthesis and Integration, Consensus and Refinement, and Final Output. Each layer represents a step in aligning evidence with practice.</alt-text>
</graphic>
</fig>
<p>Six key themes emerged:</p>
<list list-type="order">
<list-item><p>AI for case and scenario generation</p></list-item>
</list>
<p>Large language models (LLMs) have shown the ability to generate realistic and diverse OSCE stations that mimic authentic patient presentations (<xref ref-type="bibr" rid="B16">16</xref>). Studies demonstrate that AI can produce structured checklists, candidate instructions, standardized patient (SP) scripts, and examiner probes with acceptable face and content validity (<xref ref-type="bibr" rid="B7">7</xref>, <xref ref-type="bibr" rid="B10">10</xref>). However, faculty review is indispensable to ensure accuracy and curricular alignment (<xref ref-type="bibr" rid="B8">8</xref>).</p>
<list list-type="simple">
<list-item><p>2. Efficiency and scalability</p></list-item>
</list>
<p>AI-assisted case generation significantly reduces the time and effort required to produce examination materials (<xref ref-type="bibr" rid="B7">7</xref>). This efficiency enables broader coverage of clinical conditions, enhances sampling of competencies, and facilitates creation of large exam banks across multiple specialties (<xref ref-type="bibr" rid="B17">17</xref>).</p>
<list list-type="simple">
<list-item><p>3. Quality and validity considerations</p></list-item>
</list>
<p>Although promising, AI-generated outputs may contain factual inaccuracies, inconsistencies, or oversimplifications. Clinical, demographic, or cultural bias remains a concern. For these reasons, human oversight is essential to safeguard validity, reliability, and fairness (<xref ref-type="bibr" rid="B11">11</xref>).</p>
<list list-type="simple">
<list-item><p>4. Integration with assessment practices</p></list-item>
</list>
<p>Embedding AI-generated scenarios within programmatic assessment frameworks supports alignment with competency-based curricula and blueprint requirements (<xref ref-type="bibr" rid="B18">18</xref>). Emerging applications also include augmenting problem-based learning (PBL), adaptive testing, and blueprint design (<xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B20">20</xref>).</p>
<list list-type="simple">
<list-item><p>5. Ethical and practical considerations</p></list-item>
</list>
<p>Key issues include data confidentiality, intellectual property rights, transparency, and risk of over-reliance on AI (<xref ref-type="bibr" rid="B11">11</xref>). Institutional policies, faculty training, and continuous QA processes are emphasized as critical enablers of safe use (<xref ref-type="bibr" rid="B15">15</xref>).</p>
<list list-type="simple">
<list-item><p>6. Practical implementation challenges and lessons learned</p></list-item>
</list>
<p>Implementing AI-assisted OSCE development presents several practical challenges. Institutional policies regarding data security, examination confidentiality, and approved digital tools may limit access to generative AI platforms, particularly for high-stakes assessments. Faculty preparedness varies widely, requiring targeted training in AI literacy, prompt design, and validation processes. Resource constraints, including limited access to secure enterprise-level AI systems and protected assessment management platforms, may further restrict implementation. During NIHS examinations, challenges included initial faculty skepticism, variability in prompt quality, and the need for additional review time to address AI-generated inconsistencies. Addressing these barriers requires institutional governance, phased implementation, and investment in faculty development and secure infrastructure.</p>
</sec>
</sec>
<sec id="s4">
<title>Future directions</title>
<p>Promising innovations include tailoring scenarios to Entrustable Professional Activities (EPAs), developing adaptive assessments, and leveraging multimodal AI (text, images, and audio) to create richer stations (<xref ref-type="bibr" rid="B15">15</xref>). Early work also highlights the role of AI in promoting equity by generating cases that reflect diverse patient populations (<xref ref-type="table" rid="T1">Table 1</xref>) (<xref ref-type="bibr" rid="B11">11</xref>).</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Literature themes and implications.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Theme</bold></th>
<th valign="top" align="left"><bold>Evidence from literature</bold></th>
<th valign="top" align="left"><bold>Implications for clinical exams</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">AI for case generation</td>
<td valign="top" align="left">LLMs generate OSCE stems, SP scripts, and checklists with acceptable validity (<xref ref-type="bibr" rid="B7">7</xref>, <xref ref-type="bibr" rid="B10">10</xref>)</td>
<td valign="top" align="left">Can expand case diversity and authenticity if faculty oversight is applied. Thus, allowing for rapid production of clinically relevant scenarios yet requiring professional assessment to verify contextual accuracy and blueprint alignment.</td>
</tr>
<tr>
<td valign="top" align="left">Efficiency and scalability</td>
<td valign="top" align="left">AI reduces faculty workload and development time (<xref ref-type="bibr" rid="B7">7</xref>)</td>
<td valign="top" align="left">Enables larger exam banks and balanced blueprint coverage by supporting broader sampling of competencies and facilitating more frequent updates to assessment materials across specialties.</td>
</tr>
<tr>
<td valign="top" align="left">Quality and validity issues</td>
<td valign="top" align="left">Risk of factual errors and embedded bias (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B15">15</xref>)</td>
<td valign="top" align="left">Requires SME validation, iterative refinement, and post-exam review to preserve psychometric defensibility, identify bias, and uphold fairness in high-stakes situations.</td>
</tr>
<tr>
<td valign="top" align="left">Integration with practice</td>
<td valign="top" align="left">Supports blueprinting, PBL, and adaptive testing (<xref ref-type="bibr" rid="B15">15</xref>)</td>
<td valign="top" align="left">Enhances curricular alignment and consistency by supporting standardized case structures throughout exam cycles and facilitating competency mapping at scale.</td>
</tr>
<tr>
<td valign="top" align="left">Ethical and practical issues</td>
<td valign="top" align="left">Risks of bias, security, and over-reliance (<xref ref-type="bibr" rid="B15">15</xref>)</td>
<td valign="top" align="left">Calls for institutional governance and training including faculty development, transparent usage guidelines, and safe AI platforms to protect test integrity.</td>
</tr>
<tr>
<td valign="top" align="left">Future directions</td>
<td valign="top" align="left">Adaptive, multimodal, and equity-focused AI (<xref ref-type="bibr" rid="B15">15</xref>)</td>
<td valign="top" align="left">Expands innovation potential in assessment design offering opportunities for more inclusive, dynamic, and authentic OSCE scenarios that reflect diverse patient populations.</td>
</tr></tbody>
</table>
</table-wrap>
<p>To support conceptual integration of the findings, <xref ref-type="fig" rid="F2">Figure 2</xref> presents a visual synthesis of the 10 tips organized into four interrelated domains. This framework illustrates how AI-supported OSCE development spans competency alignment, quality assurance, fairness and security safeguards, and continuous improvement through feedback and psychometric monitoring. The figure serves as an orienting map for the detailed recommendations that follow.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>Conceptual framework for AI-supported OSCE station development.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmed-12-1744657-g0002.tif">
<alt-text content-type="machine-generated">Flowchart titled &#x0201C;AI Generation of OSCE&#x0201D; with four columns. Alignment with CBME: Tips 1-3 focus on competency-based blueprint, assessment design, and enriching AI with resources. Quality and Reliability: Tips 4-6 emphasize mastering prompt engineering, faculty training, and quality assurance of AI cases. Security and Fairness: Tips 7-8 address bias, fairness, security, and case variations. Efficiency and Improvement: Tips 9-10 suggest using AI for materials and continuous improvement through feedback.</alt-text>
</graphic>
</fig>
<sec>
<title>Tip 1: constructing a competency-based OSCE blueprint with AI</title>
<p>The development of a robust, competency-based blueprint is paramount to the quality and effectiveness of any Objective Structured Clinical Examination (OSCE).</p>
<sec>
<title>Traditional blueprinting: processes and associated challenges</title>
<p>Historically, blueprinting establishes validity by systematically linking the examination&#x00027;s content to the corresponding curriculum objectives, clinical disciplines, and competencies. This methodology ensures the proportional representation of essential domains, including physical examination, clinical reasoning, communication, and history taking. However, the manual construction of these blueprints is significantly resource-intensive and often introduces inconsistencies, particularly when assessments span numerous specialties and utilize large question banks.</p>
</sec>
<sec>
<title>Leveraging AI for standa<italic><bold>r</bold></italic>dization and efficiency</title>
<p>AI presents a significant opportunity to standardize and substantially streamline the entire blueprinting workflow. By inputting program learning outcomes and established competency frameworks into the AI tools, the system can autonomously generate preliminary blueprint drafts that successfully align specific competencies, case types, and required skills with the appropriate level of training.</p>
<sec>
<title>Example of a targeted prompt</title>
<p>A specific prompt, such as: &#x0201C;You are an OSCE assessment designer. Task: generate an OSCE blueprint for Year 3 undergraduate medical students, including the respiratory, cardiovascular, and gastrointestinal systems, ensuring alignment with the CanMEDS roles,&#x0201D; can quickly produce a comprehensive, structured draft for expert faculty review (<xref ref-type="table" rid="T2">Table 2</xref>).</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Blueprint snapshot (AI-assisted).</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Competency domain</bold></th>
<th valign="top" align="center"><bold>&#x00023; Stations</bold></th>
<th valign="top" align="center"><bold>Level</bold></th>
<th valign="top" align="left"><bold>Assessment focus</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">History and communication</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">Year 3</td>
<td valign="top" align="left">Focused history &#x0002B; shared decision-making</td>
</tr>
<tr>
<td valign="top" align="left">Physical examination</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">Year 3</td>
<td valign="top" align="left">Cardiovascular and respiratory systems</td>
</tr>
<tr>
<td valign="top" align="left">Clinical reasoning</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">Year 3</td>
<td valign="top" align="left">Differential diagnosis and prioritization</td>
</tr>
<tr>
<td valign="top" align="left">Professionalism and ethics</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">Year 3</td>
<td valign="top" align="left">Informed consent and patient advocacy</td>
</tr></tbody>
</table>
</table-wrap>
<p>This example shows how AI can operationalize curriculum outcomes into structured blueprints, ensuring proportional representation of competencies. To enhance novelty and avoid repetitive station designs, it is advisable to provide the AI system with examples of prior institutional blueprints. As a guiding principle, &#x0201C;<italic>provide the AI system with examples of prior blueprints to avoid repetition and encourage novelty across exam cycles</italic>.&#x0201D;</p>
</sec>
</sec>
<sec>
<title>Key takeaway</title>
<p>AI can generate draft OSCE blueprints aligned with competencies and curricular outcomes, enhancing efficiency and coverage. However, expert review remains indispensable for ensuring contextual appropriateness and psychometric soundness.</p>
</sec>
</sec>
<sec>
<title>Tip 2: establish clear assessment design and case structure before using AI</title>
<p>AI is a powerful tool, but its outputs are only as effective as the inputs it receives. One of the most common pitfalls in using AI for OSCE case generation is providing vague prompts without a structured assessment framework. In the absence of clear guidance, AI may generate cases that are verbose, unfocused, or misaligned with curricular goals (<xref ref-type="bibr" rid="B20">20</xref>).</p>
<p>To mitigate this risk, faculty should define key assessment design parameters before engaging AI. Establishing these parameters provides the structural clarity necessary for AI systems to generate focused and relevant OSCE cases.</p>
<p>At a minimum, assessment designers should specify the intended station format, such as history taking, physical examination, data interpretation, counseling, or procedural skills. They should also identify the primary competency focus, for example communication, professionalism, or clinical reasoning, and clearly articulate the tasks expected of the candidate.</p>
<p>Timing constraints and scoring structure should be defined in advance. Typical timing ranges from 7 to 10 min for undergraduate OSCEs and 12&#x02013;15 min for postgraduate assessments. Scoring approaches may include checklist based methods, global rating scales, or hybrid models. Together, these design decisions ensure that AI generated OSCE cases are aligned with assessment objectives and ready for faculty validation.</p>
<sec>
<title>Structured prompting for targeted cases</title>
<p>Providing the AI with this explicit structure results in significantly more targeted outputs compared to generic requests.</p>
<sec>
<title>Example of a structured prompt</title>
<p>Prompts that specify, &#x0201C;Create a 10-minute OSCE case for a Year 3 medical student on counseling a patient with new-onset type 2 diabetes, including candidate instructions, standardized patient role, and checklist with 10 items,&#x0201D; yield superior results compared to a vague request such as, &#x0201C;Write an OSCE case about diabetes.&#x0201D;</p>
<p>To facilitate this, institutions can develop case templates that standardize structure across specialties (<xref ref-type="bibr" rid="B21">21</xref>). These templates can then be embedded into AI prompts, reducing variability and enhancing reproducibility (<xref ref-type="table" rid="T3">Table 3</xref>).</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Structured case template.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Section</bold></th>
<th valign="top" align="left"><bold>Content</bold></th>
<th valign="top" align="left"><bold>Notes</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Candidate Instructions</td>
<td valign="top" align="left">&#x0201C;You are the junior doctor in the clinic&#x02026;&#x0201D;</td>
<td valign="top" align="left">Clear, time-limited</td>
</tr>
<tr>
<td valign="top" align="left">Patient (SP) role</td>
<td valign="top" align="left">45-year-old patient with chest pain</td>
<td valign="top" align="left">Include demeanor, key phrases</td>
</tr>
<tr>
<td valign="top" align="left">Candidate tasks</td>
<td valign="top" align="left">Take focused history, counsel patient on results</td>
<td valign="top" align="left">Align with competency framework</td>
</tr>
<tr>
<td valign="top" align="left">Scoring checklist</td>
<td valign="top" align="left">12 items, each mapped to skill domain</td>
<td valign="top" align="left">Include global rating component</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec>
<title>Key takeaway</title>
<p>AI performs best when given structured guidance. Pre-defining station format, competencies, and scoring structure ensures that AI-generated OSCE cases are aligned, targeted, and ready for faculty validation.</p>
</sec>
</sec>
<sec>
<title>Tip 3: enrich AI with contextual and evidence-based resources</title>
<p>The quality of AI generated OSCE cases depends strongly on the specificity and richness of the contextual information provided in the prompt. When prompts lack sufficient clinical, institutional, or cultural context, AI systems tend to produce generic or poorly contextualized scenarios that may not align with local practice or learner needs (<xref ref-type="bibr" rid="B8">8</xref>).</p>
<p>Enriching prompts with references to clinical guidelines, institutional assessment policies, and local epidemiological patterns strengthens content validity and improves alignment with authentic clinical practice.</p>
<p>This approach is particularly important for assessment because AI can be deliberately guided using established evidence based frameworks such as Entrustable Professional Activities, CanMEDS roles, or EmiratesMEDs competencies. Anchoring prompts to these frameworks helps ensure that generated cases target clearly defined learning outcomes and assessment purposes (<xref ref-type="bibr" rid="B18">18</xref>, <xref ref-type="bibr" rid="B22">22</xref>). Embedding these frameworks into prompts improves consistency across specialties and supports competency-based medical education (CBME).</p>
<p>In practical terms, prompt enrichment may include explicit reference to relevant clinical guidelines such as those issued by the ADA, NICE, or WHO, incorporation of demographic and cultural characteristics that reflect local patient populations, and alignment with institutional or national competency frameworks such as EmiratesMEDs, CanMEDS, or ACGME. Specifying disease prevalence or priority conditions based on local health needs further enhances relevance and educational defensibility.</p>
<p>The contrast between a weak and an enriched prompt illustrates this effect clearly. A prompt such as &#x0201C;Write an OSCE case on hypertension&#x0201D; is likely to yield a broad and unspecific scenario. In contrast, a prompt that asks the AI to &#x0201C;Generate a 10 min OSCE case for a Year 4 medical student in the UAE, focusing on counseling a patient recently diagnosed with type 2 diabetes, aligned with ADA 2025 guidelines, and reflecting cultural considerations in diet and lifestyle&#x0201D; produces a more realistic, contextually grounded, and assessment ready case.</p>
<sec>
<title>Key takeaway</title>
<p>Enriching AI prompts with guidelines, cultural context, and competency frameworks ensures that OSCE cases are not only clinically accurate but also relevant, socially accountable, and aligned with curricular outcomes.</p>
</sec>
</sec>
<sec>
<title>Tip 4: master the art of prompt engineering</title>
<p>The effectiveness of AI in OSCE case generation depends largely on the quality of the prompt provided. Poorly constructed prompts often yield vague, overly simplistic, or irrelevant scenarios (<xref ref-type="bibr" rid="B8">8</xref>). Conversely, structured, precise, and layered prompts can generate highly relevant cases, complete with candidate tasks, standardized patient (SP) roles, and scoring checklists (<xref ref-type="bibr" rid="B7">7</xref>).</p>
<p><italic>Prompt engineering</italic> is, therefore, a critical skill for educators using AI in assessment design. This involves breaking down requests into clear instructions, embedding contextual details, and specifying the desired output format. For example, prompts should identify the clinical scenario, learner level, competencies targeted, case duration, and assessment format. Adding output constraints (e.g., number of checklist items) supports standardization and enhances the defensibility of OSCE scores (<xref ref-type="bibr" rid="B20">20</xref>). Importantly, educators can also ask the AI itself to refine or recreate a better prompt, allowing for iterative improvement and more tailored case outputs.</p>
<p>Well engineered prompts play a critical role in ensuring that AI generated assessment materials accurately reflect the intended competencies and assessment constructs. By explicitly defining the task, target learner level, clinical focus, and competency framework, structured prompting reduces the risk of construct irrelevant variance and minimizes the introduction of unintended difficulty or bias. This is particularly important in OSCE design, where small variations in case framing can substantially influence performance.</p>
<p>In addition, structured prompting supports reproducibility and consistency across examination cycles. When prompts are standardized and systematically documented, AI generated cases can be regenerated, reviewed, and refined in a controlled manner, supporting fairness, comparability, and quality assurance within the assessment process.</p>
<p>A weak prompt such as &#x0201C;Write an OSCE case about asthma&#x0201D; provides minimal guidance to the AI and is therefore likely to generate a broad, generic scenario with limited educational or assessment value. In contrast, a well structured prompt that specifies the learner level, clinical context, task duration, and required outputs leads to a far more targeted and usable case. For example, asking the AI to &#x0201C;Generate a 10 min OSCE case for Year 4 undergraduate students on the acute management of an adult with asthma exacerbation in the emergency department, including candidate instructions, a standardized patient script with key phrases, examiner prompts, and a 12 item checklist mapped to communication, history taking, and clinical reasoning domains&#x0201D; results in a coherent, assessment ready OSCE station that is clearly aligned with intended competencies and scoring criteria.</p>
<p>The second prompt provides specificity (clinical scenario, learner level, setting, task) and constraints (duration, checklist length, competency mapping), resulting in a structured and usable output that supports standardization and inter-rater reliability.</p>
<p>Effective prompt engineering benefits from a structured and sequential approach. Using stepwise instructions, such as requesting the AI to first generate the case stem, followed by the standardized patient role and then the scoring checklist, helps control output flow and improves internal consistency.</p>
<p>Explicitly specifying competency alignment further strengthens prompt quality. Referencing established frameworks such as CanMEDS roles or EmiratesMEDs competencies ensures that AI-generated cases are clearly linked to intended learning outcomes and assessment standards.</p>
<p>Providing clear output constraints is also essential. Defining parameters such as the number of checklist items, station duration, and expected level of detail helps standardize AI outputs and reduces unnecessary variation across cases.</p>
<p>Prompt development should be treated as an iterative process. Refining prompts based on initial AI outputs, using structured feedback loops, allows educators to progressively improve relevance, clarity, and alignment with assessment goals.</p>
<p>Finally, employing meta-prompts, such as asking the AI to assess a prompt for clarity, completeness, or alignment with objectives, can further enhance design quality. This reflective layer supports more deliberate and defensible use of AI in OSCE case development.</p>
<p>Educators should also recognize that prompt refinement is iterative: initial outputs often require follow-up instructions such as &#x0201C;revise checklist to reduce redundancy&#x0201D; or &#x0201C;adjust SP role to reflect cultural communication styles.&#x0201D; This iterative dialogue ensures accuracy and contextual fit (<xref ref-type="bibr" rid="B19">19</xref>). Iteration also reduces construct drift and helps maintain fidelity to assessment specifications.</p>
<sec>
<title>Key takeaway</title>
<p>Prompt engineering transforms AI from a &#x0201C;generic text generator&#x0201D; into a structured assistant for OSCE design. Faculty who masters this skill can unlock AI&#x00027;s potential for creating clinically authentic, competency-aligned, and exam-ready cases.</p>
</sec>
</sec>
<sec>
<title>Tip 5: build faculty capacity and provide AI training</title>
<p>AI can enhance OSCE case development, but its outputs are not self-validating. Faculty play a central role in ensuring accuracy, fairness, and alignment with curricular outcomes (<xref ref-type="bibr" rid="B18">18</xref>). Across the literature, recurring limitations; such as factual inaccuracies, cultural or demographic bias, and inconsistency in output detail, were identified, reinforcing the need for purposeful faculty preparation (<xref ref-type="bibr" rid="B8">8</xref>). Therefore, capacity building and faculty development are essential for safe and effective integration of AI into assessment design. Strengthening faculty capability also supports defensible assessment decisions and reduces variability in how AI tools are used across departments.</p>
<sec>
<title>Core competencies for AI-enabled assessment design</title>
<p>Faculty development in AI enabled assessment design should focus on building a coherent set of competencies rather than isolated technical skills. Educators need a clear understanding of what AI systems can and cannot reliably do in assessment contexts, including commonly reported limitations and failure modes documented in the literature. This foundational knowledge allows faculty to use AI as a supportive tool rather than an uncritical content generator.</p>
<p>Equally important are prompt engineering skills that support reproducible and purpose driven case generation. Faculty should be able to design prompts that specify learner level, assessment intent, competency targets, and expected outputs, thereby improving consistency and usability across assessment cycles. Training should also address validation and quality assurance practices, enabling educators to systematically review AI generated content for factual accuracy, bias, and alignment with intended constructs.</p>
<p>Ethical considerations must be embedded throughout this training, including issues related to fairness, equity, data security, and the protection of examination confidentiality. Finally, faculty should be familiar with institutional guidelines governing the responsible use of AI in assessment to ensure consistency, transparency, and appropriate oversight across programs and departments.</p>
<p>Workshops and short training modules can build these skills effectively (<xref ref-type="table" rid="T4">Table 4</xref>). Embedding AI training into faculty development programs ensures long-term sustainability (<xref ref-type="bibr" rid="B13">13</xref>). Early adopters can support institutional rollout by mentoring colleagues and participating in communities of practice that encourage shared learning and reduce duplication of effort (<xref ref-type="bibr" rid="B15">15</xref>).</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Mini-curriculum for AI training in assessment.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Module</bold></th>
<th valign="top" align="left"><bold>Learning objectives</bold></th>
<th valign="top" align="left"><bold>Format</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Introduction to AI in assessment</td>
<td valign="top" align="left">Explain capabilities, limitations, and use cases</td>
<td valign="top" align="left">Seminar &#x0002B; demo</td>
</tr>
<tr>
<td valign="top" align="left">Prompt engineering</td>
<td valign="top" align="left">design structured prompts for OSCE case generation</td>
<td valign="top" align="left">Hands-on workshop</td>
</tr>
<tr>
<td valign="top" align="left">Validation and QA</td>
<td valign="top" align="left">Apply frameworks for reviewing AI-generated cases</td>
<td valign="top" align="left">Peer-review session</td>
</tr>
<tr>
<td valign="top" align="left">Ethics and fairness</td>
<td valign="top" align="left">Identify risks of bias, inequity, and exam security issues</td>
<td valign="top" align="left">Case discussion</td>
</tr>
<tr>
<td valign="top" align="left">Practical application</td>
<td valign="top" align="left">Generate and refine an OSCE case with AI support</td>
<td valign="top" align="left">Simulation exercise</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<title>Key takeaway</title>
<p>Faculty development is the cornerstone of safe AI integration. Training educators in AI literacy, prompt design, and validation methods empowers them to harness AI responsibly, ensuring assessment quality and fairness.</p>
</sec>
</sec>
<sec>
<title>Tip 6: validate and quality-assure AI-generated cases</title>
<p>Even when prompts are carefully structured, AI generated OSCE cases may still contain factual inaccuracies, cultural mismatches, or internal inconsistencies. If these issues are not systematically identified and corrected, they can undermine exam validity, fairness, and reliability (<xref ref-type="bibr" rid="B23">23</xref>). For this reason, AI generated cases should not be adopted directly for formative or summative assessment without a clearly defined validation and quality assurance process.</p>
<p>Validation should begin with expert review, in which subject matter experts examine AI outputs for clinical accuracy, alignment with current guidelines, and relevance to the local clinical and educational context (<xref ref-type="bibr" rid="B21">21</xref>). This step is essential for identifying subtle clinical errors or inappropriate assumptions that may not be immediately apparent in automated outputs.</p>
<p>Following expert review, cases should be cross checked against the assessment blueprint to confirm that they address the intended competencies and are appropriately positioned within the overall examination structure (<xref ref-type="bibr" rid="B3">3</xref>). A subsequent peer review by an independent faculty member can further strengthen quality by identifying gaps, redundancies, or inconsistencies in task design, scoring criteria, or case flow.</p>
<p>Pilot testing provides an additional safeguard by exposing the case to a small group of students or residents prior to formal use. This step helps identify ambiguities, timing problems, and imbalances within checklists or examiner prompts, which are particularly common in early AI generated drafts that vary in depth, pacing, and clarity (<xref ref-type="bibr" rid="B24">24</xref>). After implementation, ongoing psychometric monitoring is required to evaluate case performance using indicators such as difficulty indices, discrimination, and inter rater reliability, thereby confirming that the case functions as intended in practice (<xref ref-type="bibr" rid="B9">9</xref>).</p>
<p>Although this process closely resembles traditional quality assurance procedures for OSCE development, it must be deliberately adapted to address AI specific risks, including hallucinated content and embedded biases. Systematic validation therefore remains a critical safeguard when integrating AI generated cases into assessment programs (<xref ref-type="table" rid="T5">Table 5</xref>).</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>AI case validation workflow.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Step</bold></th>
<th valign="top" align="left"><bold>Process</bold></th>
<th valign="top" align="left"><bold>Outcome</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">AI case generation</td>
<td valign="top" align="left">Draft OSCE case using structured prompt</td>
<td valign="top" align="left">Baseline scenario created</td>
</tr>
<tr>
<td valign="top" align="left">SME review</td>
<td valign="top" align="left">Expert checks for accuracy and guideline adherence</td>
<td valign="top" align="left">Errors corrected, context aligned</td>
</tr>
<tr>
<td valign="top" align="left">Peer review</td>
<td valign="top" align="left">Second reviewer evaluates structure and checklist</td>
<td valign="top" align="left">Standardization and refinement</td>
</tr>
<tr>
<td valign="top" align="left">Pilot testing</td>
<td valign="top" align="left">Small-scale trial with learners/SPs</td>
<td valign="top" align="left">Identifies timing or clarity issues</td>
</tr>
<tr>
<td valign="top" align="left">Psychometric analysis</td>
<td valign="top" align="left">Post-exam metrics reviewed</td>
<td valign="top" align="left">Confirms validity and reliability</td>
</tr></tbody>
</table>
</table-wrap>
<p>This streamlined workflow illustrates how AI output integrates into existing QA systems while requiring adaptations for AI-specific risks.</p>
<sec>
<title>Key takeaway</title>
<p>AI can accelerate case creation but cannot replace human oversight. A structured validation workflow ensures that AI-generated OSCE stations are accurate, fair, and psychometrically robust.</p>
</sec>
</sec>
<sec>
<title>Tip 7: address bias, ensure fairness, and safeguard exam security</title>
<p>AI systems learn from large datasets that often embed historical, cultural, or clinical biases. Left unchecked, these biases can appear in OSCE cases, perpetuating stereotypes or underrepresenting important patient groups (<xref ref-type="bibr" rid="B23">23</xref>). For example, AI may default to male patients for cardiac cases or neglect culturally sensitive expressions of symptoms (<xref ref-type="bibr" rid="B25">25</xref>). Such biases undermine fairness and may disadvantage certain candidate groups.</p>
<p>Additionally, AI introduces new exam security risks. Cases generated or stored on unsecured platforms could be accessed by candidates, raising concerns about content leakage and academic integrity (<xref ref-type="bibr" rid="B26">26</xref>). Institutions must therefore combine bias detection with robust security protocols.</p>
<p>Promoting fairness and safeguarding exam security are critical when integrating AI into OSCE development. One essential step is systematic bias screening, which helps counteract AI&#x00027;s tendency to reproduce default demographic patterns or generate unrepresentative cases. Faculty should ensure diversity across patient demographics, including age, gender, ethnicity, and socioeconomic status, avoid stereotypical portrayals such as linking specific conditions or behaviors to particular groups, and include clinical conditions that reflect local epidemiology.</p>
<p>Fairness should also be evaluated through structured audits of AI-generated cases. Mapping cases to blueprint domains helps ensure balanced representation across competencies and content areas, while review by subject matter expert panels allows identification of unintended differences in case difficulty across learner subgroups.</p>
<p>Cultural adaptation is another key consideration. Standardized patient scripts should be reviewed and adapted to reflect locally appropriate communication styles and sociocultural norms. Involving regional educators in reviewing language, cues, and contextual details helps ensure authenticity and reduces the risk of cultural mismatch.</p>
<p>Robust security measures are necessary to protect exam integrity. Institutions should use secure, institutional, or encrypted AI platforms and avoid public tools for generating or storing final examination content. AI-generated cases should be housed within secure assessment management systems that maintain access logs, and case variants should be rotated across exam cycles to reduce predictability and content leakage.</p>
<p>Finally, fairness and security require ongoing monitoring after exam administration. Performance data should be analyzed across variables such as gender, language, or educational background to detect differential item functioning. Stations that demonstrate evidence of unfairness or bias should be revised or retired to maintain the integrity and defensibility of the assessment.</p>
<sec>
<title>Example&#x02014;NIHS case review</title>
<p>During an NIHS OSCE cycle, an AI-generated communication case described a single-parent household where the mother was assumed to be the caregiver. Faculty reviewers flagged this as gender-stereotyped and revised the case to make the caregiver role neutral. Subsequent psychometric analysis showed no gender-based performance gap, supporting fairness.</p>
<p>This example demonstrates how fairness checks and psychometric monitoring intersect in evaluating AI-generated content.</p></sec>
<sec>
<title>Key takeaway</title>
<p>Bias and security are non-negotiable in AI-assisted assessment. Institutions should adopt a bias audit checklist, cultural review, and secure storage protocols to ensure fairness, protect exam integrity, and maintain public trust in high-stakes assessments.</p>
</sec>
</sec>
<sec>
<title>Tip 8: generate diverse and realistic case variations</title>
<p>Reusing identical OSCE stations across exam cycles increases the risk of content leakage and predictability. This undermines fairness by rewarding candidates who have prior access to cases, rather than those demonstrating competence (<xref ref-type="bibr" rid="B3">3</xref>). Traditionally, creating multiple case versions is resource-intensive, limiting the number of variations available for high-stakes exams.</p>
<p>AI can generate multiple permutations of a core case by varying demographics, context, or complicating factors while preserving the intended learning outcomes (<xref ref-type="bibr" rid="B27">27</xref>). This not only strengthens exam security but also improves authenticity by reflecting the variability seen in real-world clinical practice (<xref ref-type="bibr" rid="B9">9</xref>).</p>
<p>Variation generation also helps mitigate recurring AI limitations; such as overly generic outputs, by prompting more contextualized and diverse scenarios from a single validated core case.</p>
<p>The generation of multiple case variants using AI should begin with clearly defining the core elements that must remain constant across all versions. These include the primary diagnosis, the intended competencies to be assessed, and the essential checklist items that underpin scoring and decision making. Fixing these elements helps ensure construct consistency and protects against unintended shifts in difficulty.</p>
<p>Once the core elements are established, selected variables can be deliberately modified to create meaningful variation. These may include patient demographics such as age, gender, or ethnicity, as well as differences in clinical context such as emergency, outpatient, or rural settings. Additional variation can be introduced through comorbidities, including the presence or absence of relevant risk factors, and psychosocial factors such as caregiver involvement or language barriers, provided these do not alter the underlying construct being assessed.</p>
<p>Structured prompts should then be used to instruct the AI to generate a limited number of variants, typically three to five, while explicitly maintaining comparable difficulty, scope, and station timing. Clear constraints within the prompt are essential to prevent drift in case complexity or assessment focus across versions.</p>
<p>Each generated version must undergo independent validation to confirm clinical accuracy, fairness, and alignment with the assessment blueprint, following the same quality assurance principles applied to single AI generated cases. Finally, validated case variants can be rotated across examination circuits or assessment sessions to reduce recall bias and enhance exam security, while preserving comparability across cohorts (<xref ref-type="table" rid="T6">Table 6</xref>).</p>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Case variations for acute chest pain.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Version</bold></th>
<th valign="top" align="left"><bold>Demographics</bold></th>
<th valign="top" align="left"><bold>Context</bold></th>
<th valign="top" align="left"><bold>Key variation</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">A</td>
<td valign="top" align="left">58M, smoker, diabetic</td>
<td valign="top" align="left">ED after exercise</td>
<td valign="top" align="left">Classic ACS with risk factors</td>
</tr>
<tr>
<td valign="top" align="left">B</td>
<td valign="top" align="left">45F, no comorbidities</td>
<td valign="top" align="left">At rest after emotional stress</td>
<td valign="top" align="left">Differentiate ACS vs. panic attack</td>
</tr>
<tr>
<td valign="top" align="left">C</td>
<td valign="top" align="left">70M, hypertensive</td>
<td valign="top" align="left">Mild confusion with dyspnea</td>
<td valign="top" align="left">Geriatric considerations</td>
</tr>
<tr>
<td valign="top" align="left">D</td>
<td valign="top" align="left">52F, postpartum</td>
<td valign="top" align="left">Pain radiating to back</td>
<td valign="top" align="left">Include PE as differential</td>
</tr></tbody>
</table>
</table-wrap>
<p>This approach ensures that all versions test the same learning outcomes (history, risk factor identification, urgent management) but in diverse and realistic contexts.</p>
<sec>
<title>Key takeaway</title>
<p>AI-generated case permutations enhance exam security, broaden content diversity, and mirror clinical variability. Faculty should carefully define fixed vs. flexible elements, validate each version, and strategically rotate them across exam cycles.</p>
<p>When managed systematically, variation generation also reduces AI&#x00027;s tendency toward repetition and ensures more equitable, realistic, and defensible assessments.</p>
</sec>
</sec>
<sec>
<title>Tip 9: use AI to generate supporting materials for OSCE stations</title>
<p>Developing an OSCE case requires much more than writing a clinical vignette. Each station needs candidate instructions, standardized patient (SP) scripts, examiner checklists, global rating scales, and sometimes supplementary resources such as lab results, imaging, or patient education materials. Traditionally, creating these materials is highly time-consuming and requires coordination among multiple faculty (<xref ref-type="bibr" rid="B27">27</xref>).</p>
<p>AI can streamline this process by generating structured supporting materials from a validated case stem. This not only reduces faculty workload but also enhances consistency across exam circuits, improving fairness and reproducibility (<xref ref-type="bibr" rid="B7">7</xref>).</p>
<sec>
<title>Steps to generate supporting materials with AI</title>
<p>The development of supporting materials using AI should begin only after a case has been fully validated through expert review and blueprint alignment, as outlined in Tip 6. Using unvalidated cases as inputs risks propagating clinical inaccuracies or construct misalignment across all associated materials. A validated core case provides a stable foundation for generating standardized patient instructions, examiner tools, and supplementary resources.</p>
<p>Standardized patient scripts can then be generated to support consistent portrayal across candidates. These scripts should explicitly describe the patient&#x00027;s affect, such as appearing anxious or distressed, and clearly specify verbal cues, including spontaneous complaints and information that should be disclosed only if directly elicited. Non verbal guidance, such as facial expressions, posture, or body language, should also be included. Particular emphasis should be placed on clarity and precision, as vague or overly narrative AI generated scripts are prone to misinterpretation by standardized patients.</p>
<p>Examiner tools should be created in parallel to ensure alignment with the intended assessment constructs. AI can be prompted to generate structured checklists with anchored scoring categories, such as 0 to 2 or 0 to 3 scales, alongside global rating scales that capture overall clinical reasoning and communication skills (<xref ref-type="bibr" rid="B24">24</xref>). Anchored scoring frameworks are especially important for supporting inter rater reliability and limiting variability introduced by loosely structured AI outputs.</p>
<p>AI may also be used to generate supplementary materials, including laboratory results, ECG summaries, imaging descriptions, or patient information leaflets for communication focused stations. These materials require careful review, as AI generated clinical data may contain internal inconsistencies or values that do not align with the clinical narrative of the case.</p>
<p>Finally, all supporting materials should be reviewed collectively to ensure internal consistency. Candidate instructions, standardized patient scripts, examiner checklists, and supplementary data must align in terms of clinical details, timing, and expected performance to preserve fairness and assessment validity.</p>
<p>An example of an AI generated standardized patient script excerpt illustrates this approach. The patient is described as appearing anxious and frequently holding the chest, with an opening line stating that the pain started 45 min ago. If asked, the patient reports nausea and sweating while denying cough, and non verbal behavior includes grimacing and pressing a hand against the sternum. Such structured detail supports consistent portrayal and clearer interpretation by standardized patients (<xref ref-type="table" rid="T7">Table 7</xref>).</p>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>Checklist (excerpt).</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Item</bold></th>
<th valign="top" align="center"><bold>Score (0&#x02013;2)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Asked onset, duration, character of pain</td>
<td valign="top" align="center">0&#x02013;2</td>
</tr>
<tr>
<td valign="top" align="left">Asked about radiation (jaw, arm, back)</td>
<td valign="top" align="center">0&#x02013;2</td>
</tr>
<tr>
<td valign="top" align="left">Inquired about cardiovascular risk factors</td>
<td valign="top" align="center">0&#x02013;2</td>
</tr>
<tr>
<td valign="top" align="left">Suggested urgent ECG</td>
<td valign="top" align="center">0&#x02013;2</td>
</tr></tbody>
</table>
</table-wrap>
<p>Global rating scale:</p>
<list list-type="simple">
<list-item><p>1 = unsatisfactory: missed key data, poor organization</p></list-item>
<list-item><p>2 = borderline: partial data, limited prioritization</p></list-item>
<list-item><p>3 = satisfactory: systematic, identified ACS features</p></list-item>
<list-item><p>4 = excellent: comprehensive, prioritized urgent action</p></list-item>
</list></sec>
<sec>
<title>Key takeaway</title>
<p>AI can generate SP scripts, examiner guides, and supplementary data from a single validated case, saving time and ensuring consistency. Faculty review remains essential to ensure clinical accuracy and cultural appropriateness.</p>
</sec>
</sec>
<sec>
<title>Tip 10: ensure continuous improvement through psychometric feedback and iteration</title>
<p>Introducing AI into OSCE case generation is not a one-time innovation but a continuous process. Even well-designed, validated stations require ongoing evaluation to confirm that they function as intended in real testing environments. Psychometric feedback and iterative refinement ensure that AI-generated cases remain valid, reliable, and fair over time (<xref ref-type="bibr" rid="B3">3</xref>).</p>
<p>This ongoing evaluation is particularly important because several limitations identified across the literature; such as inconsistency in output structure, variable difficulty, and occasional factual errors, can only be detected through post-administration monitoring.</p>
<sec>
<title>Steps for continuous improvement</title>
<p>Continuous improvement of AI generated OSCE stations should be grounded in systematic post examination evaluation. After each assessment cycle, key psychometric metrics should be collected, including item difficulty based on score distributions, discrimination indices that indicate how well the station differentiates between higher and lower performing candidates, and inter rater reliability across examiners (<xref ref-type="bibr" rid="B24">24</xref>). Together, these indicators help determine whether AI generated stations perform comparably to traditionally developed ones and support defensible assessment decisions.</p>
<p>Quantitative data should be complemented by structured feedback from candidates, examiners, and standardized patients. Candidate feedback is particularly useful for identifying issues related to clarity of instructions and perceived fairness, while examiner and standardized patient feedback can reveal ambiguities, unrealistic portrayals, or practical challenges in case execution that may not be evident from psychometric data alone.</p>
<p>Based on these findings, cases should be revised iteratively. This may involve refining checklists, clarifying instructions, or adjusting station timing to better match the intended level of performance. AI can be used efficiently at this stage to regenerate revised versions, provided that the validated core content and competency targets are retained. Such iterative revision helps mitigate construct drift and ensures that the station continues to assess the intended competencies across multiple examination cycles.</p>
<p>Maintaining a clear audit trail is an essential component of this process. Records of revisions, associated psychometric results, and the rationale for changes support transparency, accountability, and institutional learning. Finally, insights gained through this continuous improvement cycle should be integrated into faculty development activities. Sharing psychometric findings and practical lessons in training workshops can strengthen prompt design skills and enhance faculty capacity for critical review of AI generated assessment materials.</p>
<sec>
<title>Example&#x02014;Iterative refinement at NIHS</title>
<p>During an NIHS exam cycle, an AI-generated chest pain case initially showed poor discrimination (index = 0.18). Post-exam review revealed redundant checklist items and an overly generous marking scheme. Faculty revised the station, simplifying the checklist and clarifying SP cues. In the next cycle, the case achieved a discrimination index of 0.36, meeting accepted psychometric standards. This example demonstrates how psychometric monitoring can diagnose AI-related design issues and guide targeted refinement.</p>
</sec>
</sec>
<sec>
<title>Key takeaway</title>
<p>AI-assisted OSCE case generation should be treated as an iterative cycle rather than a static process. Systematic use of psychometric data and user feedback ensures continuous improvement, supporting long-term validity, reliability, and fairness.</p>
</sec>
</sec>
</sec>
<sec sec-type="discussion" id="s5">
<title>Discussion</title>
<p>This paper provides 10 evidence-informed recommendations for the responsible integration of artificial intelligence (AI) into the development of Objective Structured Clinical Examination (OSCE) stations. Together, these tips address the dual goals of efficiency and quality assurance, recognizing that AI can accelerate case generation but requires structured oversight to maintain validity, reliability, and fairness.</p>
<p>AI&#x00027;s capacity to generate diverse, structured, and customizable scenarios has the potential to reduce faculty workload and improve blueprint coverage (<xref ref-type="bibr" rid="B7">7</xref>, <xref ref-type="bibr" rid="B13">13</xref>). However, its outputs are not self-validating and may include factual inaccuracies, biases, or cultural mismatches (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B23">23</xref>). The recommendations presented here highlights how to collectively provide a coherent strategy for integrating AI tools into assessment workflows while preserving essential human oversight.</p>
<sec>
<title>Alignment with competency-based assessment</title>
<p>Tips 1&#x02013;3 highlight the importance of blueprinting, structured case design, and contextual enrichment. The broader implication is that AI enhances competency-based assessment only when its use is anchored in explicit curricular expectations and contextualized with local clinical realities. AI does not inherently ensure competency alignment; this alignment is created through deliberate educational design choices embedded within prompt structure and case specifications (<xref ref-type="bibr" rid="B21">21</xref>).</p>
</sec>
<sec>
<title>Ensuring quality and reliability</title>
<p>Tips 4&#x02013;6 focus on prompt engineering, faculty training, and validation processes, which are critical to mitigating the risks of flawed or biased outputs. The validation workflow, which combines subject matter expert (SME) review, blueprint cross-checks, peer feedback, and psychometric monitoring, parallels best practices in traditional OSCE design while adapting them to the AI era (<xref ref-type="bibr" rid="B3">3</xref>). This iterative, multi-step validation ensures stations are not only clinically accurate but also psychometrically robust.</p>
<p>This suggests a hybrid quality model in which human expertise and psychometric evidence jointly regulate the reliability of AI-assisted assessments.</p>
</sec>
<sec>
<title>Safeguarding fairness and security</title>
<p>Tips 7 and 8 emphasize equity and exam security, highlighting the need for bias audits, cultural adaptation, and the use of diverse case permutations. AI can unintentionally reproduce stereotypes or exclude minority patient profiles (<xref ref-type="bibr" rid="B25">25</xref>); deliberate diversity in AI prompts, combined with fairness audits and post-exam analysis for differential item functioning, helps safeguard equity. Generating case variations also mitigates the risks of item leakage, strengthening the integrity of high-stakes exams (<xref ref-type="bibr" rid="B27">27</xref>).</p>
<p>When these findings are interpreted widely, fairness and security appear not as discrete tasks, but as systemic safeguards required for justifiable AI usage in high-stakes assessments. AI prompts must actively counteract demographic defaults, and case modification methodologies achieve two goals: increasing authenticity while decreasing predictability and item exposure.</p>
</sec>
<sec>
<title>Enhancing efficiency and authenticity</title>
<p>Tips 9 and 10 illustrate how AI can streamline the generation of supporting materials (SP scripts, examiner checklists, multimedia resources) and facilitate continuous improvement through psychometric feedback. These innovations reduce administrative burden, improve consistency across circuits, and foster authenticity by mirroring real-world variability in clinical encounters (<xref ref-type="bibr" rid="B24">24</xref>). Beyond efficiency, an important implication is that AI generated materials, when validated, can enrich authenticity by offering dynamic variations and multimodal resources that more closely mirror real clinical environments. Moreover, integrating post exam psychometric data into AI assisted revision cycles reflects a shift toward continuous quality enhancement rather than one time case design.</p>
</sec>
<sec>
<title>Implications for future research</title>
<p>Despite early support for the feasibility of AI assisted OSCE generation (<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B18">18</xref>), findings from focus group discussions revealed notable skepticism and highlighted several critical areas requiring further investigation. Participants expressed particular concern about the reliability and psychometric defensibility of AI generated materials, reflecting uncertainty about whether such outputs can consistently meet established assessment standards.</p>
<p>One key priority identified was the need for rigorous validation of psychometric properties, including reliability and generalizability, across diverse educational and clinical contexts. Addressing these concerns is essential to demonstrate consistency of output and to strengthen evidence for content validity. Participants also emphasized the importance of examining whether AI generated case variations are effective in maintaining exam security, as predictability and item leakage were perceived as ongoing challenges despite the use of multiple case versions.</p>
<p>Further inquiry was recommended into the applicability of AI generated assessments beyond undergraduate medical education. Participants questioned whether similar approaches could be extended to interprofessional and competency based assessments across other health professions, raising issues of generalizability and contextual transferability. In addition, the long term impact of faculty development initiatives emerged as an important area for research, particularly in light of the wide variation in confidence, readiness, and risk awareness observed among faculty users.</p>
<p>Collectively, addressing these research gaps will help establish a stronger empirical foundation for institutional policy development and accreditation guidance, ensuring that the adoption of AI in assessment is aligned with principles of quality, fairness, and equity.</p>
</sec>
<sec>
<title>Limitations of the developmental workflow</title>
<p>While the developmental workflow efficiently combined experiential evidence, focus group insights, and expert consensus, these sources inevitably had limits. Experiential perspectives are frequently linked to local practices and institutional cultures, thereby limiting the applicability of findings to other educational settings. Furthermore, expert agreement is inevitably influenced by the opinions and assumptions of participating educators, notably their expectations about AI readiness or established evaluation standards. Variations in participants&#x00027; prior experience with AI technologies may possibly have altered the prioritizing of perceived difficulties or opportunities. These limits do not diminish the validity of the recommendations, but rather highlight the need for ongoing empirical validation across a wide range of programs and situations.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="s6">
<title>Conclusion</title>
<p>Artificial intelligence (AI) is reshaping clinical assessment design by offering innovative solutions to long-standing challenges in OSCE development. Evidence from recent reviews and guidance indicates growing feasibility and utility of AI tools when applied within competency-based assessment frameworks (<xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B18">18</xref>).</p>
<p>This paper provides a structured, evidence-informed framework of Ten Tips to guide safe integration of AI into high-stakes examinations. By leveraging AI for blueprinting, case generation, supporting materials, and case variations, assessment teams can increase efficiency, expand content diversity, and enhance authenticity (<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B24">24</xref>, <xref ref-type="bibr" rid="B27">27</xref>). At the same time, structured validation, faculty training, bias audits, and psychometric monitoring are essential safeguards to ensure validity, reliability, fairness, and exam security (<xref ref-type="bibr" rid="B3">3</xref>, <xref ref-type="bibr" rid="B23">23</xref>).</p>
<p>AI should serve as a complement, not a replacement, to faculty expertise. Successful implementation balances the speed and scalability of AI with human judgment, cultural sensitivity, and ethical oversight (<xref ref-type="bibr" rid="B18">18</xref>).</p>
<p>Future work should evaluate psychometric outcomes of AI-generated stations across contexts, examine security benefits of case permutations, and develop governance models that align innovation with accountability (<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B23">23</xref>). Responsible adoption offers a pathway to more efficient, equitable, and authentic assessments that better prepare learners for modern clinical practice.</p></sec>
</body>
<back>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>IZ: Conceptualization, Methodology, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. HA: Conceptualization, Writing &#x02013; review &#x00026; editing. IE: Conceptualization, Writing &#x02013; review &#x00026; editing. FS: Conceptualization, Writing &#x02013; review &#x00026; editing. SC: Writing &#x02013; review &#x00026; editing. AS: Writing &#x02013; review &#x00026; editing. MAH: Conceptualization, Writing &#x02013; review &#x00026; editing. ME: Conceptualization, Methodology, Project administration, Supervision, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The author(s) declared that this work was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declared that generative AI was used in the creation of this manuscript. During the preparation of this manuscript, the author(s) used generative artificial intelligence (AI) tools (ChatGPT, OpenAI, GPT-5, 2025 version) to assist with language refinement, formatting consistency, and structural organization. Additionally, AI was used under author supervision to generate illustrative examples of OSCE station prompts, blueprints, and checklists for demonstration purposes. All AI-generated examples were critically reviewed, verified, and edited by the author(s) to ensure medical accuracy, contextual relevance, and alignment with educational objectives. No generative AI system was used to produce or alter the conceptual framework, interpretations, or conclusions presented in the paper. All substantive ideas, recommendations, and analyses were developed by the author(s) based on their professional expertise, literature review, and institutional experience. After AI use, all content was thoroughly checked for accuracy, originality, and ethical integrity. The author(s) take full responsibility for the final content of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Frenk</surname> <given-names>J</given-names></name> <name><surname>Chen</surname> <given-names>L</given-names></name> <name><surname>Bhutta</surname> <given-names>ZA</given-names></name> <name><surname>Cohen</surname> <given-names>J</given-names></name> <name><surname>Crisp</surname> <given-names>N</given-names></name> <name><surname>Evans</surname> <given-names>T</given-names></name> <etal/></person-group>. <article-title>Health professionals for a new century: transforming education to strengthen health systems in an interdependent world</article-title>. <source>Lancet.</source> (<year>2010</year>) <volume>376</volume>:<fpage>1923</fpage>&#x02013;<lpage>58</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0140-6736(10)61854-5</pub-id><pub-id pub-id-type="pmid">21112623</pub-id></mixed-citation>
</ref>
<ref id="B2">
<label>2.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Khan</surname> <given-names>KZ</given-names></name> <name><surname>Gaunt</surname> <given-names>K</given-names></name> <name><surname>Ramachandran</surname> <given-names>S</given-names></name> <name><surname>Pushkar</surname> <given-names>P</given-names></name></person-group>. <article-title>The objective structured clinical examination (OSCE): AMEE guide no</article-title>. 81. Part II: organisation &#x00026; administration. <source>Med Teach.</source> (<year>2013</year>) <volume>35</volume>:<fpage>e1447</fpage>&#x02013;<lpage>63</lpage>. doi: <pub-id pub-id-type="doi">10.3109/0142159X.2013.818635</pub-id></mixed-citation>
</ref>
<ref id="B3">
<label>3.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pell</surname> <given-names>G</given-names></name> <name><surname>Fuller</surname> <given-names>R</given-names></name> <name><surname>Homer</surname> <given-names>M</given-names></name> <name><surname>Roberts</surname> <given-names>T</given-names></name></person-group>. <article-title>How to measure the quality of the OSCE: a review of metrics &#x02013; AMEE guide no</article-title>. 49. <source>Med Teach.</source> <volume>32</volume>, <fpage>802</fpage>&#x02013;<lpage>811</lpage>. doi: <pub-id pub-id-type="doi">10.3109/0142159X.2010.507716</pub-id><pub-id pub-id-type="pmid">20854155</pub-id></mixed-citation>
</ref>
<ref id="B4">
<label>4.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nyangeni</surname> <given-names>T</given-names></name> <name><surname>ten Ham-Baloyi</surname> <given-names>W</given-names></name> <name><surname>van Rooyen</surname> <given-names>DRM</given-names></name></person-group>. <article-title>Strengthening the planning and design of objective structured clinical examinations</article-title>. <source>Health SA</source>. (<year>2024</year>) <volume>29</volume>:<fpage>2693</fpage>. doi: <pub-id pub-id-type="doi">10.4102/hsag.v29i0.2693</pub-id><pub-id pub-id-type="pmid">39229317</pub-id></mixed-citation>
</ref>
<ref id="B5">
<label>5.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Berbenyuk</surname> <given-names>A</given-names></name> <name><surname>Powell</surname> <given-names>L</given-names></name> <name><surname>Zary</surname> <given-names>N</given-names></name></person-group>. <article-title>Feasibility and educational value of clinical cases generated using large language models</article-title>. <source>Stud Health Technol Inform.</source> (<year>2024</year>) <volume>316</volume>:<fpage>1524</fpage>&#x02013;<lpage>8</lpage>. doi: <pub-id pub-id-type="doi">10.3233/SHTI240705</pub-id><pub-id pub-id-type="pmid">39176494</pub-id></mixed-citation>
</ref>
<ref id="B6">
<label>6.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Khakpaki</surname> <given-names>A</given-names></name></person-group>. <article-title>Advancements in artificial intelligence transforming medical education: a comprehensive overview</article-title>. <source>Med Educ Online.</source> (<year>2025</year>) <volume>30</volume>:<fpage>2542807</fpage>. doi: <pub-id pub-id-type="doi">10.1080/10872981.2025.2542807</pub-id><pub-id pub-id-type="pmid">40798935</pub-id></mixed-citation>
</ref>
<ref id="B7">
<label>7.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Misra</surname> <given-names>SM</given-names></name> <name><surname>Suresh</surname> <given-names>S</given-names></name></person-group>. <article-title>Artificial intelligence and objective structured clinical examinations: using ChatGPT to revolutionize clinical skills assessment in medical education</article-title>. <source>J Med Educ Curric Dev.</source> (<year>2024</year>) <volume>11</volume>:<fpage>23821205241263475</fpage>. doi: <pub-id pub-id-type="doi">10.1177/23821205241263475</pub-id><pub-id pub-id-type="pmid">39070287</pub-id></mixed-citation>
</ref>
<ref id="B8">
<label>8.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Taylor-Drigo</surname> <given-names>CA</given-names></name> <name><surname>Kumar</surname> <given-names>A</given-names></name></person-group>. <article-title>Strengths and limitations of using ChatGPT: a preliminary examination of generative AI in medical education</article-title>. <source>medRxiv</source>. (<year>2025</year>). doi: <pub-id pub-id-type="doi">10.1101/2025.03.12.25323842</pub-id></mixed-citation>
</ref>
<ref id="B9">
<label>9.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ali</surname> <given-names>M</given-names></name> <name><surname>Rehman</surname> <given-names>S</given-names></name> <name><surname>Cheema</surname> <given-names>E</given-names></name></person-group>. <article-title>Impact of generative AI on the academic performance and test anxiety of pharmacy students in OSCE: a randomized controlled trial</article-title>. <source>Research Square</source>. (<year>2024</year>). doi: <pub-id pub-id-type="doi">10.21203/rs.3.rs-5283600/v1</pub-id></mixed-citation>
</ref>
<ref id="B10">
<label>10.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>&#x000D6;nc&#x000FC;</surname> <given-names>S</given-names></name> <name><surname>Torun</surname> <given-names>F</given-names></name> <name><surname>&#x000DC;lk&#x000FC;</surname> <given-names>HH</given-names></name></person-group>. <article-title>AI-powered standardised patients: evaluating ChatGPT-4o&#x00027;s impact on clinical case management in intern physicians</article-title>. <source>BMC Med Educ.</source> (<year>2025</year>) <volume>25</volume>:<fpage>278</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12909-025-06877-6</pub-id><pub-id pub-id-type="pmid">39979969</pub-id></mixed-citation>
</ref>
<ref id="B11">
<label>11.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Garc&#x000ED;a-L&#x000F3;pez</surname> <given-names>IM</given-names></name> <name><surname>Trujillo-Li&#x000F1;&#x000E1;n</surname> <given-names>L</given-names></name></person-group>. <article-title>Ethical and regulatory challenges of generative AI in education: a systematic review</article-title>. <source>Front Educ.</source> (<year>2025</year>) <volume>10</volume>:<fpage>1565938</fpage>. doi: <pub-id pub-id-type="doi">10.3389/feduc.2025.1565938</pub-id></mixed-citation>
</ref>
<ref id="B12">
<label>12.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Winkler</surname> <given-names>PW</given-names></name> <name><surname>Zsidai</surname> <given-names>B</given-names></name> <name><surname>Hamrin Senorski</surname> <given-names>E</given-names></name> <name><surname>Pruneski</surname> <given-names>JA</given-names></name> <name><surname>Hirschmann</surname> <given-names>MT</given-names></name> <name><surname>Ley</surname> <given-names>C</given-names></name> <etal/></person-group>. <article-title>A practical guide to the implementation of AI in orthopaedic research&#x02014;Part 7: risks, limitations, safety and verification of medical AI systems</article-title>. <source>J Exp Orthop.</source> (<year>2025</year>) <volume>12</volume>:<fpage>e70247</fpage>. doi: <pub-id pub-id-type="doi">10.1002/jeo2.70247</pub-id><pub-id pub-id-type="pmid">40276496</pub-id></mixed-citation>
</ref>
<ref id="B13">
<label>13.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rinc&#x000F3;n</surname> <given-names>EHH</given-names></name> <name><surname>Jimenez</surname> <given-names>D</given-names></name> <name><surname>Aguilar</surname> <given-names>LAC</given-names></name> <name><surname>Fl&#x000F3;rez</surname> <given-names>JMP</given-names></name> <name><surname>Tapia</surname> <given-names>&#x000C1;ER</given-names></name> <name><surname>Pe&#x000F1;uela</surname> <given-names>CLJ</given-names></name></person-group>. <article-title>Mapping the use of artificial intelligence in medical education: a scoping review</article-title>. <source>BMC Med Educ.</source> (<year>2025</year>) <volume>25</volume>:<fpage>526</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12909-025-07089-8</pub-id><pub-id pub-id-type="pmid">40221725</pub-id></mixed-citation>
</ref>
<ref id="B14">
<label>14.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Amiri</surname> <given-names>H</given-names></name> <name><surname>Peiravi</surname> <given-names>S</given-names></name> <name><surname>Rezazadeh Shojaee</surname> <given-names>SS</given-names></name> <name><surname>Rouhparvarzamin</surname> <given-names>M</given-names></name> <name><surname>Nateghi</surname> <given-names>MN</given-names></name> <name><surname>Etemadi</surname> <given-names>MH</given-names></name> <etal/></person-group>. <article-title>Medical, dental, and nursing students&#x00027; attitudes and knowledge towards artificial intelligence: a systematic review and meta-analysis</article-title>. <source>BMC Med Educ.</source> (<year>2024</year>). 24:412. doi: <pub-id pub-id-type="doi">10.1186/s12909-024-05406-1</pub-id><pub-id pub-id-type="pmid">38622577</pub-id></mixed-citation>
</ref>
<ref id="B15">
<label>15.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gordon</surname> <given-names>M</given-names></name> <name><surname>Daniel</surname> <given-names>M</given-names></name> <name><surname>Ajiboye</surname> <given-names>A</given-names></name> <name><surname>Uraiby</surname> <given-names>H</given-names></name> <name><surname>Xu</surname> <given-names>NY</given-names></name> <name><surname>Bartlett</surname> <given-names>R</given-names></name> <etal/></person-group>. <article-title>A scoping review of artificial intelligence in medical education: BEME Guide No</article-title>. 84. <source>Med Teach</source>. (<year>2024</year>) <volume>46</volume>:<fpage>446</fpage>&#x02013;<lpage>70</lpage>. doi: <pub-id pub-id-type="doi">10.1080/0142159X.2024.2314198</pub-id><pub-id pub-id-type="pmid">38423127</pub-id></mixed-citation>
</ref>
<ref id="B16">
<label>16.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Park</surname> <given-names>Y-J</given-names></name> <name><surname>Guo</surname> <given-names>E</given-names></name> <name><surname>Sachdeva</surname> <given-names>M</given-names></name> <name><surname>Ma</surname> <given-names>B</given-names></name> <name><surname>Mirali</surname> <given-names>S</given-names></name> <name><surname>Rankin</surname> <given-names>B</given-names></name> <etal/></person-group>. <article-title>OSCEai dermatology: augmenting dermatologic medical education with Large Language Model GPT-4</article-title>. <source>Can Med Educ J</source>. (<year>2025</year>) <volume>16</volume>:<fpage>29</fpage>&#x02013;<lpage>31</lpage>. doi: <pub-id pub-id-type="doi">10.36834/cmej.80056</pub-id></mixed-citation>
</ref>
<ref id="B17">
<label>17.</label>
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Mishra</surname> <given-names>GV</given-names></name> <name><surname>Luharia</surname> <given-names>AA</given-names></name> <name><surname>Naqvi</surname> <given-names>W</given-names></name> <name><surname>Sood</surname> <given-names>A</given-names></name></person-group>. <article-title>Artificial intelligence in OSCE: innovations and implications for medical education assessment &#x02013; a systematic review</article-title>. In: <source>2024 2nd DMIHER International Conference on Artificial Intelligence in Healthcare, Education and Industry (IDICAIEI)</source> (<publisher-loc>Wardha</publisher-loc>). (<year>2024</year>). p. <fpage>1</fpage>&#x02013;<lpage>5</lpage>. doi: <pub-id pub-id-type="doi">10.1109/IDICAIEI61867.2024.10842789</pub-id></mixed-citation>
</ref>
<ref id="B18">
<label>18.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Masters</surname> <given-names>K</given-names></name> <name><surname>MacNeil</surname> <given-names>H</given-names></name> <name><surname>Benjamin</surname> <given-names>J</given-names></name> <name><surname>Carver</surname> <given-names>T</given-names></name> <name><surname>Nemethy</surname> <given-names>K</given-names></name> <name><surname>Valanci-Aroesty</surname> <given-names>S</given-names></name> <etal/></person-group>. <article-title>Artificial intelligence in health professions education assessment: AMEE Guide No</article-title>. 178. <source>Med Teach</source>. (<year>2025</year>) <volume>47</volume>:<fpage>1410</fpage>&#x02013;<lpage>24</lpage>. doi: <pub-id pub-id-type="doi">10.1080/0142159X.2024.2445037</pub-id><pub-id pub-id-type="pmid">39787028</pub-id></mixed-citation>
</ref>
<ref id="B19">
<label>19.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Heston</surname> <given-names>TF</given-names></name></person-group>. <article-title>Prompt engineering for students of medicine and their teachers</article-title>. <source>arXiv</source> [preprint]. (<year>2023</year>). arXiv:2308.11628. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2308.11628</pub-id></mixed-citation>
</ref>
<ref id="B20">
<label>20.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zaghir</surname> <given-names>J</given-names></name> <name><surname>Naguib</surname> <given-names>M</given-names></name> <name><surname>Bjelogrlic</surname> <given-names>M</given-names></name> <name><surname>N&#x000E9;v&#x000E9;ol</surname> <given-names>A</given-names></name> <name><surname>Tannier</surname> <given-names>X</given-names></name> <name><surname>Lovis</surname> <given-names>C</given-names></name></person-group>. <article-title>Prompt engineering paradigms for medical applications: scoping review and recommendations for better practices</article-title>. <source>J Med Internet Res.</source> (<year>2024</year>) <volume>26</volume>:<fpage>e60501</fpage>. doi: <pub-id pub-id-type="doi">10.2196/60501</pub-id></mixed-citation>
</ref>
<ref id="B21">
<label>21.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Daniels</surname> <given-names>VJ</given-names></name> <name><surname>Pugh</surname> <given-names>D</given-names></name></person-group>. <article-title>Twelve tips for developing an OSCE that measures what you want</article-title>. <source>Med Teach.</source> (<year>2018</year>) <volume>40</volume>:<fpage>1208</fpage>&#x02013;<lpage>13</lpage>. doi: <pub-id pub-id-type="doi">10.1080/0142159X.2017.1390214</pub-id></mixed-citation>
</ref>
<ref id="B22">
<label>22.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>ten Cate</surname> <given-names>O</given-names></name> <name><surname>Chen</surname> <given-names>HC</given-names></name> <name><surname>Hoff</surname> <given-names>RG</given-names></name> <name><surname>Peters</surname> <given-names>H</given-names></name> <name><surname>Bok</surname> <given-names>H</given-names></name> <name><surname>van der Schaaf</surname> <given-names>M</given-names></name></person-group>. <article-title>Curriculum development for the workplace using Entrustable Professional Activities (EPAs): AMEE Guide No</article-title>. 99. <source>Med Teach.</source> (<year>2015</year>) <volume>37</volume>:<fpage>983</fpage>&#x02013;<lpage>1002</lpage>. doi: <pub-id pub-id-type="doi">10.3109/0142159X.2015.1060308</pub-id><pub-id pub-id-type="pmid">26172347</pub-id></mixed-citation>
</ref>
<ref id="B23">
<label>23.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Amann</surname> <given-names>J</given-names></name> <name><surname>Blasimme</surname> <given-names>A</given-names></name> <name><surname>Vayena</surname> <given-names>E</given-names></name> <name><surname>Frey</surname> <given-names>D</given-names></name> <name><surname>Madai</surname> <given-names>VI</given-names></name></person-group>. <article-title>Explainability for artificial intelligence in healthcare: a multidisciplinary perspective</article-title>. <source>BMC Med Inform Decis Mak.</source> (<year>2020</year>) <volume>20</volume>:<fpage>310</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12911-020-01332-6</pub-id><pub-id pub-id-type="pmid">33256715</pub-id></mixed-citation>
</ref>
<ref id="B24">
<label>24.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hodges</surname> <given-names>B</given-names></name> <name><surname>Regehr</surname> <given-names>G</given-names></name> <name><surname>McNaughton</surname> <given-names>N</given-names></name> <name><surname>Tiberius</surname> <given-names>R</given-names></name> <name><surname>Hanson</surname> <given-names>M</given-names></name></person-group>. <article-title>OSCE checklists do not capture increasing levels of expertise</article-title>. <source>Acad Med.</source> (<year>1999</year>) <volume>74</volume>:<fpage>1129</fpage>&#x02013;<lpage>34</lpage>. doi: <pub-id pub-id-type="doi">10.1097/00001888-199910000-00017</pub-id><pub-id pub-id-type="pmid">10536636</pub-id></mixed-citation>
</ref>
<ref id="B25">
<label>25.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pupic</surname> <given-names>N</given-names></name> <name><surname>Ghaffari-zadeh</surname> <given-names>A</given-names></name> <name><surname>Hu</surname> <given-names>R</given-names></name> <name><surname>Singla</surname> <given-names>R</given-names></name> <name><surname>Darras</surname> <given-names>K</given-names></name> <name><surname>Karwowska</surname> <given-names>A</given-names></name> <etal/></person-group>. <article-title>An evidence-based approach to artificial intelligence education for medical students: a systematic review</article-title>. <source>PLoS Digit Health</source>. (<year>2023</year>) <volume>2</volume>:<fpage>e0000255</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pdig.0000255</pub-id><pub-id pub-id-type="pmid">38011214</pub-id></mixed-citation>
</ref>
<ref id="B26">
<label>26.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Burstein</surname> <given-names>J</given-names></name> <name><surname>LaFlair</surname> <given-names>GT</given-names></name></person-group>. <article-title>Where assessment validation and responsible AI meet</article-title>. <source>arXiv [preprint].</source> (<year>2024</year>). arXiv:2411.02577. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2411.02577</pub-id></mixed-citation>
</ref>
<ref id="B27">
<label>27.</label>
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lam</surname> <given-names>G</given-names></name> <name><surname>Shammoon</surname> <given-names>Y</given-names></name> <name><surname>Coulson</surname> <given-names>A</given-names></name> <name><surname>Lalloo</surname> <given-names>F</given-names></name> <name><surname>Maini</surname> <given-names>A</given-names></name> <name><surname>Amin</surname> <given-names>A</given-names></name> <etal/></person-group>. <article-title>Utility of large language models for creating clinical assessment items</article-title>. <source>Med Teach.</source> (<year>2025</year>) <volume>47</volume>:<fpage>878</fpage>&#x02013;<lpage>82</lpage>. doi: <pub-id pub-id-type="doi">10.1080/0142159X.2024.2382860</pub-id><pub-id pub-id-type="pmid">39186054</pub-id></mixed-citation>
</ref>
</ref-list>
<fn-group>
<fn fn-type="custom" custom-type="edited-by" id="fn0001">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/534138/overview">Thiago C. Moulin</ext-link>, Uppsala University, Sweden</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by" id="fn0002">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1753189/overview">Santosh Chokkakula</ext-link>, Chungbuk National University, Republic of Korea</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2216861/overview">Alexandre Hudon</ext-link>, Montreal University, Canada</p>
</fn>
</fn-group>
</back>
</article>