<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="methods-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1644084</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Privacy-, linguistic-, and information-preserving synthesis of clinical documentation through generative agents</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>van Velzen</surname>
<given-names>Mark</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn0013"><sup>&#x2020;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2868846/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author" corresp="yes" equal-contrib="yes">
<name>
<surname>van der Willigen</surname>
<given-names>Robert F.</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<xref ref-type="author-notes" rid="fn0013"><sup>&#x2020;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2947462/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>de Beer</surname>
<given-names>Vincent J.</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3074846/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>de Graaf-Waar</surname>
<given-names>Helen I.</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Janssen</surname>
<given-names>Esther R. C.</given-names>
</name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>van Leeuwen</surname>
<given-names>Sjemaine</given-names>
</name>
<xref ref-type="aff" rid="aff8"><sup>8</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>van der Willigen</surname>
<given-names>Micha F.</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>van der Willigen</surname>
<given-names>Martijn J.</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Renardus</surname>
<given-names>Gavin</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>El Maaroufi</surname>
<given-names>Rayan</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Satimin</surname>
<given-names>Sven J.</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hartog</surname>
<given-names>Larissa M.</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hulsen</surname>
<given-names>Tim</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff9"><sup>9</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>van Meeteren</surname>
<given-names>Nico L. U.</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff10"><sup>10</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Scheper</surname>
<given-names>Mark C.</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff11"><sup>11</sup></xref>
<xref ref-type="aff" rid="aff12"><sup>12</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2569733/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Data Supported Healthcare: Data-Science Unit, Research Center Innovations in Care, Rotterdam University of Applied Sciences</institution>, <addr-line>Rotterdam</addr-line>, <country>Netherlands</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Anesthesiology and Department of Cariothoracic Surgery, Erasmus Medical Center</institution>, <addr-line>Rotterdam</addr-line>, <country>Netherlands</country></aff>
<aff id="aff3"><sup>3</sup><institution>HR Datalab EAS, School of Engineering and Applied Science, Rotterdam University of Applied Sciences</institution>, <addr-line>Rotterdam</addr-line>, <country>Netherlands</country>,</aff>
<aff id="aff4"><sup>4</sup><institution>School of Communication, Media and Information Technology, Rotterdam University of Applied Sciences</institution>, <addr-line>Rotterdam</addr-line>, <country>Netherlands</country></aff>
<aff id="aff5"><sup>5</sup><institution>Department of Orthopedic Surgery, VieCuri Medical Centre</institution>, <addr-line>Venlo</addr-line>, <country>Netherlands</country></aff>
<aff id="aff6"><sup>6</sup><institution>Radboud Institute for Health Sciences, IQ Health, Radboud University Medical Center</institution>, <addr-line>Nijmegen</addr-line>, <country>Netherlands</country></aff>
<aff id="aff7"><sup>7</sup><institution>School of Allied Health, HAN University of Applied Sciences</institution>, <addr-line>Nijmegen</addr-line>, <country>Netherlands</country></aff>
<aff id="aff8"><sup>8</sup><institution>Medifit Bewegingscentrum</institution>, <addr-line>Oss</addr-line>, <country>Netherlands</country></aff>
<aff id="aff9"><sup>9</sup><institution>Data Science &#x0026; AI Engineering, Philips</institution>, <addr-line>Eindhoven</addr-line>, <country>Netherlands</country></aff>
<aff id="aff10"><sup>10</sup><institution>Top Sector Life Sciences and Health (Health~Holland)</institution>, <addr-line>The Hague</addr-line>, <country>Netherlands</country></aff>
<aff id="aff11"><sup>11</sup><institution>Allied Health Professions, Faculty of Medicine and Science, Macquarrie University</institution>, <addr-line>Sydney, NSW</addr-line>, <country>Australia</country></aff>
<aff id="aff12"><sup>12</sup><institution>Solid Start Coalition, Erasmus Medical Center</institution>, <addr-line>Rotterdam</addr-line>, <country>Netherlands</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0014">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/59759/overview">Tuan D. Pham</ext-link>, Queen Mary University of London, United Kingdom</p>
</fn>
<fn fn-type="edited-by" id="fn0015">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1315523/overview">Ricardo Valentim</ext-link>, Federal University of Rio Grande do Norte, Brazil</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2962798/overview">Mahesh Kumar Goyal</ext-link>, Google (United States), United States</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Robert F. van der Willigen, <email>r.f.van.der.willigen@hr.nl</email></corresp>
<fn fn-type="equal" id="fn0013"><p><sup>&#x2020;</sup>These authors have contributed equally to this work and share first authorship</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1644084</elocation-id>
<history>
<date date-type="received">
<day>09</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>25</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 van Velzen, van der Willigen, de Beer, de Graaf-Waar, Janssen, van Leeuwen, van der Willigen, van der Willigen, Renardus, El Maaroufi, Satimin, Hartog, Hulsen, van Meeteren and Scheper.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>van Velzen, van der Willigen, de Beer, de Graaf-Waar, Janssen, van Leeuwen, van der Willigen, van der Willigen, Renardus, El Maaroufi, Satimin, Hartog, Hulsen, van Meeteren and Scheper</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The widespread adoption of generative agents (GAs) is reshaping the healthcare landscape. Nonetheless, broad utilization is impeded by restricted access to high-quality, interoperable clinical documentation from electronic health records (EHRs) due to persistent legal, ethical, and technical barriers. Synthetic health data generation (SHDG), leveraging pre-trained large language models (LLMs) instantiated as GAs, could offer a practical solution by creating synthetic patient information that mimics genuine EHRs. The use of LLMs, however, is not without issues; significant concerns remain regarding privacy, potential bias propagation, the risk of generating inaccurate or misleading content, and the lack of transparency in how these models make decisions. We therefore propose a privacy-, linguistic-, and information-preserving SHDG protocol that employs multiple context-aware, role-specific GAs. Guided by targeted prompting and authentic EHRs&#x2014;serving as structural and linguistic templates&#x2014;role-specific GAs can, in principle, operate collaboratively through multi-turn interactions. We theorized that utilizing GAs in this fashion permits LLMs not only to produce synthetic EHRs that are accurate, consistent, and contextually appropriate, but also to expose the underlying decision-making process. To test this hypothesis, we developed a no-code GA-driven SHDG workflow as a proof of concept, which was implemented within a predefined, multi-layered data science infrastructure (DSI) stack&#x2014;an integrated ensemble of software and hardware designed to support rapid prototyping and deployment. The DSI stack streamlines implementation for healthcare professionals, improving accessibility, usability, and cybersecurity. To deploy and validate GA-assisted workflows, we implemented a fully automated SHDG evaluation framework&#x2014;co-developed with GenAI technology&#x2014;which holistically compares the informational and linguistic features of synthetic, anonymized, and real EHRs at both the document and corpus levels. Our findings highlight that SHDG implemented through GAs offers a scalable, transparent, and reproducible methodology for unlocking the potential of clinical documentation to drive innovation, accelerate research, and advance the development of learning health systems. The source code, synthetic datasets, toolchains and prompts created for this study can be accessed at the GitHub repository: <uri xlink:href="https://github.com/HR-DataLab-Healthcare/RESEARCH_SUPPORT/tree/main/PROJECTS/Generative_Agent_based_Data-Synthesis">https://github.com/HR-DataLab-Healthcare/RESEARCH_SUPPORT/tree/main/PROJECTS/Generative_Agent_based_Data-Synthesis</uri>.</p>
</abstract>
<kwd-group>
<kwd>healthcare</kwd>
<kwd>data synthesis</kwd>
<kwd>privacy</kwd>
<kwd>generative agents</kwd>
<kwd>linguistics</kwd>
<kwd>information theory</kwd>
<kwd>synthetic health data generation (SHDG)</kwd>
<kwd>clinical natural language processing (NLP)</kwd>
</kwd-group>
<counts>
<fig-count count="4"/>
<table-count count="6"/>
<equation-count count="0"/>
<ref-count count="126"/>
<page-count count="23"/>
<word-count count="18370"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Medicine and Public Health</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>While the use of GenAI in healthcare offers substantial promise and is expected to become an integral part of regular clinical practice, its widespread adoption is limited by several critical challenges (<xref ref-type="bibr" rid="ref93">Sai et al., 2024</xref>). These include fragmented and often inaccessible data silos, significant variability in the quality of real-world data, and complex ethical and legal considerations. Efforts to address these barriers are underway, including the implementation of federated learning approaches, the establishment of robust data governance frameworks that incorporate standards such as HL7 FHIR and FAIR principles, and the development of evolving regulatory measures&#x2014;such as the European Union AI Act and the General Data Protection Regulation. Collectively, these initiatives aim to foster a landscape of responsible and ethical innovation in healthcare AI (<xref ref-type="bibr" rid="ref72">Mons et al., 2017</xref>; <xref ref-type="bibr" rid="ref13">Busch et al., 2024</xref>; <xref ref-type="bibr" rid="ref118">Woisetschl&#x00E4;ger et al., 2024</xref>; <xref ref-type="bibr" rid="ref50">Ibrahim et al., 2025</xref>; <xref ref-type="bibr" rid="ref65">Liu et al., 2025</xref>). For readers seeking concrete real-world clinical practice use cases of GenAI in healthcare information systems, see <xref ref-type="bibr" rid="ref93">Sai et al. (2024)</xref> and <xref ref-type="bibr" rid="ref89">Reddy (2024)</xref>.</p>
<p>Real-world data as recorded in EHRs consists primarily of free-text narratives that often contain valuable clinical insights that are not captured by structured data formats (<xref ref-type="bibr" rid="ref75">Negro-Calduch et al., 2021</xref>; <xref ref-type="bibr" rid="ref98">Seinen et al., 2024</xref>). However, the synthesis of meaningful health-related text has received little attention in GenAI literature (<xref ref-type="bibr" rid="ref74">Murtaza et al., 2023</xref>; <xref ref-type="bibr" rid="ref50">Ibrahim et al., 2025</xref>; <xref ref-type="bibr" rid="ref66">Loni et al., 2025</xref>; <xref ref-type="bibr" rid="ref92">Rujas et al., 2025</xref>). This represents a missed opportunity, not only in terms of data synthesis methodology, but also because free-text narratives often contain hidden and nuanced clinical information that is essential for meaningful patient understanding, accelerated early diagnosis, and improved clinical decision making (<xref ref-type="bibr" rid="ref75">Negro-Calduch et al., 2021</xref>; <xref ref-type="bibr" rid="ref98">Seinen et al., 2024</xref>; <xref ref-type="bibr" rid="ref97">Schut et al., 2025</xref>). Moreover, while methods for synthesizing structured EHR data are well established, the generation of synthetic free text from real-world EHRs remains a comparatively underdeveloped area (<xref ref-type="bibr" rid="ref74">Murtaza et al., 2023</xref>; <xref ref-type="bibr" rid="ref50">Ibrahim et al., 2025</xref>; <xref ref-type="bibr" rid="ref92">Rujas et al., 2025</xref>).</p>
<p>High quality synthetic health data generation (SHDG) that emulates real-world clinical documents is emerging as a vital strategy to enable safe, scalable, and privacy-preserving GenAI implementation in health and care settings (<xref ref-type="bibr" rid="ref103">Smolyak et al., 2024</xref>; <xref ref-type="bibr" rid="ref66">Loni et al., 2025</xref>). To be precise, SHDG is a privacy-enhancing technology that entails the generation of synthetic data based on real-world datasets. These synthetic datasets are designed to retain the essential statistical patterns and relationships of the original data, without containing any directly identifiable information. The goal is to enable analyses on synthetic data that produce results closely mirroring those obtained from the real data (<xref ref-type="bibr" rid="ref74">Murtaza et al., 2023</xref>; <xref ref-type="bibr" rid="ref28">Drechsler and Haensch, 2024</xref>; <xref ref-type="bibr" rid="ref50">Ibrahim et al., 2025</xref>; <xref ref-type="bibr" rid="ref92">Rujas et al., 2025</xref>). Ideally fully interoperable and machine-readable, synthetic datasets provide a foundation for the development, testing, and validation of innovative applications ranging from personalized healthcare models to solutions that alleviate administrative workload (<xref ref-type="bibr" rid="ref50">Ibrahim et al., 2025</xref>; <xref ref-type="bibr" rid="ref92">Rujas et al., 2025</xref>).</p>
<p>The growing promise and pitfalls of SHDG is best understood in the context of major breakthroughs in natural language processing (NLP) and GenAI, spanning early statistical approaches to today&#x2019;s transformer-based language models. Synthetic data generation gained momentum in the early 1990s exemplified by Rubin&#x2019;s (<xref ref-type="bibr" rid="ref90">Rubin, 1993</xref>) multiple imputation framework and Little&#x2019;s efforts on statistical disclosure through data masking (<xref ref-type="bibr" rid="ref64">Little, 1993</xref>). Since 2010 the adoption of machine learning and deep learning has expanded the applications of synthetic data, especially in healthcare (<xref ref-type="bibr" rid="ref30">Eigenschink et al., 2023</xref>; <xref ref-type="bibr" rid="ref74">Murtaza et al., 2023</xref>; <xref ref-type="bibr" rid="ref28">Drechsler and Haensch, 2024</xref>; <xref ref-type="bibr" rid="ref36">Goyal and Mahmoud, 2024</xref>; <xref ref-type="bibr" rid="ref83">Pezoulas et al., 2024</xref>). Deep learning models can automatically extract and represent intricate patterns from large datasets, thereby facilitating more accurate and efficient processing of information. Early approaches used models called convolutional and recurrent neural networks, which allowed machines to automatically process unstructured datasets such as images and speech without the need for feature analysis. However, deep learning model development is labor-intensive and complex (<xref ref-type="bibr" rid="ref61">LeCun et al., 2015</xref>; <xref ref-type="bibr" rid="ref35">Goodfellow et al., 2018</xref>).</p>
<p>The use of Generative Adversarial Networks (GANs) has led to significant advances in data synthesis and can be classified as both deep learning and GenAI approaches. GANs rely on adversarial training, in which two deep neural networks&#x2014;the generator and the discriminator&#x2014;compete with one another until the discriminator can no longer distinguish between real and synthetic data, reaching equilibrium (<xref ref-type="bibr" rid="ref35">Goodfellow et al., 2018</xref>; <xref ref-type="bibr" rid="ref7">Baowaly et al., 2019</xref>; <xref ref-type="bibr" rid="ref50">Ibrahim et al., 2025</xref>). It is the most widely adopted AI technology in the fields of health and healthcare for generating synthetic data. GAN models were originally developed to produce realistic pictures&#x2014;such as images from magnetic resonance imaging (MRI) or dermatoscopy&#x2014;as well as to create continuous data. Continuous data includes numerical values that can take on any value within a range, like blood pressure, glucose levels, or patient age. In addition, specialized GAN models have also been used to generate time-series data, such as signals from electrocardiograms (ECG) and electroencephalograms (EEG) (<xref ref-type="bibr" rid="ref50">Ibrahim et al., 2025</xref>).</p>
<p>The advent of transformer-based architectures revolutionized NLP by significantly enhancing language understanding (<xref ref-type="bibr" rid="ref115">Vaswani et al., 2017</xref>). Groundbreaking pre-trained language models such as &#x201C;bidirectional encoder representations from transformer&#x201D; (BERT) laid the groundwork for this transition by introducing mechanisms for capturing nuanced contextual relationships within text. Building upon this foundation, domain-specific adaptations like BioBERT and Clinical-BERT, trained on biomedical literature and clinical notes, respectively, substantially increased the fidelity with which deep learning could represent the intricate semantics of medical language through vectorization and embeddings of text into numerical expressions (<xref ref-type="bibr" rid="ref5">Alsentzer et al., 2019</xref>; <xref ref-type="bibr" rid="ref62">Lee et al., 2020</xref>; <xref ref-type="bibr" rid="ref11">Bommasani et al., 2021</xref>; <xref ref-type="bibr" rid="ref50">Ibrahim et al., 2025</xref>). Capturing clinically meaningful concepts through tokenization, vectorization, and text embedding, however, remains a formidable challenge. For example, clinical notes contain valuable contextual information but are characterized by a variety of nomenclatures, abbreviations, misspellings, and synonyms both within and across healthcare disciplines (<xref ref-type="bibr" rid="ref14">Cannon and Lucci, 2010</xref>; <xref ref-type="bibr" rid="ref27">Doan et al., 2014</xref>; <xref ref-type="bibr" rid="ref71">Meystre et al., 2017</xref>; <xref ref-type="bibr" rid="ref75">Negro-Calduch et al., 2021</xref>; <xref ref-type="bibr" rid="ref98">Seinen et al., 2024</xref>; <xref ref-type="bibr" rid="ref97">Schut et al., 2025</xref>).</p>
<p>The evolution of transformer-based architectures culminated in the development of Large Language Models (LLMs) that are not only capable of language understanding&#x2014;like BERT&#x2014;but are also capable of language generation&#x2014;like OpenAI&#x2019;s GPT3&#x2014;, called foundational models (<xref ref-type="bibr" rid="ref11">Bommasani et al., 2021</xref>). LLMs further amplified generative and NLP capabilities&#x2014;made possible by scaling up model architectures, leveraging enhanced computational power, and training on extremely large datasets. The emergence of novel functionality is associated with the so-called &#x201C;scaling effect&#x201D;&#x2014;a phenomenon that was initially unforeseen and, to this day, remains inadequately understood (<xref ref-type="bibr" rid="ref54">Kaplan et al., 2020</xref>; <xref ref-type="bibr" rid="ref22">Coveney and Succi, 2025</xref>). Recent comprehensive surveys underscore the complexity and ongoing debate surrounding the generality of scaling laws, highlighting both their impressive predictive successes and significant limitations&#x2014;especially when LLMs are applied in situations quite different from the functionality they were originally trained for, or when they face new types of data they have not seen before (<xref ref-type="bibr" rid="ref63">Li et al., 2023</xref>; <xref ref-type="bibr" rid="ref99">Sengupta et al., 2025</xref>). One of the most striking emergent properties of LLMs is their ability to respond to prompting&#x2014;where users provide specific instructions or examples in natural language to guide the model&#x2019;s output. This phenomenon allows for highly flexible and tailored use of LLMs without the need for additional fine-tuning, thereby broadening their practical utility in clinical document generation and comprehension. For example, prompting allows for zero-shot and few-shot learning, wherein LLMs can generate appropriate responses to tasks or questions with either no examples (zero-shot) or just a handful of provided examples (few-shot), greatly enhancing their versatility to synthesize realistic clinical narratives (<xref ref-type="bibr" rid="ref12">Brown et al., 2020</xref>).</p>
<p>The rapid development of efficient, fine-tunable Small Language Models (SLMs) and multimodal SLMs is rapidly transforming healthcare. Unlike their bigger LLM counterparts SLMs can operate on-premises servers supporting local deployment, which significantly reduces costs, carbon footprint, and privacy concerns. These advances&#x2014;as well as techniques like quantization, which reduce model size and accelerate inference time&#x2014;are making it feasible to bring powerful language understanding and generation capabilities directly to the point of care, enabling clinical decision support, documentation, and patient interaction without reliance on high-end cloud infrastructure (<xref ref-type="bibr" rid="ref95">Schick and Sch&#x00FC;tze, 2021</xref>; <xref ref-type="bibr" rid="ref26">Dibia et al., 2024</xref>; <xref ref-type="bibr" rid="ref34">Garg et al., 2025</xref>; <xref ref-type="bibr" rid="ref56">Kim et al., 2025</xref>; <xref ref-type="bibr" rid="ref121">Xie et al., 2025</xref>).</p>
<p>However, as GenAI becomes increasingly integrated into clinical workflows, it faces unique challenges specific to the medical domain. One of the most persistent and technically demanding issues is the precise segmentation and representation of multi-word clinical terms&#x2014;such as &#x201C;low blood pressure&#x201D; or &#x201C;low back pain&#x201D; as well as their common abbreviations like &#x201C;LBP.&#x201D; Language models, including both LLMs and SLMs, often struggle to consistently recognize and encode such terms as unified clinical concepts. Inconsistent tokenization or lack of domain-specific context can lead to fragmented or distorted semantic representations, which in turn may compromise the accuracy of clinical information extraction, decision support, or narrative synthesis. This underscores the ongoing need for the development of more sophisticated prompting strategies tailored for healthcare, and the integration of concept embeddings, clinical practice guidelines, and real-world sample data to improve the realism and utility of synthetic clinical data (<xref ref-type="bibr" rid="ref9">Beam et al., 2020</xref>; <xref ref-type="bibr" rid="ref21">Chung et al., 2023</xref>; <xref ref-type="bibr" rid="ref40">Han et al., 2023</xref>; <xref ref-type="bibr" rid="ref63">Li et al., 2023</xref>).</p>
</sec>
<sec id="sec2">
<label>2</label>
<title>GA-assisted SHDG workflows</title>
<p>To overcome the challenges of semantic fragmentation and context loss, recent research has shifted toward collaborative, multi-agent workflows to jointly tackle the complexity of clinical narratives and multimodal data. This new class of GenAI&#x2014;known as Agentic AI or Generative Agents (GAs)&#x2014;features multiple autonomous agents, each specializing in a particular task or sensory domain such as vision, language, audio, or touch (<xref ref-type="bibr" rid="ref16">Chan et al., 2023</xref>; <xref ref-type="bibr" rid="ref81">Park et al., 2023</xref>). That is, each agent can preside over its own dedicated LLM, prompted to address the unique challenges of its modality and dedicated task. By collaborating iteratively and sharing insights across these different modalities, GAs are able to process and integrate information from multiple sources&#x2014;text, images, speech&#x2014;generating content that is coherent and contextually rich (<xref ref-type="bibr" rid="ref87">Qiu et al., 2024</xref>; <xref ref-type="bibr" rid="ref37">Gridach et al., 2025</xref>; <xref ref-type="bibr" rid="ref46">Hettiarachchi, 2025</xref>; <xref ref-type="bibr" rid="ref84">Piccialli et al., 2025</xref>; <xref ref-type="bibr" rid="ref96">Schneider, 2025</xref>).</p>
<p>Unlike traditional GenAI, GAs can gather real-time data from various sources, use different tools, design custom workflows, and refine their strategies through feedback, making them highly flexible and context aware. This collaborative approach allows specialized agents to solve complex tasks&#x2014;such as SHDG&#x2014;within a single workflow (<xref ref-type="bibr" rid="ref87">Qiu et al., 2024</xref>). Some advanced workflows also incorporate language-agnostic concept models, which can further enhance the quality and performance of data synthesis (<xref ref-type="bibr" rid="ref24">Daull et al., 2023</xref>; <xref ref-type="bibr" rid="ref8">Barrault et al., 2024</xref>; <xref ref-type="bibr" rid="ref122">Xie et al., 2024</xref>; <xref ref-type="bibr" rid="ref53">Jin et al., 2025</xref>).</p>
<p>Building on this flexible and context-aware foundation, role-specific GAs are able to distribute cognitive workloads and leverage the strengths of multiple specialized agents within a single workflow (<xref ref-type="bibr" rid="ref122">Xie et al., 2024</xref>; <xref ref-type="bibr" rid="ref53">Jin et al., 2025</xref>). This not only allows for more effective problem solving but also enables the application of domain-specific expertise and enhances system robustness across a range of complex real-world tasks. For example, by leveraging advanced language models such as GPT-4, GAs can support workflows in scientific literature review writing (<xref ref-type="bibr" rid="ref1">Abdurahman et al., 2025</xref>), sentiment analysis (<xref ref-type="bibr" rid="ref114">Vasireddy et al., 2024</xref>), and the identification of mental health symptoms such as social anxiety from clinical interview data (<xref ref-type="bibr" rid="ref78">Ohse et al., 2024</xref>). GAs also demonstrate promise in interpreting and structuring unstructured data from EHRs and clinical documentation (<xref ref-type="bibr" rid="ref123">Yang et al., 2025</xref>). In particular, GPT-4.1 can complement&#x2014;and sometimes rival&#x2014;human expertise in global health education and data analysis (<xref ref-type="bibr" rid="ref107">Thandla et al., 2024</xref>).</p>
<p>Low-code/no-code LLM platforms, such as Flowise,<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref> Langflow,<xref ref-type="fn" rid="fn0002"><sup>2</sup></xref> and AutoGen<xref ref-type="fn" rid="fn0003"><sup>3</sup></xref> (<xref ref-type="bibr" rid="ref60">Kumar, 2023</xref>), are transforming the way GA-assisted SHDG workflows could be developed (<xref ref-type="bibr" rid="ref52">Jeong, 2025</xref>). These platforms use drag-and-drop interfaces and natural language prompts, making it much easier to create, modify, and optimize workflows that leverage LLMs without the requirement of in-depth coding expertise. As a result, healthcare professionals and other non-technical users can now play an active role in designing and improving intelligent systems relevant to their needs. Moreover, a recent development called &#x201C;Vibe-coding&#x201D;&#x2014;coined by <xref ref-type="bibr" rid="ref55">Karpathy (2025)</xref>&#x2014;illustrates this shift by enabling conversational co-development. Here, developers interact with GenAI tools&#x2014;like GitHub Copilot&#x2014;using plain language to iteratively refine and adjust code (<xref ref-type="bibr" rid="ref55">Karpathy, 2025</xref>; <xref ref-type="bibr" rid="ref67">Mayo, 2025</xref>).</p>
<p>In what follows, we present a protocolized GA-driven proof-of-concept for SHDG use cases aimed at creating novel synthetic clinical documents that emulate genuine EHRs while safeguarding patient privacy, preserving linguistic integrity, and maintaining informational accuracy under conditions of limited access to authentic EHR datasets.</p>
</sec>
<sec sec-type="materials|methods" id="sec3">
<label>3</label>
<title>Materials and methods</title>
<p>The protocol described here aims to expand the responsible use and deployment of GenAI healthcare solutions to a broad spectrum of end-users&#x2014;including those without specialized AI expertise&#x2014;by providing clear, step-by-step guidance for designing workflows that leverage GAs for SHDG. To further support open-source adoption, reproducibility and practical application of our GA-assisted SHDG workflows we provide a GitHub repository.<xref ref-type="fn" rid="fn0004"><sup>4</sup></xref></p>
<p>At the heart of our methodology is a modular data science infrastructure (DSI) Stack (<xref ref-type="fig" rid="fig1">Figure 1A</xref>), designed to organize and streamline the process of generating clinical data by breaking it down into a series of well-defined, interconnected workflows. It starts with ingesting and converting anonymized clinical notes from PDF to Markdown (FLOW01) (see text footnote 4), followed by pseudonymization to protect patient privacy (FLOW02) (see text footnote 4). Synthetic notes are then produced using LLMs, comparing standard prompting (FLOW03) (see text footnote 4) with a Generative Agent approach (FLOW03_AGENT_BASED). The pipeline ends with benchmarking (FLOW04) (see text footnote 4) using metrics for diversity, vocabulary similarity, semantic alignment, and classifier-based machine discernibility&#x2014;enabling the comparison of the informational and linguistic features of synthetic, anonymized, and real EHRs at both the document and corpus levels. Included is a hands-on guide for responsible deployment of GA-assisted SHDG-workflows,<xref ref-type="fn" rid="fn0005"><sup>5</sup></xref> implemented through open-source data science platforms&#x2014;Hugging Face Spaces and Flowise&#x2014;and powered by LLMs accessed via public cloud services such as Azure. This hybrid approach enables rapid prototyping and controlled sharing of workflows, while API key&#x2013;secured inference endpoints ensure privacy compliance. The GitHub repository also provides lines of example source code (FLOW03) (see text footnote 4) that enables implementation of a privacy-first alternative to public cloud AI services called Ollama, allowing users to run and manage large language models locally while maintaining full data control and privacy compliance.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Data science infrastructure (DSI) stack. <bold>(A)</bold> Schematic of the DSI stack, structured as modular, interoperable layers founded on key IT principles: abstraction and modularization, separation of concerns, interoperability and standardization, scalability, and resilience. Each layer&#x2014;[1] &#x2026; [8]&#x2014;fulfils a distinct function, from data storage and processing to analytics and deployment. It supports flexible, maintainable, and scalable data science pipelines. The DSI stack aligns with the Double Diamond design model (<ext-link xlink:href="https://www.designcouncil.org.uk/our-resources/the-double-diamond/" ext-link-type="uri">https://www.designcouncil.org.uk/our-resources/the-double-diamond/</ext-link>). Lower layers focus on &#x201C;Doing the right things&#x201D;&#x2014;data gathering and integration&#x2014;, while upper layers emphasize &#x201C;Doing things right&#x201D;&#x2014;curation, iteration, deployment. <bold>(B)</bold> Visualization of the roles and involvement of data scientists versus data engineers across the DSI stack. While data scientists are predominantly active in the human-oriented, upper layers of feature engineering, model development, and deployment, data engineers are primarily engaged in the machine-oriented, foundational layers involving warehousing, compute, and toolchains. This panel highlights the complementary skill sets necessary for an effective and robust data science infrastructure.</p>
</caption>
<graphic xlink:href="frai-08-1644084-g001.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Diagram comparing data science and data engineering roles. Panel A shows stages from warehousing to deployment, divided into data science and engineering tasks. Data science involves deployment, feature engineering, and model development. Data engineering includes warehousing, compute, and toolchain tasks. Coding infrastructure is central. Data science focuses on doing things right, while data engineering focuses on doing the right things. Panel B contrasts human-oriented data scientists with machine-oriented data engineers, emphasizing their differing involvement levels.</alt-text>
</graphic>
</fig>
<sec id="sec4">
<label>3.1</label>
<title>Data science infrastructure stack</title>
<p>Synthesizing and validating EHRs requires a well-defined data science infrastructure. This involves designing a robust data pipeline, a systematic sequence of processes that transforms raw data into high-quality synthetic datasets and generates actionable insights (<xref ref-type="bibr" rid="ref70">Meng, 2021</xref>; <xref ref-type="bibr" rid="ref109">Tuulos, 2022</xref>).</p>
<p>Building a pipeline requires the seamless integration of diverse hardware and software components within a pre-defined DSI stack. In this architecture, each layer&#x2014;from data ingestion to deployment&#x2014;builds upon the previous one, creating a cohesive framework. This layered approach streamlines the transformation of real-world EHR samples and clinical practice guidelines into novel, synthesized datasets. By following this structured process, patient privacy and regulatory compliance are maintained, enabling the safe and effective use of synthetic data for research, analytics, and clinical decision support (<xref ref-type="bibr" rid="ref86">Priebe et al., 2021</xref>; <xref ref-type="bibr" rid="ref109">Tuulos, 2022</xref>; <xref ref-type="bibr" rid="ref44">Hechler et al., 2023</xref>).</p>
<p>Our DSI stack&#x2014;as depicted in <xref ref-type="fig" rid="fig1">Figure 1</xref>&#x2014;is organized into eight layers: [1] Data Warehousing; [2] Compute Resources; [3] Toolchain; [4] Workflow Orchestration; [5] Software Architecture; [6] Model Development; [7] Feature Engineering; [8] Data Product Deployment. The dependency on IT-hardware progressively increases towards the bottom of the stack (<xref ref-type="bibr" rid="ref59">Krishnakumar et al., 2023</xref>). Next, we describe each of the relevant layers from the bottom up, highlighting their importance and practical application. Note that layers [6] Model Development and [7] Feature Engineering were not applicable to our specific use case and will therefore not be discussed further. Moreover, our DSI stack aligns with the double diamond design model (<xref ref-type="bibr" rid="ref58">Kochanowska et al., 2022</xref>): lower layers focus on <italic>&#x201C;Doing the right things,&#x201D;</italic> while upper layers emphasize <italic>&#x201C;Doing things right.&#x201D;</italic></p>
<sec id="sec5">
<label>3.1.1</label>
<title>Pseudonymization</title>
<p>Integrating real-world sample data into the synthesization process enhances the diversity of synthetic data, thereby improving its linguistic quality and informational characteristics to better reflect real-world EHR narratives (<xref ref-type="bibr" rid="ref21">Chung et al., 2023</xref>). Clinical narratives were collected from 13 patients undergoing treatment for low back pain at a single physiotherapy clinic within a primary care setting, spanning the duration of their therapeutic trajectories. All patients provided informed consent for the use of their EHRs to assist in the generation of synthetic data. All EHRs (N&#x202F;=&#x202F;13) were manually anonymized by deleting patient names, addresses, social security numbers, contact details, and insurance details.</p>
<p>Pseudonymization served as an essential data pre-processing step prior to warehousing (Section 3.1.2). This involved replacing the names of referring physicians and treating physiotherapists with fictive names, thereby restoring the natural structure of the EHRs. This procedure ensured that sample datasets destined for warehousing were thoroughly de-identified in accordance with Dutch and broader European privacy and regulatory standards. To achieve this, we developed a GenAI-based Named Entity Recognition (NER) workflow, customized for privacy categories, to systematically identify and replace personal identifiers in Markdown files derived from EHR sample PDF documents. Data entry fields for entities such as names, addresses, contact details, birth dates, &#x201C;burgerservicenummers&#x201D; (BSNs), insurance details, and financial data were detected and either removed or pseudonymized in compliance with privacy guidelines.</p>
<p>A Jupyter notebook executed custom Python code [GitHub Repository (see text footnote 4): FLOW01] to configure the Azure OpenAI API SDK; submit each document to Azure OpenAI&#x2019;s GPT-4.1; and apply a tailored system prompt to both identify and pseudonymize specified entities while preserving the Markdown format. Processed files were stored with their original formatting intact for subsequent analysis.</p>
</sec>
<sec id="sec6">
<label>3.1.2</label>
<title>Warehousing</title>
<p>Data warehousing serves as the foundational layer of the DSI stack, providing centralized aggregation and accessibility for static, unstructured datasets. It is crucial for storing both the generated synthetic data (output for developing and testing solutions) and the real-world sample data and clinical practice guidelines (as input knowledge bases for the synthesis process). Besides, the integration of anonymized real-world data from EHR systems, the inclusion of codebooks for labels and abbreviations, and clinical practice guidelines into data warehouses limits chance of hallucination but enhances the diversity, linguistic quality, and therefore the clinical relevance of synthetic data (<xref ref-type="bibr" rid="ref21">Chung et al., 2023</xref>; <xref ref-type="bibr" rid="ref63">Li et al., 2023</xref>).</p>
<p>Storing data in accessible formats such as markdown (MD), structured query language (SQL), comma-separated values (CSV), portable document format (PDF), JavaScript object notation (JSON), or images is integral to facilitating interoperability and data sharing within clinical environments. These widely used formats support the storage and exchange of both structured and unstructured data, making it easier for diverse health information systems to work together (<xref ref-type="bibr" rid="ref42">Hart et al., 2016</xref>). For example, JSON is used in AI-workflows because it is both human-readable and machine-readable. It is particularly helpful in health care and other fields for quickly and efficiently transferring data between systems, applications, or devices.</p>
</sec>
<sec id="sec7">
<label>3.1.3</label>
<title>Compute</title>
<p>The next DSI stack layer is compute, which refers to scalable data processing capacity or computational power. Its purpose is to manage and scale the performance of a predefined set of computational instructions&#x2014;referred to as a computational workload or task. The type of data science use case dictates its compute requirements, including the need for CPUs, GPUs, TPUs, or internal memory.</p>
<p>To assist the targeted end-users&#x2014;non-AI specialists&#x2014;, we decided to employ a hybrid compute solution, combining standard desktop computers or laptops (local computing) with powerful external computing resources available over the internet (public cloud services like Azure, AWS, or Google Cloud). The latter is essential for utilizing state-of-the-art LLMs. These models require specialized high-performance computing hardware and massive computational resources that far exceed the capabilities of standard desktops or laptops.</p>
<p>Public cloud enables scalable LLM deployment with on-demand access to high-end GPUs/TPUs, large memory, and parallel computing, surpassing local infrastructure limits. Since state-of-the-art LLMs are only accessible via cloud-based APIs (<xref ref-type="table" rid="tab1">Table 1</xref>), cloud adoption is essential for high-end performance. However, organizations with sufficient local computing power (e.g., EHR vendors, hospitals) may opt for on-premises GenAI model deployment.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Key evaluation criteria for selecting LLMs.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Consideration</th>
<th align="left" valign="top">Explanation</th>
<th align="left" valign="top">Importance for model selection</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Model architecture</td>
<td align="left" valign="top">The underlying design of the model (e.g., transformer-based, LSTM, etc.).</td>
<td align="left" valign="top">Influences model capabilities, efficiency, and suitability for specific NLP tasks.</td>
</tr>
<tr>
<td align="left" valign="top">Model size &#x0026; parameters</td>
<td align="left" valign="top">Number of parameters indicating model complexity and capacity</td>
<td align="left" valign="top">Larger models often perform better but require more computational resources; balance needed based on use case.</td>
</tr>
<tr>
<td align="left" valign="top">Inference speed &#x0026; latency</td>
<td align="left" valign="top">Time taken to generate outputs during use.</td>
<td align="left" valign="top">Critical for real-time applications and user experience; faster models enable scalable deployment.</td>
</tr>
<tr>
<td align="left" valign="top">Performance metric scores</td>
<td align="left" valign="top">Quantitative measures like accuracy, perplexity, BLEU, ROUGE on relevant benchmarks.</td>
<td align="left" valign="top">Helps objectively compare models&#x2019; language understanding and generation quality.</td>
</tr>
<tr>
<td align="left" valign="top">Context window size</td>
<td align="left" valign="top">Maximum input length (tokens) the model can process at once.</td>
<td align="left" valign="top">Larger context windows allow handling longer documents or conversations without losing coherence.</td>
</tr>
<tr>
<td align="left" valign="top">Fine-tuning &#x0026; customizability</td>
<td align="left" valign="top">Ability to adapt the model to specific domains or tasks via additional training.</td>
<td align="left" valign="top">Enables tailoring model behavior to unique organizational needs and improves task-specific performance.</td>
</tr>
<tr>
<td align="left" valign="top">Pretraining data &#x0026; knowledge cutoff</td>
<td align="left" valign="top">The scope and recency of data the model was trained on.</td>
<td align="left" valign="top">Determines how current and relevant the model&#x2019;s knowledge is.</td>
</tr>
<tr>
<td align="left" valign="top">Multimodal capabilities</td>
<td align="left" valign="top">Support for inputs beyond text, such as images or video.</td>
<td align="left" valign="top">Expands potential applications, enabling richer interactions and cross-modal understanding.</td>
</tr>
<tr>
<td align="left" valign="top">Use case alignment</td>
<td align="left" valign="top">Suitability of the model&#x2019;s strengths to the specific application or domain.</td>
<td align="left" valign="top">Ensures optimal performance and ROI by matching model capabilities with business goals.</td>
</tr>
<tr>
<td align="left" valign="top">Safety, bias &#x0026; ethical considerations</td>
<td align="left" valign="top">Mechanisms to reduce harmful, biased, or inappropriate outputs.</td>
<td align="left" valign="top">Ensures responsible AI use, compliance with regulations, and trustworthiness.</td>
</tr>
<tr>
<td align="left" valign="top">Licensing &#x0026; accessibility</td>
<td align="left" valign="top">Terms of use, availability (open source vs. proprietary), and cost implications.</td>
<td align="left" valign="top">Affects budget, deployment flexibility, and compliance with organizational policies.</td>
</tr>
<tr>
<td align="left" valign="top">Ecosystem &#x0026; integration</td>
<td align="left" valign="top">Availability of APIs, developer tools, and compatibility with existing systems.</td>
<td align="left" valign="top">Facilitates easier implementation, faster development cycles, and operational efficiency.</td>
</tr>
<tr>
<td align="left" valign="top">Enterprise readiness</td>
<td align="left" valign="top">Support for scalability, data privacy, user data control, and cloud provider support.</td>
<td align="left" valign="top">Important for secure, compliant, and robust deployment in production environments.</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Listed are essential considerations&#x2014;including usability, ethical implications, and enterprise requirements&#x2014;compiled together to assist non-AI specialists in selecting an LLM that aligns with their specific needs. The selected criteria are based on a synthesis of recent expert analyses and practitioner frameworks.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="sec8">
<label>3.1.4</label>
<title>Toolchain</title>
<p>The toolchain layer ensures the correct functioning of the desired workflow orchestration. Our protocol leverages rapid prototyping platforms to synthesize and validate EHRs using state-of-the-art GA-assisted SHDG workflows. These platforms offer an intuitive drag-and-drop user interface, simplifying implementation by eliminating the need for data engineering expertise in LLM deployment. This makes GenAI-technology more accessible for non-AI specialists by facilitating browser-based access (<xref ref-type="bibr" rid="ref60">Kumar, 2023</xref>; <xref ref-type="bibr" rid="ref52">Jeong, 2025</xref>).</p>
<p>We implemented a Docker-based<xref ref-type="fn" rid="fn0006"><sup>6</sup></xref> architecture to enable secure, reproducible workflows across on-premises and public cloud environments. By deploying containerized workflows via API key&#x2013;secured inference endpoints, we ensure scalable resource allocation, consistent performance, and robust access control (<xref ref-type="bibr" rid="ref2">Abhishek and Rao, 2021</xref>; <xref ref-type="bibr" rid="ref3">Ait et al., 2025</xref>). Inference endpoints offer a user-friendly interface for submitting inputs&#x2014;such as prompts for synthetic EHR generation&#x2014;and receiving outputs, allowing on-demand use of custom GA-assisted workflows without exposing users to underlying infrastructure or model complexity (<xref ref-type="bibr" rid="ref32">Fu et al., 2025</xref>; <xref ref-type="bibr" rid="ref39">Gupta, 2025</xref>).</p>
<p>Hugging Face Spaces<xref ref-type="fn" rid="fn0007"><sup>7</sup></xref> offers a public cloud platform for rapidly building, sharing, and interacting with containerized GenAI applications through intuitive interfaces, automatically hosting workflows at a public URL. This supports Docker-based deployment of low-code/no-code tools such as Flowise, Langflow and AutoGen for developing multi-agent AI workflows (<xref ref-type="bibr" rid="ref60">Kumar, 2023</xref>; <xref ref-type="bibr" rid="ref52">Jeong, 2025</xref>). Flowise enables users to visually compose, configure, and deploy LLMs without programming expertise, while Langflow allows for direct code customization. AutoGen Studio facilitates rapid prototyping and orchestration of LLM-based multi-agent systems via a low-code Python framework. For example, an implementation guide titled &#x201C;Learn how to deploy Flowise on Hugging Face&#x201D; is available online.<xref ref-type="fn" rid="fn0008"><sup>8</sup></xref></p>
<p>The usefulness of Open-source LLMs like LLaMA, Qwen, DeepSeek, and Phi in clinical settings is hampered by the frequent production of unsupported facts, contradictions, and omissions&#x2014;collectively known as hallucinations&#x2014;which present a substantial safety risk. Evaluation of the MIMIC-IV dataset revealed that while these models can accurately capture up to 83% of admission reasons and key events, their performance dropped dramatically for critical follow-up recommendations, with comprehensive coverage as low as 29% (<xref ref-type="bibr" rid="ref23">Das et al., 2025</xref>).</p>
<p>Considering both the above outlined limitations and the selection criteria presented in <xref ref-type="table" rid="tab1">Table 1</xref>, we selected OpenAI&#x2019;s GPT-4.1 (version:2025-04-14) as our preferred LLM. GPT-4.1 offers a one million token context window, which allows for comprehensive analysis of extensive patient records, and demonstrates robust instruction-following and advanced reasoning abilities (<xref ref-type="bibr" rid="ref79">OpenAI, 2025</xref>). The criteria listed in <xref ref-type="table" rid="tab1">Table 1</xref> reflect insights drawn from recent expert reviews and leading practitioner models (<xref ref-type="bibr" rid="ref88">QuantSpark, 2023</xref>; <xref ref-type="bibr" rid="ref51">Inoue, 2024</xref>; <xref ref-type="bibr" rid="ref91">Ruczynski, 2024</xref>; <xref ref-type="bibr" rid="ref20">Chojnacki, 2025</xref>; <xref ref-type="bibr" rid="ref73">Morris et al., 2025</xref>).</p>
<p>In addition, GPT-4.1 incorporates advanced domain adaptation, sophisticated fact-checking mechanisms, and alignment strategies to reduce hallucination rates and enhance faithful adherence to source texts. Here, <italic>&#x201C;Advanced domain adaptation&#x201D;</italic> refers to the ability of an LLM to adjust its understanding and generation of text to fit the specific language, conventions, and knowledge of a particular field or domain&#x2014;such as medicine (<xref ref-type="bibr" rid="ref102">Singhal et al., 2023</xref>). Moreover, GPT-4.1 appears to demonstrate advanced comprehension of medical and healthcare language, enabling more accurate interpretation of complex EHR narratives and improved detection of inconsistencies (<xref ref-type="bibr" rid="ref117">Walturn, 2025</xref>).</p>
</sec>
<sec id="sec9">
<label>3.1.5</label>
<title>Workflow orchestration</title>
<p>To synthesize Dutch clinical narratives related to low back pain in physiotherapy using genuine EHRs, we developed a GA-assisted, no-code workflow built on a rapid prototyping platform that allows end users to quickly construct and test GenAI solutions (<xref ref-type="fig" rid="fig2">Figure 2</xref>). This platform is organized into modular components&#x2014;known as modules&#x2014;each serving a specific purpose and designed to be easily rearranged or modified.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Visual representation of a no-code, multi-agent workflow for synthesizing EHRs. Data flows through connected tools and agents, enabling an iterative, structured generation process without manual coding. The here shown GA-assisted SHDG workflow begins with a Recursive Character Text Splitter that divides the uploaded PDF file containing anonymized EHR data into manageable chunks. These chunks are processed using Azure OpenAI Embeddings and stored in an In-Memory Vector Store. A Retriever Tool (RAG) then queries the stored embeddings to provide relevant context. The Azure ChatOpenAI component, configured with the GPT-4.0-mini model, interacts with stored agent memory (SQLite Agent Memory) and coordinates with two agents: the Supervisor&#x2014;acting as a senior physiotherapist specialized in low back pain&#x2014;who manages task instructions and workflow control, and directs the Tech Researcher&#x2014;acting as a practicing physiotherapist (general or specialized&#x2014;who executes prompts to generate synthetic Dutch-language EHR notes). Note: The workflow&#x2014;accessed via a web interface&#x2014;maintains contextual memory across user queries for seamless, multi-turn interactions and stops automatically when the supervisor determines completion. This design enables rapid prototyping of SHDG solutions by healthcare researchers and practitioners without requiring advanced expertise in AI. For more information on the adopted technologies and their implementation, see the Toolchain Section (Section 3.1.4 of the DSI stack).</p>
</caption>
<graphic xlink:href="frai-08-1644084-g002.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">No-code multi-agent EHR synthesizing workflow diagram showing interconnected modules. Key components include Recursive Character Text Splitter, Pdf File input, In-Memory Vector Store, Retriever Tool, Azure OpenAI Embeddings, Azure ChatOpenAI, SQLite Agent Memory, Supervisor, and Worker. Each module has specific input and output parameters, facilitating the processing and synthesis of electronic health record data.</alt-text>
</graphic>
</fig>
<p>Central to our approach is a multi-agent architecture (<xref ref-type="fig" rid="fig3">Figure 3</xref>), comprising a supervisor agent and one or more worker agents, each guided by tailored prompts that correspond to their unique responsibilities. To make the process intuitive, we utilized a &#x201C;What-IF&#x201D; scenario: imagine a clinical team with a general practitioner, a specialist, and a medical scribe collaborating to create a realistic synthetic patient note. In this analogy, the supervisor agent acts like the lead clinician&#x2014;overseeing the entire process, distributing tasks, prioritizing activities, monitoring progress, and ensuring the workflow stays on track and determines when the task is finished&#x2014;while the worker agents take on specialized roles such as drafting clinical content or formatting records, analogous to the scribe and specialist. Each agent receives role-specific guidance through prompt engineering (<xref ref-type="bibr" rid="ref18">Chen et al., 2025a</xref>), ensuring adherence to clinical standards and the accurate, coherent assembly of narratives. This modular, role-based structure not only enables seamless coordination and iterative refinement among AI agents, but also produces synthetic EHRs with high clinical validity and practical value for healthcare workers.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Example of a single-turn input/output interaction when applying the multi-agent workflow for generating realistic, structured EHRs, as described in <xref ref-type="fig" rid="fig2">Figure 2</xref>. In this scenario, the End User&#x2014;a practicing physiotherapist&#x2014;uses a web-based interface to request the creation of 20 synthetic but realistic EHRs in Dutch. The request specifies detailed content and formatting requirements, including: a concise patient history summary, a clearly stated help-seeking question, an ICF-based diagnosis, measurable treatment goals, and a treatment plan aligned with KNGF guidelines. All records must use professional Dutch clinical language with correct abbreviations. The Supervisor ensures these specifications are complete and unambiguous before the Tech Researcher produces a synthetic yet realistic Dutch-language EHR. Each record contains the requested summary, patient demographics, presenting complaint, ICF-based functional and contextual factors, SMART goals, an individualized treatment plan, and SOAP-formatted progress notes. For illustration purposes, only Patient Dossier 14 from the generated set is shown here. The Supervisor reviews this output, confirms it meets all requirements, and marks the task as finished. <italic>Color coding:</italic> Blue&#x2014;End User; Pink&#x2014;Supervisor Agent; Green&#x2014;Tech Researcher (worker Agent).</p>
</caption>
<graphic xlink:href="frai-08-1644084-g003.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Graphic displaying a structured layout for creating realistic physiotherapeutic patient files in Dutch, addressing low back pain. It includes sections on anamnesis, diagnosis following ICF, SMART therapeutic goals, and treatment plans. Text boxes offer guidelines for language use, workflow examples, and specify that 20 unique patient files are to be generated. The graphic is divided into roles: End-user, supervisor, and tech researcher, with a patient dossier example detailed. A status box at the bottom-right corner indicates "status: FINISHED."</alt-text>
</graphic>
</fig>
<sec id="sec10">
<label>3.1.5.1</label>
<title>Supervisor agent prompt</title>
<p>The supervisor agent was equipped with a <italic>&#x201C;system prompt&#x201D;</italic>&#x2014;a foundational set of instructions that defines the LLM&#x2019;s overarching persona, scope, and governance strategy. Specifically, this prompt directed the supervisor agent to represent an experienced physiotherapist overseeing the clinical perspective of a registered physiotherapist tasked with generating authentic Dutch EHRs for low back pain cases. The prompt specified how to generate clinical documentation according to the International Classification of Functioning, Disability and Health (ICF) framework domains (<xref ref-type="bibr" rid="ref119">World Health Organization, 2001</xref>), and adherence to the Dutch Royal Society for Physiotherapy (KNGF) guideline on low back pain (<xref ref-type="bibr" rid="ref106">Swart et al., 2021</xref>). This ensured that every generated record aligned with current best practices in physiotherapy documentation.</p>
<p>The prompt further instructed the supervisor agent to restrict outputs solely to the requested EHR content, thereby enforcing compliance and preventing the inclusion of extraneous or sensitive information. Functionally, it mandated the supervisor agent to orchestrate the division of labor among worker agents, manage task handoff, establish execution priorities, consolidate contributions, and verify completion of all record elements in accordance with professional documentation standards.</p>
</sec>
<sec id="sec11">
<label>3.1.5.2</label>
<title>Worker agent prompt</title>
<p>Under the coordination of the supervisor agent, each worker agent received individualized &#x201C;worker prompts&#x201D; tailored to its specialized domain within the workflow. These prompts offered detailed, task-specific instructions for generating a single EHR taking into account the clinical nuances of physiotherapy care for (sub)acute or chronic low back pain, as well as relevant and documentation standards (<xref ref-type="bibr" rid="ref29">Driehuis et al., 2019</xref>; <xref ref-type="bibr" rid="ref106">Swart et al., 2021</xref>). Most notably, the worker prompts required structured output, as listed in <xref ref-type="table" rid="tab2">Table 2</xref>.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Targeted worker agent prompting.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Consideration</th>
<th align="left" valign="top">Explanation</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Anamnesis summary</td>
<td align="left" valign="top">Craft a concise, professional account of the patient&#x2019;s medical history, the impact of symptoms, coping mechanisms, and clinical context; ensure precise specification of symptom duration (acute, subacute, or chronic) and maintain professional standards of written Dutch.</td>
</tr>
<tr>
<td align="left" valign="top">Physical therapy diagnosis</td>
<td align="left" valign="top">Deliver a comprehensive, multidimensional diagnostic formulation, detailing impairments, activity limitations, participation restrictions, relevant contextual (personal and environmental) factors, risk/prognostic indicators, and a reformulation of the patient&#x2019;s explicit help seeking question&#x2014;all mapped to ICF domains.</td>
</tr>
<tr>
<td align="left" valign="top">Treatment goals</td>
<td align="left" valign="top">Articulate SMART (Specific, Measurable, Achievable, Relevant, Time-bound), patient-centered, and function-oriented short- and long-term goals, with reference to clinical metrics (e.g., NPRS, QLBDS) strictly as criteria, not as goals themselves. Target dates for each goal are specified according to best practice.</td>
</tr>
<tr>
<td align="left" valign="top">Treatment plan</td>
<td align="left" valign="top">Compose an intervention strategy, incorporating manual therapy, exercise programs, educational components, psychosomatic physiotherapy and other modalities, substantiated by the KNGF guidelines and explicitly related to the established treatment goals.</td>
</tr>
<tr>
<td align="left" valign="top">SOEP progress notes</td>
<td align="left" valign="top">Generate between three and eight detailed progress notes, each corresponding to an individual treatment session, structured in the SOEP (Subjective, Objective, Evaluation, Plan) format. These notes are intended to reflect realistic clinical variation, including both therapeutic progression and stagnation or need for adjustment.</td>
</tr>
<tr>
<td align="left" valign="top">Language and style</td>
<td align="left" valign="top">All documentation must be rendered in idiomatic, professional Dutch with expanded abbreviations (e.g., MT, NPRS, LBP), and maintain a narrative and tone that emulates authentic Dutch physiotherapy records as demonstrated in the reference examples.</td>
</tr>
<tr>
<td align="left" valign="top">Referencing examples and output format</td>
<td align="left" valign="top">Worker Agents are instructed to use pseudonymized sample EHRs solely as a stylistic and structural reference, ensuring every generated dossier remains unique.</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Detailed are the specific domains addressed by the worker agent, along with corresponding explanations, to guide the synthesis of clinically valid and contextually appropriate physiotherapy records for low back pain.</p>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec id="sec12">
<label>3.1.6</label>
<title>Software architecture and deployment</title>
<p>Data product deployment is the final layer of our DSI stack, where prototype workflows transform into web-accessible applications. This layer ensures that data products&#x2014;such as LLM GPT-4.1 and Hugging Face Spaces&#x2014;are securely and reliably made available through Inference Endpoints secured with API key authorization (as was discussed in the Toolchain Section 3.1.4). This also protects patient privacy and complies with data management and regulatory standards, including the GDPR<xref ref-type="fn" rid="fn0009"><sup>9</sup></xref> and the EU AI Act<xref ref-type="fn" rid="fn0010"><sup>10</sup></xref> (<xref ref-type="bibr" rid="ref43">Haug, 2018</xref>; <xref ref-type="bibr" rid="ref47">Hoofnagle et al., 2019</xref>; <xref ref-type="bibr" rid="ref31">European Parliament and Council, 2024</xref>). Additionally, it manages resources and access controls to maintain data security and organization. By simplifying deployment through no-code/low-code platforms, discussed in the next section, with reusable components and clear specifications, this layer helps deliver trustworthy, easy-to-use solutions that support clinical decision-making, research, and analytics.</p>
</sec>
</sec>
<sec id="sec13">
<label>3.2</label>
<title>No-code proof-of-concept</title>
<p>This section details a rapid-prototyping implementation to demonstrate the feasibility and effectiveness of generating synthetic EHRs using GA-assisted SHDG workflows (<xref ref-type="fig" rid="fig2">Figures 2</xref>, <xref ref-type="fig" rid="fig3">3</xref>). By combining modularity and no-code principles, our proof-of-concept illustrates a practical, user-friendly approach for healthcare professionals to generate, validate, and experiment with synthetic health data. It supports iterative development, transparency, and ease of integration with other systems or data science workflows.</p>
<p>Here, we outline the specific tools and configurations used, illustrating how the DSI stack layers translate a specific data science use case into a functional workflow. The protocol presented here serves as a practical step-by-step guide for replicating our approach and showcases the capabilities of the proposed architecture in addressing the challenges of clinical text synthesis.</p>
<p>For demonstrative purposes, we utilized public Hugging Face Spaces infrastructure in combination with Flowise to facilitate deployment (see Section 3.1.4). This setup allows custom-made workflows to be shared publicly or privately, making them accessible via a web interface or API, and supports secure credential management for connecting to external LLMs and API services.</p>
<p>The workflow (<xref ref-type="fig" rid="fig2">Figure 2</xref>) to synthesize EHRs, required additional modules for document ingestion, embedding generation (using Azure API key credentials), agent memory management (using a SQLite database) and memory retrieval through a vector store. <xref ref-type="table" rid="tab3">Table 3</xref> provides a functional overview of the Flowise modules shown in <xref ref-type="fig" rid="fig3">Figure 3</xref>. To digest unstructured sample data (e.g., PDFs), we applied a recursive character text splitter. Traditionally, parameters such as chunk size and chunk overlap are critical: smaller chunks (500&#x2013;1,000 characters) improve granularity but may fragment context, while larger chunks (&#x003E;1,500 characters) support coherence but risk exceeding model context windows. Moderate overlap (100&#x2013;200 characters) helps maintain semantic continuity between segments. However, recent studies suggest that late chunking&#x2014;where segmentation occurs after the model embedding step rather than before&#x2014;can preserve global context more effectively and enhance downstream performance, particularly in retrieval-augmented tasks (<xref ref-type="bibr" rid="ref38">G&#x00FC;nther et al., 2024</xref>).</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Workflow stages, modules, and operational details for no-code GA-assisted synthetic EHR processing.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Workflow stage</th>
<th align="left" valign="top">Module</th>
<th align="left" valign="top">Operational details</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top" rowspan="4">Ingestion and preprocessing parsing</td>
<td align="left" valign="top">A/B recursive character text splitter</td>
<td align="left" valign="top"><bold>Function:</bold> Splits a large text document into semantically meaningful &#x201C;chunks&#x201D; (e.g., 1,000 characters / 200-character overlap) to meet processing constraints.<break/><bold>Input:</bold> Large text document (e.g., PDF file).<break/><bold>Output:</bold> Prepared data chunks for downstream processing within input size limits.</td>
</tr>
<tr>
<td align="left" valign="top">PDF File</td>
<td align="left" valign="top"><bold>Function:</bold> Extracts text content from uploaded PDFs.<break/><bold>Input:</bold> Text splitter output; end-user uploads a PDF file (e.g., a real-world EHR).<break/><bold>Output:</bold> Structured text format suitable for processing (one document per page).</td>
</tr>
<tr>
<td align="left" valign="top">In-memoryvector store</td>
<td align="left" valign="top"><bold>Function:</bold> Stores embeddings for fast semantic EHR searching and referencing in agent workflows.<break/><bold>Input:</bold> Embeddings (via Azure OpenAI) derived from document chunks (e.g., from PDF files).<break/><bold>Output:</bold> Embeddings stored for quick retrieval.</td>
</tr>
<tr>
<td align="left" valign="top">Retriever tool</td>
<td align="left" valign="top"><bold>Function:</bold> Queries the vector store to retrieve relevant EHR content based on prompts or keywords.<break/><bold>Input:</bold> Query from workflow (e.g., &#x201C;provide context&#x201D;).<break/><bold>Output:</bold> Most contextually relevant EHR chunks for the next workflow step.</td>
</tr>
<tr>
<td align="left" valign="top">Embeddings</td>
<td align="left" valign="top">Azure OpenAI embeddings</td>
<td align="left" valign="top"><bold>Function:</bold> Generates numerical vector representations of text chunks for similarity search and LLM processing.<break/><bold>Input:</bold> Name and credentials for the text embedder (e.g., text-embedding-3-large).<break/><bold>Output:</bold> Numerical embeddings of text chunks.</td>
</tr>
<tr>
<td align="left" valign="top">Agent memory management</td>
<td align="left" valign="top">SQLite agent memory</td>
<td align="left" valign="top"><bold>Function:</bold> Maintains multi-turn interaction memory for context continuity.<break/><bold>Input:</bold> Additional parameters (if any).<break/><bold>Output:</bold> Persistent memory of conversation history, prior actions, and current state.</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Agent orchestration of multi-turn interactions</td>
<td align="left" valign="top">Supervisor</td>
<td align="left" valign="top"><bold>Function:</bold> Central controller &#x2014; interprets tasks, routes them to workers or tools, ensures proper sequencing, and coordinates memory and moderation.<break/><bold>Input:</bold> LLM (e.g., GPT-4o-mini), agent memory, supervisor prompt/role (plus optional parameters).<break/><bold>Output:</bold> Coordinated orchestration between tools, agents, and workflow steps.</td>
</tr>
<tr>
<td align="left" valign="top">Worker</td>
<td align="left" valign="top"><bold>Function:</bold> Executes reasoning, analysis, or synthesis tasks using retrieved context and domain knowledge.<break/><bold>Input:</bold> Tools (via supervisor), worker prompt/role (plus optional parameters).<break/><bold>Output:</bold> Structured answers, summaries, or insights per prompt.</td>
</tr>
<tr>
<td align="left" valign="top">User interaction and language model reasoning</td>
<td align="left" valign="top">Azure ChatOpenAI</td>
<td align="left" valign="top"><bold>Function:</bold> Provides a conversational interface between user and workflow, leveraging memory and context for responses.<break/><bold>Input:</bold> Prompts from the user with parameters (e.g., credentials, temperature, mode name).<break/><bold>Output:</bold> User-facing responses and orchestrated agent actions.</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We employed GPT-4.1 LLMs&#x2014;using Azure API key credentials&#x2014;for supervised reasoning and text generation. A key model parameter for any LLM is temperature, which controls generative diversity: lower values (~0.2&#x2013;0.4) yield deterministic, guideline-conform output, while higher values (~0.7&#x2013;0.9) promote creative variability (<xref ref-type="bibr" rid="ref82">Peeperkorn et al., 2024</xref>). For clinical synthesis, we assumed that a temperature between 0.3&#x2013;0.5 best balances realism and consistency; however, future experiments are required to empirically evaluate model performance at varying settings. The <italic>&#x201C;agentflow&#x201D;</italic> as implemented within the Flowise no-code framework can be downloaded as a JSON file from our GitHub Repository (see text footnote 1) (GA-assisted SHDG workflow).</p>
<p>The supervisor agent represents an experienced physiotherapist who ensures that documentation conforms to ICF domains, linguistic plausibility, and overall fidelity (<xref ref-type="fig" rid="fig3">Figure 3</xref>). The worker agent (<xref ref-type="fig" rid="fig3">Figure 3</xref>) embodies one of several physiotherapy profiles (e.g., generalist, manual therapist, exercise therapist, psychosomatic physiotherapist) and is tasked with generating synthetic clinical narratives. To reduce the risk of hallucination and ensure domain-conformant output, we implemented a retrieval-augmented generation (RAG) pipeline. This supported the contextual grounding of worker agent output using a vectorized memory store filled with real-world sample data, clinical practice guidelines, and documentation standards used by Dutch physiotherapists. This architecture allows agents to generate clinically realistic output grounded in both empirical input and normative context (<xref ref-type="bibr" rid="ref21">Chung et al., 2023</xref>; <xref ref-type="bibr" rid="ref63">Li et al., 2023</xref>).</p>
</sec>
<sec id="sec14">
<label>3.3</label>
<title>SHDG automation through GenAI-assisted co-development</title>
<p>Here we describe how we automatized the entire synthetic EHR pipeline; using a novel engineering approach called: GenAI-assisted co-development. This technique fosters collaboration between human developers and GAs (<xref ref-type="bibr" rid="ref81">Park et al., 2023</xref>) allowing human domain specialists to iteratively refine code through natural language prompts and AI-tools like GitHub Copilot, Perplexity, and Gemini. Copilot helps with coding by suggesting and writing code, Perplexity helps find answers by providing clear, sourced information, and Gemini acts as an all-around smart assistant that can respond to questions, summarize text, and assist with a variety of tasks. As such, GenAI-assisted co-development fundamentally redefines the paradigm of human-AI collaboration (<xref ref-type="bibr" rid="ref80">Orru et al., 2023</xref>; <xref ref-type="bibr" rid="ref111">Ulfsnes et al., 2024</xref>; <xref ref-type="bibr" rid="ref15">Casper et al., 2025</xref>; <xref ref-type="bibr" rid="ref67">Mayo, 2025</xref>).</p>
<p>For instance, GAs like AlphaEvolve (<xref ref-type="bibr" rid="ref25">DeepMind, 2025</xref>) exemplify the power of GenAI-assisted co-development by autonomously facilitating iterative code improvement. AlphaEvolve operates as an evolutionary coding agent that leverages the orchestration of multiple LLMs, enabling ongoing refinement of algorithmic solutions through a cycle of edits and evaluator feedback. This process mirrors the collaborative workflow described earlier (Section 3.2), whereby human expertise and GAs interleave in a continuous dialogue&#x2014;using natural language guidance to steer, critique, and optimize computational problem-solving. In this way, AlphaEvolve and similar systems advance the core objective of our methodology: to solve complex scientific and computational challenges by seamlessly integrating human insight with autonomous generative capabilities (<xref ref-type="bibr" rid="ref77">Novikov et al., 2025</xref>).</p>
<sec id="sec15">
<label>3.3.1</label>
<title>GA-assisted SHDG workflow validation</title>
<p>We started by confirming that our GA-assisted SHDG workflow (Section 3.2) was effective in generating meaningful synthetic EHRs. This preliminary validation step involved human specialists (authors MV, MS) assessing the synthetic EHR samples for realism, internal coherence, and adherence to professional clinical documentation standards. We confirmed that the generated records were representative for deployment in downstream healthcare applications and research.</p>
</sec>
<sec id="sec16">
<label>3.3.2</label>
<title>PDF-to-Markdown conversion</title>
<p>As a foundational step, we used Gemini 2.5 Flash as the core LLM in our GenAI-Assisted Software Development workflow to generate and refine Python code&#x2014;contained in Jupyter notebooks&#x2014;for automating PDF-to-Markdown conversion. Guided by natural language prompts written by human domain experts, Gemini 2.5 Flash authored code that extracts text from PDFs (via Python packages such as OpenAI, PyMuPDF, and glob), interfaces with Azure OpenAI&#x2019;s GPT-4.1 for Markdown formatting, and supports batch processing of files. This collaborative approach enabled efficient, maintainable code development, combining Gemini 2.5 Flash&#x2019;s reasoning and coding abilities with GPT-4.1&#x2019;s language understanding to deliver scalable and accurate document conversion. Note, Human supervision&#x2014; following a human-in-the-loop approach, in which humans remain actively involved in reviewing, verifying, and refining AI outputs&#x2014;was essential to ensure that the AI-generated code functioned correctly. The code used is available online via our GitHub Repository (see text footnote 1) (FLOW01).</p>
</sec>
<sec id="sec17">
<label>3.3.3</label>
<title>Pseudonymization</title>
<p>We prompted Gemini 2.5 Flash to integrate a pseudonymization step using a second call. This involved a second call to the Azure OpenAI GPT-4.1 API with a tailored system prompt designed to identify and pseudonymize specified named entities while preserving the Markdown format. For details about the selected named entities, we refer to Section 3.1.1. The code used is available online via our GitHub Repository (see text footnote 1) (FLOW02).</p>
</sec>
<sec id="sec18">
<label>3.3.4</label>
<title>EHR-synthesis co-developed with GenAI technology</title>
<p>Subsequently, we instructed Gemini 2.5 Flash to generate Python code that implements the synthetic data generation process. This process used the pseudonymized Markdown files as contextual examples for GPT-4.1, guided by the natural language prompts previously specified for the supervisor agent and worker agents (see Section 3.1.5). A comprehensive explanation of the code is provided, and it is publicly available as a downloadable Jupyter notebook from our GitHub Repository (see text footnote 1) (FLOW03).</p>
</sec>
<sec id="sec19">
<label>3.3.5</label>
<title>Benchmark framework &#x0026; analysis</title>
<p>Finally, Gemini 2.5 Flash was tasked with developing a programmatic approach to assess how closely synthetic data mirrors the informational and linguistic characteristics of pseudonymized real-world EHRs. This assessment strictly adheres to an evaluation framework comprising ten distinct metrics, as detailed in the next Section 3.4, to evaluate the indistinguishability of synthetic from real data. The code used is available online via our GitHub Repository (see text footnote 1) (FLOW04).</p>
</sec>
</sec>
<sec id="sec20">
<label>3.4</label>
<title>Quantitative assessment of synthetic data quality</title>
<p>To systematically assess the fidelity of GA-assisted SHDG, we recognized that no single metric would suffice. Therefore, we selected ten distinct metrics to jointly evaluate both the informational and linguistic qualities of clinical documents on both the individual document and corpus levels. These metrics, summarized in <xref ref-type="table" rid="tab4">Tables 4</xref>, <xref ref-type="table" rid="tab5">5</xref>, were chosen to provide clear and quantifiable assessments that are informative for both AI specialists and clinical experts.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>Metrics used to evaluate surface-level similarities.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Metric</th>
<th align="left" valign="top">Category</th>
<th align="left" valign="top">Purpose</th>
<th align="left" valign="top">Interpretation (desired score)</th>
<th align="left" valign="top">Interpretation (undesired score)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Average word count</td>
<td align="left" valign="top">Structural fidelity</td>
<td align="left" valign="top">Compares word count per document between datasets.</td>
<td align="left" valign="top">Similar average word counts between synthetic and real data.</td>
<td align="left" valign="top">Consistent divergence (higher/lower) in average word count, suggesting content over/under-generation.</td>
</tr>
<tr>
<td align="left" valign="top">Average unique word count</td>
<td align="left" valign="top">Linguistic</td>
<td align="left" valign="top">Compares the diversity of vocabulary per document.</td>
<td align="left" valign="top">Similar average unique word counts, indicating comparable linguistic richness.</td>
<td align="left" valign="top">Significant differences, suggesting issues with vocabulary diversity.</td>
</tr>
<tr>
<td align="left" valign="top">Average document length (characters)</td>
<td align="left" valign="top">Structural fidelity</td>
<td align="left" valign="top">Compares overall document size in characters.</td>
<td align="left" valign="top">Similar average document lengths between synthetic and real data.</td>
<td align="left" valign="top">Consistent divergence (higher/lower) in average document length, suggesting content over/under-generation.</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Structural and linguistic metrics for evaluating similarities between original, pseudonymized, and synthetic documents, with interpretation guidelines for desirable and undesirable results.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="tab5">
<label>Table 5</label>
<caption>
<p>Metrics used to evaluate linguistic and information level similarities.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Metric</th>
<th align="left" valign="top">Category</th>
<th align="left" valign="top">Purpose</th>
<th align="left" valign="top">Interpretation (desired score)</th>
<th align="left" valign="top">Interpretation (undesired score)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Shannon&#x2019;s entropy (characters)</td>
<td align="left" valign="top">Textual diversity</td>
<td align="left" valign="top">Quantifies richness and unpredictability at character level for the entire corpus.</td>
<td align="left" valign="top">Similar entropy values to real data.</td>
<td align="left" valign="top">Much lower entropy (overly repetitive) or much higher entropy (excessively random/incoherent).</td>
</tr>
<tr>
<td align="left" valign="top">Shannon&#x2019;s entropy (words)</td>
<td align="left" valign="top">Textual diversity</td>
<td align="left" valign="top">Quantifies richness and unpredictability at word level for the entire corpus.</td>
<td align="left" valign="top">Similar entropy values to real data, indicating comparable vocabulary diversity.</td>
<td align="left" valign="top">Much lower entropy (overly repetitive vocabulary) or much higher entropy (excessively random/incoherent word choice).</td>
</tr>
<tr>
<td align="left" valign="top">Jensen-Shannon divergence</td>
<td align="left" valign="top">Word distribution similarity</td>
<td align="left" valign="top">Measures the statistical distance between word probability distributions of two corpora (range 0&#x2013;1).</td>
<td align="left" valign="top">Low JSD (closer to 0), implying similar word frequency patterns and vocabulary overlap.</td>
<td align="left" valign="top">High JSD (closer to 1), indicating marked differences in vocabulary or word usage patterns.</td>
</tr>
<tr>
<td align="left" valign="top">Average Bigram Pointwise Mutual Information(PMI)</td>
<td align="left" valign="top">Naturalness of word associations</td>
<td align="left" valign="top">Quantifies the average strength of association between adjacent words.</td>
<td align="left" valign="top">Comparable PMI, indicates synthetic text mimics natural bigram.</td>
<td align="left" valign="top">Significant differences, suggesting unnatural word pairings or phrasings.</td>
</tr>
<tr>
<td align="left" valign="top">BLEU score</td>
<td align="left" valign="top">Lexical similarity / surface-level overlap</td>
<td align="left" valign="top">Quantifies n-gram overlap between synthetic and reference texts (range 0&#x2013;100).</td>
<td align="left" valign="top">Higher BLEU score (closer to 100), indicating greater literal overlap in n-grams.</td>
<td align="left" valign="top">Lower BLEU score (e.g., 4.6), indicating very low literal overlap; suggests synthetic text does not closely replicate exact phrasing, potentially acceptable if novelty is a goal.</td>
</tr>
<tr>
<td align="left" valign="top">BERTScore</td>
<td align="left" valign="top">Semantic alignment</td>
<td align="left" valign="top">Assesses semantic similarity using contextual embeddings (F1 typically 0&#x2013;1).</td>
<td align="left" valign="top">High F1 score (closer to 1), indicating strong semantic alignment and meaning preservation.</td>
<td align="left" valign="top">Lower scores suggest synthetic data differs from the sample data. This indicates the newly generated data is not an exact replica of its origin</td>
</tr>
<tr>
<td align="left" valign="top">Classifier performance (AUC)/(AUPRC)</td>
<td align="left" valign="top">Inseparability</td>
<td align="left" valign="top">Tests how easily a classifier can distinguish real (pseudonymized) from synthetic data.<break/>Indicates the &#x201C;realism&#x201D; of synthetic data. (range 0&#x2013;1).</td>
<td align="left" valign="top">AUC/AUPRC approaches 0.5, implying classifier cannot effectively differentiate (high mimicry).<break/>Lower values are desirable for synthetic data quality.</td>
<td align="left" valign="top">AUC/AUPRC &#x2248; 1.0: Classifier easily separates classes; unrealistic synthetic data.<break/>Good indication of &#x201C;machine-discernibility.&#x201D;<break/>Limitations:<break/>Sensitive to dataset size<break/>Depends on classifier choice</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Metrics assessing overall similarity between real and synthetic clinical text corpora&#x2014;including diversity, vocabulary, semantic alignment, and machine discernibility&#x2014;emphasizing that desired scores reflect close corpus-level resemblance across these dimensions.</p>
</table-wrap-foot>
</table-wrap>
<p>Our framework goes beyond surface-level resemblance&#x2014;such as basic structure&#x2014;by also examining deeper linguistic and semantic properties. Specifically, we evaluate whether synthetic texts authentically mirror genuine EHRs in their phrasing, stylistic features, and conveyed meanings. To ensure this, we analyzed pooled corpora of pseudonymized real and synthetic data (<xref ref-type="table" rid="tab5">Table 5</xref>), assessing similarities in word sequences, style, and intent.</p>
<p>Detailed implementation notes, equations and code for each metric are available in our public GitHub Repository (see text footnote 1) (FLOW04).</p>
<sec id="sec21">
<label>3.4.1</label>
<title>Document level assessment</title>
<p>Three different metrics &#x2014;the original clinical documents, their pseudonymized counterparts, and synthetic documents generated through the GA-assisted SHDG process&#x2014;were used to provide insight into the structural fidelity of the generated documents, ensuring that basic textual characteristics were faithfully reproduced in the synthetic samples. Assessment of averaged word count, average unique word count, and average document length (measured in characters) provided a straightforward means for comparing the overall size and composition of documents between the original, pseudonymized, and synthetic datasets.</p>
<p><xref ref-type="table" rid="tab4">Table 4</xref> provides an overview of structural and linguistic metrics used to evaluate the fidelity of GA-assisted SHDG at the document level. It specifically focuses on comparing the original, pseudonymized, and synthetic documents in terms of word count, vocabulary diversity, and document length. The table offers guidance for interpreting whether the observed scores indicate that synthetic documents adequately replicate the structure and linguistic richness of the real data or reveal potential discrepancies.</p>
<p>Collectively, the document-level metrics of <xref ref-type="table" rid="tab4">Table 4</xref> allow for direct comparison of volume and informational content across the three datasets. Consistent differences in document length, word count, or vocabulary diversity between synthetic and real documents can signal problems in the data generation process, such as systematic under- or over-generation. Early detection of such discrepancies enables targeted improvements, ensuring that synthetic data more accurately reflects the completeness and verbosity found in authentic datasets.</p>
</sec>
<sec id="sec22">
<label>3.4.2</label>
<title>Corpus level assessment</title>
<p>The inherent heterogeneity in EHRs&#x2014;stemming from both the variety in patients and the diversity in documentation by healthcare professionals&#x2014;can greatly impact comparability between real and synthetic data. Furthermore, as the aim of synthetic data is often to generate more data than is originally available (thus overcoming limited data availability), metrics other than pairwise comparison of individual documents are needed to evaluate the linguistic and informational similarities between datasets. To assess the actual comparability between the real pseudonymized and the synthetic clinical text, we pooled the individual documents into two corpora.</p>
<p><xref ref-type="table" rid="tab5">Table 5</xref> provides a comprehensive set of metrics for assessing the fidelity of synthetic clinical text at the corpus level. Given the inherent variability in real-world electronic health records and the goal of creating synthetic datasets that closely resemble the originals, these metrics move beyond simple pairwise document comparisons to evaluate broader linguistic and informational features. The table outlines measures of textual diversity, vocabulary similarity, semantic alignment, and machine discernibility, each with its specific interpretation. Collectively, these metrics enable a holistic evaluation of how well the synthetic corpus replicates the complexity, nuance, and realism of the real clinical text, providing an in-depth view of corpus-level similarity across multiple dimensions.</p>
</sec>
</sec>
<sec id="sec23">
<label>3.5</label>
<title>Sample dataset description</title>
<p>The original dataset comprised N&#x202F;=&#x202F;13 EHRs in PDF format, all relating to Dutch patients suffering from lower back pain. These real-world documents featured a combination of structured and unstructured text, including clinical notes, reports, and other pertinent patient information. To characterize these EHRs in terms of linguistic quality and informational content, a custom Python script was developed through GenAI-assisted co-development (see Section 3.3 for a detailed description).</p>
<p>For each file, various parameters were extracted and calculated, including <italic>storage size</italic> (in MB), <italic>textual content size</italic> (measured as total words, unique words after tokenization and lowercasing, and total characters as a measure of document length), and the primary language detected within the textual content. <italic>Structural elements</italic> such as the presence and count of tables, figures (images), and annotations were also identified. Additionally, <italic>Shannon entropy</italic> was computed at both the character and word level to quantify the average uncertainty or randomness in the text, thereby providing insight into its complexity and predictability. The <italic>canonical Jensen-Shannon Divergence (JSD)</italic> was calculated to compare the word distribution of each document with the overall word distribution across the dataset, reflecting the distinctiveness of each document&#x2019;s language use. Finally, the <italic>average pointwise mutual information (PMI)</italic> for word bigrams was determined to assess the strength of association between commonly co-occurring words. Note, the use of PMI was inspired by the mutual information approach used in the neurophysiology study by <xref ref-type="bibr" rid="ref112">van der Willigen et al. (2024)</xref>, which measures the dependence between complex spectral-temporal sound representations. Both methods apply principles of information theory to quantify meaningful relationships related to NLP, though in distinct data domains (language vs. auditory processing) and at different scales (word pairs vs. neural coding of sound features).</p>
</sec>
</sec>
<sec sec-type="results" id="sec24">
<label>4</label>
<title>Results</title>
<p>We start by reporting on a document-level assessment (Section 4.1), examining whether the generated data are contextually and semantically consistent with real EHRs. This is followed by a corpus-level assessment (Section 4.2) to assess whether our GA-assisted SHDG protocol adhered to established clinical data standards and preserved the statistical properties of the sample dataset (for details, see Section 3.5).</p>
<sec id="sec25">
<label>4.1</label>
<title>Document level assessment</title>
<p>The first 4 rows of <xref ref-type="fig" rid="fig4">Figure 4</xref>&#x2014;Size (MB), Word Count, Unique Words, Document Length (Chars)&#x2014;provide a structural element characterization of the original real-world, pseudonymized, and synthetic datasets, respectively. Each real-world EHR document&#x2014;provided in PDF format&#x2014;was analyzed for both linguistic quality and informational content. Structural document components&#x2014;such as tables and images&#x2014;were also quantified; across the original EHRs, between two and four tables were identified per file, while no figures (images) were detected in any document.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Document level assessment using a holistic benchmark framework for quantitative evaluation of synthetic data quality (see <xref ref-type="table" rid="tab4">Table 4</xref>). The figure presents a matrix of violin plots comparing the distributions of eight features&#x2014;Size (MB), Word Count, Unique Words, Document Length (Chars), Character Entropy, Word Entropy, Average Pointwise Mutual Information (PMI), and Jensen-Shannon (JS) Distance&#x2014;across three datasets: <bold>(A)</bold> real-world PDF (<italic>N</italic>&#x202F;=&#x202F;13), <bold>(B)</bold> pseudonymized markdown (<italic>N</italic>&#x202F;=&#x202F;13), and <bold>(C)</bold> synthetic markdown (<italic>N</italic>&#x202F;=&#x202F;20). Note, each feature is encoded by a unique color for visual clarity (see Legend upper right side of the figure). Violin plots combine aspects of box plots and kernel density plots to provide a nuanced visualization of distributional characteristics. Specifically, the width of each violin at a given value represents the estimated probability density of the data at that value, as calculated by a kernel density estimator. This allows for the depiction of multimodality, skewness, and overall distributional shape, beyond summary statistics such as mean or quartiles. In this matrix, each subplot includes both the violin plot and overlaid scatter points indicating individual data instances, thereby facilitating both distributional and sample-level comparison of features across the three datasets.</p>
</caption>
<graphic xlink:href="frai-08-1644084-g004.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Violin plots comparing various metrics across three datasets: Real-World PDF, Pseudonymized Markdown, and Synthetic Markdown. Metrics shown are size in megabytes, word count, unique words, document length in characters, character entropy, word entropy, average pointwise mutual information, and Jensen-Shannon divergence distance. Each metric is color-coded and the plots illustrate distributions with density and outlier points.</alt-text>
</graphic>
</fig>
<p>When converting the original, real-world sample data from PDF to pseudonymised Markdown format, file sizes dropped noticeably. This is because PDF files retain a large amount of information&#x2014;including embedded fonts, images, and detailed layout instructions&#x2014;to ensure consistent appearance across devices. In contrast, Markdown files contain only the essential textual content and minimal formatting information, omitting media and complex layout data, which makes them much more compact (first row, <xref ref-type="fig" rid="fig4">Figure 4</xref>), However, during the pseudonymization phase, we added information to the documents (see Section 3.1.1). Consequently, the word count, and document length of the pseudonymized data increased somewhat (second row, <xref ref-type="fig" rid="fig4">Figure 4</xref>), reflecting the inclusion of additional descriptive tokens and labels in the file.</p>
<p>The overall structure of the synthetic data differed substantially from the pseudonymized data: on average, synthetic clinical notes were approximately 30% shorter in length (9,412 vs. 13,370). This reduction suggests that synthetic documents contained significantly less content per record, potentially omitting important clinical details or context. While real-world EHRs often include duplicate or redundant information (<xref ref-type="bibr" rid="ref76">Nijor et al., 2022</xref>; <xref ref-type="bibr" rid="ref104">Steinkamp et al., 2022</xref>), we observed that the synthetic data lacked such redundancy. This absence likely accounts for the notable decrease in word count and document length in the synthetic notes, even though the number of unique words remained similar between the two datasets.</p>
<p>At the per-document level, the mean bigram PMI was also slightly lower for synthetic documents (6.26) compared to pseudonymized documents (6.40); however, this difference was not statistically significant (Mann&#x2013;Whitney U&#x202F;=&#x202F;161.00, <italic>p</italic>&#x202F;=&#x202F;0.2611). This suggests that although, on average, synthetic texts have somewhat weaker word pairings, individual documents do not consistently differ in bigram association strength, indicating that variability in bigram usage exists within both pseudonymized and synthetic documents.</p>
</sec>
<sec id="sec26">
<label>4.2</label>
<title>Corpus level assessment</title>
<p>We compared our synthetic clinical text to the real-world, pseudonymized clinical notes at corpus level (Section 4.2) using informational and linguistic measures, including Shannon&#x2019;s Entropy and average bigram PMI. The results are shown in <xref ref-type="table" rid="tab6">Table 6</xref>.</p>
<table-wrap position="float" id="tab6">
<label>Table 6</label>
<caption>
<p>Corpus level assessment: pseudonymized (Pseudo) versus synthetic (Synth) text corpora.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Metric <italic>interpretation</italic></th>
<th align="center" valign="top">Pseudo mean</th>
<th align="center" valign="top">Synth mean</th>
<th align="center" valign="top">Mann-whitney U (U-stat)</th>
<th align="center" valign="top"><italic>p</italic>-value</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top" colspan="5">Textual diversity metrics</td>
</tr>
<tr>
<td align="left" valign="top">Corpus shannon entropy (character)<break/><italic>Pseudo &#x003E; Synth character diversity</italic></td>
<td align="center" valign="top">4.8565</td>
<td align="center" valign="top">4.6936</td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">&#x2014;</td>
</tr>
<tr>
<td align="left" valign="top">Corpus shannon entropy (word)<break/><italic>Synth &#x003E; Pseudo word diversity</italic></td>
<td align="center" valign="top">9.3543</td>
<td align="center" valign="top">9.9799</td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">&#x2014;</td>
</tr>
<tr>
<td align="left" valign="top">Mean per-document shannon entropy (character)<break/><italic>Pseudo &#x003E; Synth, significant difference</italic></td>
<td align="center" valign="top">4.8449</td>
<td align="center" valign="top">4.6846</td>
<td align="center" valign="top">U = 260.00</td>
<td align="center" valign="top"><italic>p</italic>&#x003C;0.001</td>
</tr>
<tr>
<td align="left" valign="top">Mean per-document shannon entropy (word)<break/><italic>Synth &#x003E; Pseudo, significant difference</italic></td>
<td align="center" valign="top">8.3075</td>
<td align="center" valign="top">8.6939</td>
<td align="center" valign="top">U = 0.00</td>
<td align="center" valign="top"><italic>p</italic>&#x003C;0.001</td>
</tr>
<tr>
<td align="left" valign="top" colspan="5">Distributional differences</td>
</tr>
<tr>
<td align="left" valign="top">JSD (word dist. between corpora)<break/><italic>Moderate divergence</italic></td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">0.3770</td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">&#x2014;</td>
</tr>
<tr>
<td align="left" valign="top">Mean Per-Doc JSD (word-level)<break/>vs. combined corpus<break/><italic>Synth more divergent,</italic><break/><italic>significant difference</italic></td>
<td align="center" valign="top">0.2297</td>
<td align="center" valign="top">0.2618</td>
<td align="center" valign="top">U = 12.00</td>
<td align="center" valign="top"><italic>p</italic>&#x003C;0.001</td>
</tr>
<tr>
<td align="left" valign="top" colspan="5">Linguistic associations</td>
</tr>
<tr>
<td align="left" valign="top">Corpus average bigram PMI (min freq=3)<break/><italic>Pseudo &#x003E; Synth word pair associations</italic></td>
<td align="center" valign="top">6.9187</td>
<td align="center" valign="top">5.9833</td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">&#x2014;</td>
</tr>
<tr>
<td align="left" valign="top">Mean per-document bigram PMI<break/><italic>Not statistically significant</italic></td>
<td align="center" valign="top">6.4007</td>
<td align="center" valign="top">6.2604</td>
<td align="center" valign="top">U = 161.00</td>
<td align="center" valign="top"><italic>P</italic>=0.2611</td>
</tr>
<tr>
<td align="left" valign="top" colspan="5">Document structure</td>
</tr>
<tr>
<td align="left" valign="top">Average document length (characters)<break/><italic>Pseudonymized docs much longer</italic></td>
<td align="center" valign="top">13,370.15</td>
<td align="center" valign="top">9,412.25</td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">&#x2014;</td>
</tr>
<tr>
<td align="left" valign="top" colspan="5">Surface &#x0026; Semantic similarity</td>
</tr>
<tr>
<td align="left" valign="top">BLEU score (synthetic vs. pseudonymized)<break/><italic>Very low n-gram overlap,</italic><break/><italic>low surface similarity</italic></td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">4.6179</td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">&#x2014;</td>
</tr>
<tr>
<td align="left" valign="top">BERTScore (synthetic vs. pseudonymized)<break/><italic>Moderate similarity</italic></td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">0.6447</td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">&#x2014;</td>
</tr>
<tr>
<td align="left" valign="top" colspan="5">Classification metrics</td>
</tr>
<tr>
<td align="left" valign="top">Precision<break/><italic>Semantic/textual discrimination</italic></td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">0.6556</td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">&#x2014;</td>
</tr>
<tr>
<td align="left" valign="top">Recall pseudonymized&#x2014;F1<break/><italic>Similarity measure</italic></td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">0.6499</td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">&#x2014;</td>
</tr>
<tr>
<td align="left" valign="top">Classifier AUC (pseudo vs. synthetic)<break/><italic>Perfectly distinguishes</italic><break/><italic>lower = better for synthetic</italic></td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">1.0000</td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">&#x2014;</td>
</tr>
<tr>
<td align="left" valign="top">Classifier AUPRC (pseudo vs. synthetic)<break/><italic>Perfectly distinguishes (lower = better for synthetic)</italic></td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">1.0000</td>
<td align="center" valign="top">&#x2014;</td>
<td align="center" valign="top">&#x2014;</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Evaluation metrics used&#x2014;see Table 5, for detailed description&#x2014;are character and word diversity (Shannon entropy), distributional differences (JSD), linguistic associations (Average Bigram PMI), document length, surface and semantic similarity (BLEU and BERTScore), and the ability of machine learning classifiers to distinguish the two types of documents (AUC and AUPRC). Note, the Mann&#x2013;Whitney U-test&#x2014;reporting both statistic and p-value&#x2014;is used to determine whether observed differences between the corpora are statistically significant. This non-parametric test was chosen because it does not assume a normal distribution, making it suitable for text-derived metrics that often violate this assumption, and thereby provides a robust evaluation of whether differences reflect meaningful distinctions rather than random variation.</p>
</table-wrap-foot>
</table-wrap>
<p>The corpus-level Shannon entropy quantifies the diversity and unpredictability of character and word usage within a corpus. For character-level entropy, the pseudonymized corpus (4.8565) exhibited a higher value than the synthetic corpus (4.6936), indicating that texts in the pseudonymized set utilize a greater variety of characters or employ characters in a less predictable manner. This suggests that the process of synthetically generating text may introduce constraints or redundancies at the character level, resulting in reduced diversity.</p>
<p>Conversely, word-level corpus entropy was higher in the synthetic corpus (9.9799) compared to the pseudonymized corpus (9.3543). This reflects a broader or less predictable word usage in the synthetic data, potentially attributable to the generative process introducing new combinations of words or emphasizing novelty. Thus, while synthetic data appears to be less varied at the character level, it is more varied at the word level than the original pseudonymized corpus.</p>
<p>Beyond overall corpus-level entropy, per-document analysis further clarifies differences in diversity and distribution between the pseudonymized and synthetic corpora. Mean per-document Shannon entropy at the character level was higher for the pseudonymized documents (4.8449) than for synthetic ones (4.6846), and this difference was statistically significant (Mann&#x2013;Whitney U&#x202F;=&#x202F;260.00, <italic>p</italic>&#x202F;&#x003C;&#x202F;0.001). This corroborates the corpus-level finding, indicating that, on an individual document basis, pseudonymized texts are consistently more diverse and less predictable regarding character usage than their synthetic counterparts.</p>
<p>Moreover, mean per-document Shannon entropy at the word level was significantly higher for synthetic documents (8.6939) than for pseudonymized ones (8.3075) (Mann&#x2013;Whitney U&#x202F;=&#x202F;0.00, <italic>p</italic>&#x202F;&#x003C;&#x202F;0.001). Thus, at the document level, synthetic texts exhibit greater unpredictability and a broader vocabulary than those found in the pseudonymized set, as was the case for corpus-level entropy.</p>
<p>For Average Bigram PMI, which quantifies the strength of association between word pairs by measuring how much more likely two words are to occur together than would be expected by chance, we observed that the synthetic corpus exhibited weaker word pair associations compared to the pseudonymized corpus. Specifically, the corpus-average bigram PMI was lower in the synthetic data (5.98) than in the pseudonymized data (6.92). This indicates that bigrams in synthetic texts are less strongly associated, or less conventional, than those present in the real-world clinical narratives. This finding likely reflects the inherent repetitiveness and predictability of authentic EHRs, in which common phraseology and co-occurring terms are documented repeatedly, thereby increasing the frequency and association strength of certain word pairs. In contrast, synthetically generated texts may introduce more varied or less natural co-occurrences, leading to overall weaker bigram associations.</p>
<p>In contrast to Shannon&#x2019;s Entropy and average bigram PMI, metrics such as JSD, BLEU score, BERTScore, and classifier performance directly provide a comparative value as output (<xref ref-type="table" rid="tab6">Table 6</xref>). While the synthetic data demonstrates several meaningful similarities to real data, there are also notable shortcomings.</p>
<p>The mean per-document Jensen-Shannon divergence (JSD) of word distributions relative to the combined corpus is higher in synthetic documents (0.2618) than in pseudonymized ones (0.2297), a statistically significant difference (Mann&#x2013;Whitney U&#x202F;=&#x202F;12.00, <italic>p</italic>&#x202F;&#x003C;&#x202F;0.001). This signifies that individual synthetic documents tend to diverge more from the aggregate corpus word distribution than pseudonymized documents, implying less conformity and potentially greater variation in how topics or vocabulary are expressed in synthetic data.</p>
<p>In surface-level text analysis, an <italic>n-gram</italic> refers to a sequence of <italic>n</italic> consecutive words&#x2014;such as a bigram (two words) or a trigram (three words)&#x2014;and comparing n-grams between documents helps reveal how closely their wording and phrasing align. Surface-level text similarity, as measured by the BLEU score (4.62 out of 100), was very low, indicating minimal n-gram overlap between synthetic and pseudonymized texts. This result bears out that the synthetic data are not merely replicating or closely paraphrasing real document phrases but are instead generating genuinely novel text content. While a low BLEU score might be viewed as a negative outcome in tasks requiring close mimicry, in the context of privacy-preserving data synthesis, it is encouraging. It demonstrates that the generative Agent workflow is not merely memorizing or reproducing existing expressions from the source corpus, but creating new, diverse language that reduces risks of information leakage.</p>
<p>Semantic similarity, as assessed by BERTScore, was moderate, with an F1 score of approximately 0.65 (precision: 0.64, recall: 0.66). Thus, although the synthetic data exhibits low surface-level overlap and increased lexical diversity compared to authentic EHR notes, it nonetheless preserves a substantial portion of the underlying clinical meaning and topical content. Such moderation in semantic overlap tells us that the sentences and phrases in the synthetic data are not directly copied or closely matched, word-for-word, with those in the original clinical records (authentic EHR notes). When we look at common sequences of words (n-grams), there is very little overlap between the two sets&#x2014;meaning the synthetic data displays different combinations of words and sentences, rather than repeating those found in the real notes. In our use case of synthesizing clinical narratives, very high F1, precision, and recall measures would indicate exact copies of original data, whereas our aim was to increase textual diversity by adding real-world samples and clinical practice guidelines as knowledge bases for reasoning.</p>
<p>A machine learning classifier trained to distinguish between synthetic and pseudonymized (real) documents attained perfect discrimination, with both AUC and AUPRC scores of 1.00. This result shows that there are clear, easily learnable feature differences between the two datasets&#x2014;particularly those reflected in TF-IDF representations. Among these, document length emerged as a primary distinguishing characteristic. Consequently, while the synthetic dataset demonstrates advantages such as content novelty and satisfactory conceptual coverage, its inability to realistically replicate document length results in synthetic documents being consistently and trivially separable from real ones. This highlights the need for improved modeling of document-level properties to enhance the realism and utility of synthetic clinical text. It also provides a clear path forward: simply adjusting the length of synthetic documents to match that of authentic clinical notes, or by deleting redundant information from the authentic clinical notes, should eliminate the primary feature the classifier uses to tell them apart. In doing so, the synthetic texts would likely become much harder for automated classifiers to distinguish from real ones, substantially improving their realism and the utility of the synthetic dataset.</p>
<p>Our corpus-level assessment&#x2014;drawing on both <xref ref-type="fig" rid="fig4">Figure 4</xref> and <xref ref-type="table" rid="tab6">Table 6</xref>&#x2014;shows that clinical text derived through GA-assisted SHDG are more predictable at the character level but surpass pseudonymized notes in word-level diversity and unpredictability. This pattern likely stems from artifacts or variability introduced during text generation, with important implications for the realism and utility of synthetic data. Overall, our findings highlight that while synthetic EHRs successfully avoids direct replication and privacy risks by producing novel word combinations, they also exhibit weaker conventional phrase associations and greater divergence from word distributions observed in the real corpus in.</p>
</sec>
</sec>
<sec sec-type="discussion" id="sec27">
<label>5</label>
<title>Discussion</title>
<p>Our present work addresses a persistent barrier in digital health GenAI: the lack of accessible, interoperable, and privacy-preserving datasets that capture the diversity of real-world healthcare documentation (<xref ref-type="bibr" rid="ref30">Eigenschink et al., 2023</xref>; <xref ref-type="bibr" rid="ref74">Murtaza et al., 2023</xref>; <xref ref-type="bibr" rid="ref45">Hernandez et al., 2025</xref>; <xref ref-type="bibr" rid="ref50">Ibrahim et al., 2025</xref>; <xref ref-type="bibr" rid="ref66">Loni et al., 2025</xref>). Building on our stepwise strategy for constructing a learning health system (LHS) (<xref ref-type="bibr" rid="ref113">van Velzen et al., 2023</xref>), we argue that a fully integrated LHS will remain unattainable until these data challenges&#x2014;especially in nursing and allied health (<xref ref-type="bibr" rid="ref108">Tischendorf et al., 2025</xref>)&#x2014;are resolved.</p>
<p>Echoing &#x201C;<italic>On the Dangers of Stochastic Parrots</italic>&#x201D; (<xref ref-type="bibr" rid="ref10">Bender et al., 2021</xref>), we stress that healthcare must ask whether public cloud LLMs can be used in ways that are truly FAIR&#x2014;Findable, Accessible, Interoperable, and Reusable&#x2014;while mitigating risks such as high environmental and financial costs, opaque or biased data, and amplification of inequities. This requires curating and documenting high-quality datasets, aligning development with research and stakeholder values, and exploring approaches beyond ever-larger models.</p>
<p>A viable solution is the adoption of compact, fine-tunable SLMs. These small LLM alternatives can be deployed directly within healthcare facilities. Running locally not only reduces reliance on external cloud services but also lowers operational costs, decreases energy demands, and enhances data privacy. SLMs are increasingly capable of powering point-of-care applications&#x2014;from clinical decision support to patient communication&#x2014;while ensuring sensitive information remains within institutional boundaries (<xref ref-type="bibr" rid="ref95">Schick and Sch&#x00FC;tze, 2021</xref>; <xref ref-type="bibr" rid="ref26">Dibia et al., 2024</xref>; <xref ref-type="bibr" rid="ref34">Garg et al., 2025</xref>; <xref ref-type="bibr" rid="ref56">Kim et al., 2025</xref>; <xref ref-type="bibr" rid="ref121">Xie et al., 2025</xref>).</p>
<p>By establishing a no-code, protocol for creating GA-assisted SHDG workflows&#x2014;enabled by rapid prototyping platforms and further enhanced through fully automated, GenAI-assisted co-development&#x2014;it is achievable to significantly lower the technical threshold for GenAI engagement. Leveraging the DSI stack as a generic blueprint architecture (<xref ref-type="fig" rid="fig1">Figure 1A</xref>), our methodological approach illustrates how modular, no-code frameworks can be systematically employed to streamline and democratize the creation of intelligent healthcare workflows.</p>
<p>In particular, the use of GA-assisted workflows to support clinical reasoning for nurse specialists&#x2014;demonstrated through the &#x201C;Nandalyse&#x201D;<xref ref-type="fn" rid="fn0011"><sup>11</sup></xref> tool at the 2025 ASCENDIO conference&#x2014;represents a major shift in healthcare technology development and adoption in the Netherlands (<xref ref-type="bibr" rid="ref60">Kumar, 2023</xref>; <xref ref-type="bibr" rid="ref52">Jeong, 2025</xref>; <xref ref-type="bibr" rid="ref108">Tischendorf et al., 2025</xref>). These novel GenAI tools enable clinicians and other non-technical users to easily create and customize workflows using simple, modular, drag-and-drop interfaces, significantly lowering the barriers to participation.</p>
<p>A key aspect of our protocol for GA-assisted SHDG workflows is the explicit bridging of the gap between end users&#x2014;such as clinicians, quality officers, and researchers&#x2014;and the technical developers responsible for constructing and maintaining AI systems (as detailed in <xref ref-type="fig" rid="fig1">Figure 1B</xref>). In healthcare, this chiasm is often perpetuated by differences in language, priorities, and familiarity with digital tools. The explicit integration of human-in-the-loop (<xref ref-type="bibr" rid="ref4">Alemohammad et al., 2024</xref>) in no-code GA-assisted SHDG workflows not only supports iterative co-development (<xref ref-type="bibr" rid="ref63">Li et al., 2023</xref>) but also ensures that the system remains transparent and grounded in real-world clinical needs and documentation standards (<xref ref-type="bibr" rid="ref21">Chung et al., 2023</xref>).</p>
<p>Notably, our methodology fosters closer collaboration between users and developers by providing open-source GitHub repositories (see text footnote 4), which make design choices, workflow logic, and evaluation criteria more transparent and verifiable. However, realizing the full promise of these platforms will require continued investment in user education, ongoing refinement of documentation and support resources, and the cultivation of communities of practice around open-source synthetic data generation.</p>
<p>The successful adoption of privacy preserving GA-assisted SHDG workflows in clinical practice, healthcare professionals requires more than technical proficiency alone (<xref ref-type="bibr" rid="ref60">Kumar, 2023</xref>; <xref ref-type="bibr" rid="ref19">Chew and Ngiam, 2025</xref>; <xref ref-type="bibr" rid="ref52">Jeong, 2025</xref>); they must also possess a comprehensive understanding of key data privacy principles&#x2014;such as pseudonymization and de-identification&#x2014;to ensure patient confidentiality is maintained (<xref ref-type="bibr" rid="ref28">Drechsler and Haensch, 2024</xref>; <xref ref-type="bibr" rid="ref92">Rujas et al., 2025</xref>). Especially, clinicians should be aware of the inherent limitations of LLMs, including potential risks of bias, hallucination, and model drift, all of which may impact the fidelity and safety of synthetic health data (<xref ref-type="bibr" rid="ref4">Alemohammad et al., 2024</xref>; <xref ref-type="bibr" rid="ref66">Loni et al., 2025</xref>). <xref ref-type="bibr" rid="ref41">Harnad (2025)</xref>, notes that LLMs rely on stochastic patterns by capturing statistical regularities rather than achieving genuine semantic understanding, meaning that even fluent and coherent output may be inaccurate, irrelevant or misleading. <xref ref-type="bibr" rid="ref101">Shojaee et al. (2025)</xref> further caution that the apparent reasoning proficiency of such models can deteriorate markedly as problem complexity increases&#x2014;an &#x201C;illusion of thinking&#x201D; with significant implications for clinical safety. Collectively, these observations reinforce the imperative for multidimensional evaluation frameworks that integrate both surface-level measures (e.g., document length, lexical overlap) and deep-level metrics (e.g., semantic alignment, diversity) to rigorously assess the fidelity and practical utility of synthetic narratives across diverse, high-stakes clinical contexts (<xref ref-type="bibr" rid="ref100">Shannon, 1948</xref>; <xref ref-type="bibr" rid="ref85">Post, 2018</xref>; <xref ref-type="bibr" rid="ref124">Zhang et al., 2019</xref>).</p>
<p>Importantly, working with GenAI in Healthcare mandates a clear understanding of evaluation metrics (<xref ref-type="bibr" rid="ref30">Eigenschink et al., 2023</xref>; <xref ref-type="bibr" rid="ref1">Abdurahman et al., 2025</xref>; <xref ref-type="bibr" rid="ref50">Ibrahim et al., 2025</xref>). Because no single metric can holistically capture the quality of synthetic clinical text, a comprehensive assessment should integrate document- and corpus-level metrics (for overview see <xref ref-type="table" rid="tab4">Tables 4</xref>, <xref ref-type="table" rid="tab5">5</xref>, respectively): document length and average word count for surface features, entropy for textual diversity (<xref ref-type="bibr" rid="ref100">Shannon, 1948</xref>), BLEU for lexical overlap (<xref ref-type="bibr" rid="ref85">Post, 2018</xref>), and BERTScore for semantic alignment (<xref ref-type="bibr" rid="ref124">Zhang et al., 2019</xref>). Together, these complementary metrics provide a nuanced perspective on both the linguistic and informative value of synthetic narratives. Importantly, high performance on one dimension (e.g., diversity as measured by entropy) does not necessarily translate to strong semantic faithfulness (as measured by BERTScore). This distinction is particularly relevant when evaluating synthetic data for diverse clinical contexts, such as physiotherapy documentation in high-risk or emotionally fraught scenarios (e.g., cardiac, oncology, or orthopedic surgery). Our results (<xref ref-type="table" rid="tab6">Table 6</xref>) reinforce the necessity of using an ensemble of surface and deep metrics to robustly assess the utility and fidelity of synthetic clinical narratives, supporting their appropriate integration into research and practice.</p>
<p>Our work is not without limitations. We identified four main operational and technical challenges that must be addressed to advance GA-assisted SHDG workflows. First, comparative evaluation of on-premises versus cloud-based AI models (<xref ref-type="table" rid="tab1">Table 1</xref>) is needed to optimize trade-offs between data privacy, computational performance, and cost (<xref ref-type="bibr" rid="ref95">Schick and Sch&#x00FC;tze, 2021</xref>; <xref ref-type="bibr" rid="ref34">Garg et al., 2025</xref>; <xref ref-type="bibr" rid="ref121">Xie et al., 2025</xref>). Parameter tuning&#x2014;including the adjustment of prompt temperature and chunk size&#x2014;must be empirically refined to balance realism and diversity in generated outputs (<xref ref-type="bibr" rid="ref82">Peeperkorn et al., 2024</xref>). Second, while our GA prompts were intentionally crafted to encourage adherence to established clinical practice guidelines to generate clinically relevant outputs, we acknowledge that real-world clinical practice often diverges from these standards. Notably, intentional non-adherence to guidelines has been observed in up to 65% of cases within EHRs, frequently attributable to specific patient factors such as contraindications, comorbidities, or individual preferences (<xref ref-type="bibr" rid="ref6">Arts et al., 2016</xref>). Therefore, for GA-assisted SHDG workflows to remain practical and reflective of authentic clinical scenarios, addressing this variability is essential. Consequently, our prompt design may have inadvertently constrained the clinical diversity of the synthesized narratives&#x2014;a limitation that may have been further amplified by the small sample size of our authentic clinical EHR documentation dataset (N&#x202F;=&#x202F;13 PDF documents). Future SHDG workflows should explicitly address these limitations through testing various prompt engineering techniques to better capture the variability inherent in real-world clinical narratives, or alternatively, by increasing the size of or sample dataset to encompass a broader range of patient presentations, care contexts, and documentation styles and a wider range of intentional non-adherence to guidelines. Implementing these changes&#x2014;refining prompt-engineering techniques and expanding the dataset&#x2014;would improve the representativeness of the generated outputs and better align them with the complexities of actual clinical practice. Third, we identified additional prompt engineering issues (<xref ref-type="fig" rid="fig3">Figure 3</xref>). These included the inadvertent introduction of gender bias&#x2014;all synthetic patients turned out to be female&#x2014;and inconsistent handling of abbreviations&#x2014;where sample and markdown files contained frequent abbreviations, but synthetic data did not. We also found another prompt engineering issue that resulted in a one-size-fits-all approach in the generated therapy plans and reduced therapy frequency variability. Document length discrepancies, often resulting from repetition in sample or markdown files, also require further exploration to ensure consistent structural fidelity across datasets. Recent research underscores that prompt engineering is not a value-neutral process and that different components of a prompt can vary significantly in their robustness and susceptibility to bias. <xref ref-type="bibr" rid="ref68">Mei et al. (2025)</xref> provide a comprehensive survey of &#x201C;context engineering&#x201D; strategies for large language models, highlighting how the choice, structuring, and sequencing of contextual elements can systematically influence model outputs. <xref ref-type="bibr" rid="ref126">Zheng et al. (2025)</xref> further demonstrate that individual prompt components&#x2014;such as instructions, examples, or delimiters&#x2014;exhibit heterogeneous adversarial robustness, meaning that some parts are more vulnerable to manipulation or unintended bias than others. Related work on priming effects shows that the initial context or examples provided to a model can strongly condition its subsequent responses, amplifying or dampening biases and shaping output diversity (<xref ref-type="bibr" rid="ref125">Zhao et al., 2021</xref>; <xref ref-type="bibr" rid="ref33">Gallegos et al., 2024</xref>). <xref ref-type="bibr" rid="ref48">Huang et al. (2025)</xref> demonstrate that targeted priming can exploit intrinsic weaknesses in large language models, revealing latent vulnerabilities that may not be apparent under standard prompting conditions. <xref ref-type="bibr" rid="ref69">Meinke et al. (2024)</xref> further reveal that frontier-scale models are capable of sophisticated in-context behaviors, sometimes strategically adapting to earlier cues in ways that can subtly steer reasoning and decision-making. In addition, <xref ref-type="bibr" rid="ref33">Gallegos et al. (2024)</xref> synthesize evidence that such biases can emerge not only from pre-training data but also from contextual framing and priming during inference. Together, these findings suggest that the gender bias, abbreviation inconsistencies, and homogenized therapy plans observed in our study may stem not only from prompt content but also from the structural composition, priming effects, and resilience of the prompts themselves. This reinforces the need for systematic evaluation and refinement of both prompt components and priming strategies to mitigate bias and improve variability in generated outputs. Fourth, the normalization of clinical text presents a significant methodological challenge, as variations in terminology&#x2014;for instance, describing the same condition as &#x201C;sciatica&#x201D; versus &#x201C;lumbosacral radicular pain syndrome&#x201D;&#x2014;often reflect individual practitioner preferences as well as institutional conventions (<xref ref-type="bibr" rid="ref105">Suominen et al., 2013</xref>; <xref ref-type="bibr" rid="ref110">U.S.-National-Library-of-Medicine, 2024</xref>). This semantic variability can be quantitatively monitored using mutual information-based evaluation metrics (see Section 3.4); in our study, we applied the average bigram pointwise mutual information (PMI) metric. However, the robustness of this metric across more diverse datasets and practitioners from different clinical specialties remains to be fully validated. Thus, relying solely on either surface-level metrics or deep semantic measures is insufficient. Instead, a comprehensive evaluation of the faithfulness of SHDG demands the integration of both approaches (<xref ref-type="bibr" rid="ref30">Eigenschink et al., 2023</xref>; <xref ref-type="bibr" rid="ref103">Smolyak et al., 2024</xref>; <xref ref-type="bibr" rid="ref17">Chen et al., 2025b</xref>; <xref ref-type="bibr" rid="ref50">Ibrahim et al., 2025</xref>; <xref ref-type="bibr" rid="ref94">Scherr et al., 2025</xref>).</p>
</sec>
<sec sec-type="conclusions" id="sec28">
<label>6</label>
<title>Conclusion</title>
<p>Based on our findings, we recommend prioritizing the continued development and refinement of GA-assisted SHDG-workflows, ensuring that these toolchains remain accessible, transparent, and customizable for a diverse range of clinical users and researchers through the provision of open-source GitHub repositories (see text footnote 4). Our recommendations align with ongoing efforts by the RUAS Healthcare DataLab to advance innovation in care through collaborative, technology-driven solutions that integrate open, adaptable tools into diverse healthcare contexts.<xref ref-type="fn" rid="fn0012"><sup>12</sup></xref> In parallel, our talent program fosters and equips RUAS students with the skills and expertise required to become proficient data science professionals, thereby strengthening the capacity for data-driven innovation within the Dutch healthcare sector. To maximize impact, future initiatives should emphasize robust user education on essential data privacy concepts&#x2014;such as pseudonymization and de-identification&#x2014;and foster a deeper understanding of both the capabilities and limitations of no-code GenAI technologies, particularly regarding risks like bias, hallucination, and model drift. Comprehensive evaluation frameworks should integrate both surface-level and deep semantic metrics to robustly assess the linguistic and clinical fidelity of synthetic narratives, while methodological improvements&#x2014;including enhanced prompt engineering to address issues of bias, guideline adherence, and structural consistency, as well as adherence to normalization standards&#x2014;will be essential in capturing the diversity inherent in real-world clinical documentation. Comparative assessments of on-premises versus cloud-based deployments, plus empirical tuning of generative parameters, are also advised to optimize privacy, cyber security, cost, and performance trade-offs for varied healthcare settings. Finally, building active communities of practice and cultivating collaborative, iterative engagement between end-users and developers will be critical to realizing truly FAIR (<xref ref-type="bibr" rid="ref72">Mons et al., 2017</xref>), AI-ready LHS that can adapt to the complexities and dynamic requirements of modern clinical environments (<xref ref-type="bibr" rid="ref49">Huerta et al., 2023</xref>; <xref ref-type="bibr" rid="ref116">Verhulst et al., 2025</xref>).</p>
<p>Our protocolization of privacy-preserving, GA-assisted SHDG workflows underscores the critical importance of maintaining transparency by keeping humans actively involved in the process&#x2014;a principle known as <italic>human-in-the-loop</italic>. Specifically, we standardized the use of a modular, no-code system in which every step&#x2014;from designing AI prompts, to providing data inputs, to defining how outputs are applied&#x2014;can be easily inspected, understood, and refined without programming expertise. This approach enables users to trace how the LLM generates clinical narratives and to make step-by-step improvements over time. In parallel, the integration of an open-source GitHub repository (see text footnote 4), anchored by our modular DSI Stack (<xref ref-type="fig" rid="fig1">Figure 1A</xref>), supports the design and deployment of secure, reproducible, and scalable GA-assisted SHDG workflows across both on-premises and public cloud environments. By incorporating Docker-based containerization and API key&#x2013;secured inference endpoints via Hugging Face Spaces, we ensure controlled access, consistent performance, and community-driven enhancement, while shielding users from underlying infrastructure complexity and maintaining full transparency for healthcare and research applications.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec29">
<title>Data availability statement</title>
<p>The sample datasets referenced in this article are not publicly available, as they are proprietary to MediFit Bewegingscentrum Oss and the University of Applied Sciences Rotterdam. At present, approval to use these datasets beyond the university has not been granted. Individuals interested in accessing the sample data are kindly asked to contact Mark van Velzen at <email>m.van.velzen@hr.nl</email> with their request. The original contributions presented in the study&#x2014;source code, synthetic datasets and prompts&#x2014;are included in the article/supplementary material. Toolchain implementation details are provided to facilitate reproducibility and to encourage non-AI experts, such as healthcare workers, to confidently explore, adapt, and apply these workflows within their own professional contexts. Further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="ethics-statement" id="sec30">
<title>Ethics statement</title>
<p>Ethical approval was not required for this study involving humans, in accordance with local legislation and institutional requirements. The data were initially collected for the purpose of providing care. Prior to data extraction, all participants provided written informed consent, in accordance with the Declaration of Helsinki (<xref ref-type="bibr" rid="ref120">World Medical Association, 2025</xref>). Following informed consent, the data were anonymized for research purposes.</p>
</sec>
<sec sec-type="author-contributions" id="sec31">
<title>Author contributions</title>
<p>MV: Writing &#x2013; original draft, Formal analysis, Project administration, Methodology, Data curation, Writing &#x2013; review &#x0026; editing, Conceptualization, Investigation, Resources, Validation. RW: Writing &#x2013; original draft, Formal analysis, Resources, Methodology, Software, Project administration, Visualization, Data curation, Writing &#x2013; review &#x0026; editing, Validation, Investigation, Conceptualization, Supervision. VB: Investigation, Validation, Writing &#x2013; review &#x0026; editing, Conceptualization, Methodology, Formal analysis, Writing &#x2013; original draft. HG-W: Methodology, Conceptualization, Validation, Writing &#x2013; review &#x0026; editing, Writing &#x2013; original draft, Investigation. EJ: Writing &#x2013; review &#x0026; editing, Conceptualization. SL: Resources, Writing &#x2013; review &#x0026; editing. MiW: Software, Writing &#x2013; review &#x0026; editing, Methodology, Validation, Conceptualization, Formal analysis. MaW: Writing &#x2013; review &#x0026; editing, Validation, Software, Methodology. GR: Software, Writing &#x2013; review &#x0026; editing. RM: Software, Writing &#x2013; review &#x0026; editing. SS: Software, Writing &#x2013; review &#x0026; editing. LH: Conceptualization, Writing &#x2013; review &#x0026; editing, Software, Methodology. TH: Conceptualization, Writing &#x2013; review &#x0026; editing. NM: Writing &#x2013; review &#x0026; editing, Conceptualization. MS: Conceptualization, Writing &#x2013; review &#x0026; editing.</p>
</sec>
<sec sec-type="funding-information" id="sec32">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research and/or publication of this article.</p>
</sec>
<ack>
<p>Our sincere thanks go to Maurice A.C. van den Dobbelsteen&#x2014;innovation manager AI &#x0026; eHealth at KCZI and HR-DataLab Healthcare coordinator&#x2014;for his pivotal role in establishing our DataLab at RUAS. This project was truly a team effort, and his inspiring guidance, collaborative spirit, and insightful discussions were essential in enabling and shaping our research and Data Science talent program.</p>
</ack>
<sec sec-type="COI-statement" id="sec33">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec34">
<title>Generative AI statement</title>
<p>The authors declare that Gen AI was used in the creation of this manuscript. Generative tools, specifically Gemini Advanced, Perplexity, and Deep Research Pro, were utilized in developing the manuscript drafts to enhance the quality of human-written text produced by non-native English-speaking authors. The use of these tools was fully in accordance with relevant legal requirements and ethical guidelines, including Article 12 GDPR (transparency regarding data processing), Article 5 GDPR (purpose limitation and data minimization), Article 13 AI ACT (informing users about interactions with AI systems), and Article 50 AI ACT (transparency requirements for AI providers and users). Furthermore, all AI-generated content was carefully reviewed and edited by human authors to ensure accuracy, relevance, and adherence to academic and ethical standards. Importantly, this process was conducted in compliance with the principles set forth in the &#x201C;Nederlandse gedragscode wetenschappelijke integriteit&#x201D; (<xref ref-type="bibr" rid="ref57">Knaw et al., 2018</xref>).</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="sec35">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn0001"><p><sup>1</sup><ext-link xlink:href="https://github.com/FlowiseAI/Flowise" ext-link-type="uri">https://github.com/FlowiseAI/Flowise</ext-link></p></fn>
<fn id="fn0002"><p><sup>2</sup><ext-link xlink:href="https://github.com/langflow-ai/langflow" ext-link-type="uri">https://github.com/langflow-ai/langflow</ext-link></p></fn>
<fn id="fn0003"><p><sup>3</sup><ext-link xlink:href="https://www.microsoft.com/en-us/research/project/autogen/" ext-link-type="uri">https://www.microsoft.com/en-us/research/project/autogen/</ext-link></p></fn>
<fn id="fn0004"><p><sup>4</sup><ext-link xlink:href="https://github.com/HR-DataLab-Healthcare/RESEARCH_SUPPORT/blob/main/PROJECTS/Generative_Agent_based_Data-Synthesis/" ext-link-type="uri">https://github.com/HR-DataLab-Healthcare/RESEARCH_SUPPORT/blob/main/PROJECTS/Generative_Agent_based_Data-Synthesis/</ext-link></p></fn>
<fn id="fn0005"><p><sup>5</sup><ext-link xlink:href="https://github.com/HR-DataLab-Healthcare/RESEARCH_SUPPORT/tree/main/PROJECTS/Generative_Agent_based_Data-Synthesis/AGENT-FLOWS" ext-link-type="uri">https://github.com/HR-DataLab-Healthcare/RESEARCH_SUPPORT/tree/main/PROJECTS/Generative_Agent_based_Data-Synthesis/AGENT-FLOWS</ext-link></p></fn>
<fn id="fn0006"><p><sup>6</sup><ext-link xlink:href="https://github.com/docker/roadmap" ext-link-type="uri">https://github.com/docker/roadmap</ext-link></p></fn>
<fn id="fn0007"><p><sup>7</sup><ext-link xlink:href="https://huggingface.co/spaces" ext-link-type="uri">https://huggingface.co/spaces</ext-link></p></fn>
<fn id="fn0008"><p><sup>8</sup><ext-link xlink:href="https://docs.flowiseai.com/configuration/deployment/hugging-face" ext-link-type="uri">https://docs.flowiseai.com/configuration/deployment/hugging-face</ext-link></p></fn>
<fn id="fn0009"><p><sup>9</sup><ext-link xlink:href="https://eur-lex.europa.eu/eli/reg/2016/679/oj/eng" ext-link-type="uri">https://eur-lex.europa.eu/eli/reg/2016/679/oj/eng</ext-link></p></fn>
<fn id="fn0010"><p><sup>10</sup><ext-link xlink:href="https://digital-strategy.ec.europa.eu/en/policies/regulatory-framework-ai" ext-link-type="uri">https://digital-strategy.ec.europa.eu/en/policies/regulatory-framework-ai</ext-link></p></fn>
<fn id="fn0011"><p><sup>11</sup><ext-link xlink:href="https://github.com/HR-DataLab-Healthcare/RESEARCH_SUPPORT/tree/main/PROJECTS/Harnessing%20the%20Power%20of%20Gen-AI%20in%20Research" ext-link-type="uri">https://github.com/HR-DataLab-Healthcare/RESEARCH_SUPPORT/tree/main/PROJECTS/Harnessing%20the%20Power%20of%20Gen-AI%20in%20Research</ext-link></p></fn>
<fn id="fn0012"><p><sup>12</sup><ext-link xlink:href="https://www.rotterdamuas.com/research/projects-and-publications/innovations-in-care/healthcare-innovation-with-technology/ruas-datalab-healthcare/" ext-link-type="uri">https://www.rotterdamuas.com/research/projects-and-publications/innovations-in-care/healthcare-innovation-with-technology/ruas-datalab-healthcare/</ext-link></p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abdurahman</surname><given-names>S.</given-names></name> <name><surname>Salkhordeh Ziabari</surname><given-names>A.</given-names></name> <name><surname>Moore</surname><given-names>A. K.</given-names></name> <name><surname>Bartels</surname><given-names>D. M.</given-names></name> <name><surname>Dehghani</surname><given-names>M.</given-names></name></person-group> (<year>2025</year>). <article-title>A primer for evaluating large language models in social-science research</article-title>. <source>Adv. Methods Pract. Psychol. Sci.</source> <volume>8</volume>:<fpage>25152459251325174</fpage>. doi: <pub-id pub-id-type="doi">10.1177/25152459251325174</pub-id></citation></ref>
<ref id="ref2"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Abhishek</surname><given-names>M. K.</given-names></name> <name><surname>Rao</surname><given-names>D. R.</given-names></name></person-group> (<year>2021</year>). &#x201C;<article-title>Framework to secure docker containers</article-title>&#x201D; in <source>Fifth world conference on smart trends in systems security and sustainability (WorldS4)</source> (<publisher-loc>London, UK</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>152</fpage>&#x2013;<lpage>156</lpage>. doi: <pub-id pub-id-type="doi">10.1109/WorldS451998.2021.9514041</pub-id></citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ait</surname><given-names>A.</given-names></name> <name><surname>C&#x00E1;novas Izquierdo</surname><given-names>J. L.</given-names></name> <name><surname>Cabot</surname><given-names>J.</given-names></name></person-group> (<year>2025</year>). <article-title>On the suitability of hugging face hub for empirical studies</article-title>. <source>Empir. Softw. Eng.</source> <volume>30</volume>, <fpage>1</fpage>&#x2013;<lpage>48</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10664-024-10608-8</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Alemohammad</surname><given-names>S.</given-names></name> <name><surname>Casco-Rodriguez</surname><given-names>J.</given-names></name> <name><surname>Luzi</surname><given-names>L.</given-names></name> <name><surname>Humayun</surname><given-names>A. I.</given-names></name> <name><surname>Babaei</surname><given-names>H.</given-names></name> <name><surname>LeJeune</surname><given-names>D.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Self-consuming generative models go mad</article-title>, in: <conf-name>International conference on learning representations (ICLR)</conf-name>, (<publisher-loc>Vienna, AT</publisher-loc>). doi: <pub-id pub-id-type="doi">10.48550/arXiv.2307.01850</pub-id></citation></ref>
<ref id="ref5"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Alsentzer</surname><given-names>E.</given-names></name> <name><surname>Murphy</surname><given-names>J. R.</given-names></name> <name><surname>Boag</surname><given-names>W.</given-names></name> <name><surname>Weng</surname><given-names>W.-H.</given-names></name> <name><surname>Jin</surname><given-names>D.</given-names></name> <name><surname>Naumann</surname><given-names>T.</given-names></name> <etal/></person-group>. (<year>2019</year>). &#x201C;<article-title>Publicly available clinical BERT embeddings</article-title>&#x201D; in <source>Proceedings of the 2nd clinical natural language processing workshop</source> (<publisher-loc>Minneapolis, MN</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>72</fpage>&#x2013;<lpage>78</lpage>. doi: <pub-id pub-id-type="doi">10.18653/v1/W19-1909</pub-id></citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Arts</surname><given-names>D. L.</given-names></name> <name><surname>Voncken</surname><given-names>A. G.</given-names></name> <name><surname>Medlock</surname><given-names>S.</given-names></name> <name><surname>Abu-Hanna</surname><given-names>A.</given-names></name> <name><surname>van Weert</surname><given-names>H. C.</given-names></name></person-group> (<year>2016</year>). <article-title>Reasons for intentional guideline non-adherence: a systematic review</article-title>. <source>Int. J. Med. Inform.</source> <volume>89</volume>, <fpage>55</fpage>&#x2013;<lpage>62</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2016.02.009</pub-id>, PMID: <pub-id pub-id-type="pmid">26980359</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Baowaly</surname><given-names>M. K.</given-names></name> <name><surname>Lin</surname><given-names>C. C.</given-names></name> <name><surname>Liu</surname><given-names>C. L.</given-names></name> <name><surname>Chen</surname><given-names>K. T.</given-names></name></person-group> (<year>2019</year>). <article-title>Synthesizing electronic health records using improved generative adversarial networks</article-title>. <source>J. Am. Med. Inform. Assoc.</source> <volume>26</volume>, <fpage>228</fpage>&#x2013;<lpage>241</lpage>. doi: <pub-id pub-id-type="doi">10.1093/jamia/ocy142</pub-id>, PMID: <pub-id pub-id-type="pmid">30535151</pub-id></citation></ref>
<ref id="ref8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barrault</surname><given-names>L.</given-names></name> <name><surname>Duquenne</surname><given-names>P.-A.</given-names></name> <name><surname>Elbayad</surname><given-names>M.</given-names></name> <name><surname>Kozhevnikov</surname><given-names>A.</given-names></name> <name><surname>Alastruey</surname><given-names>B.</given-names></name> <name><surname>Andrews</surname><given-names>P.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Large concept models: language modeling in a sentence representation space</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2412.08821</pub-id></citation></ref>
<ref id="ref9"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Beam</surname><given-names>A.L.</given-names></name> <name><surname>Kompa</surname><given-names>B.</given-names></name> <name><surname>Schmaltz</surname><given-names>A.</given-names></name> <name><surname>Fried</surname><given-names>I.</given-names></name> <name><surname>Weber</surname><given-names>G.</given-names></name> <name><surname>Palmer</surname><given-names>N.</given-names></name> <etal/></person-group>. (<year>2020</year>). "<article-title>Clinical concept embeddings learned from massive sources of multimodal medical data</article-title>", in: <conf-name>Pacific symposium on Biocomputing 2020</conf-name>, (<publisher-loc>Singapore</publisher-loc>: <publisher-name>World Scientific</publisher-name>), <volume>25</volume>:<fpage>295</fpage>&#x2013;<lpage>306</lpage></citation></ref>
<ref id="ref10"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Bender</surname><given-names>E. M.</given-names></name> <name><surname>Gebru</surname><given-names>T.</given-names></name> <name><surname>McMillan-Major</surname><given-names>A.</given-names></name> <name><surname>Shmitchell</surname><given-names>S.</given-names></name></person-group> (<year>2021</year>). "<article-title>On the dangers of stochastic parrots: can language models be too big?</article-title>", in: <conf-name>Proceedings of the 2021 ACM conference on fairness, accountability, and transparency: Association for Computing Machinery</conf-name>, <fpage>610</fpage>&#x2013;<lpage>623</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3442188.3445922</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bommasani</surname><given-names>R.</given-names></name> <name><surname>Hudson</surname><given-names>D. A.</given-names></name> <name><surname>Adeli</surname><given-names>E.</given-names></name> <name><surname>Altman</surname><given-names>R.</given-names></name> <name><surname>Arora</surname><given-names>S.</given-names></name> <name><surname>von Arx</surname><given-names>S.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>On the opportunities and risks of foundation models</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2108.07258</pub-id></citation></ref>
<ref id="ref12"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Brown</surname><given-names>T. B.</given-names></name> <name><surname>Mann</surname><given-names>B.</given-names></name> <name><surname>Ryder</surname><given-names>N.</given-names></name> <name><surname>Subbiah</surname><given-names>M.</given-names></name> <name><surname>Kaplan</surname><given-names>J.</given-names></name> <name><surname>Dhariwal</surname><given-names>P.</given-names></name> <etal/></person-group> (<year>2020</year>). <article-title>Language models are few-shot learners</article-title>, in: <conf-name>Proceedings of the 34th international conference on neural information processing systems</conf-name>, (<publisher-loc>NY, USA</publisher-loc>: <publisher-name>Curran Associates Inc</publisher-name>.). doi: <pub-id pub-id-type="doi">10.5555/3495724.3495883</pub-id></citation></ref>
<ref id="ref13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Busch</surname><given-names>F.</given-names></name> <name><surname>Kather</surname><given-names>J. N.</given-names></name> <name><surname>Johner</surname><given-names>C.</given-names></name> <name><surname>Moser</surname><given-names>M.</given-names></name> <name><surname>Truhn</surname><given-names>D.</given-names></name> <name><surname>Adams</surname><given-names>L. C.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Navigating the European union artificial intelligence act for healthcare</article-title>. <source>NPJ Digit Med</source> <volume>7</volume>:<fpage>210</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41746-024-01213-6</pub-id>, PMID: <pub-id pub-id-type="pmid">39134637</pub-id></citation></ref>
<ref id="ref14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cannon</surname><given-names>J.</given-names></name> <name><surname>Lucci</surname><given-names>S.</given-names></name></person-group> (<year>2010</year>). <article-title>Transcription and EHRS. Benefits of a blended approach</article-title>. <source>J. AHIMA</source> <volume>81</volume>, <fpage>36</fpage>&#x2013;<lpage>40</lpage>.</citation></ref>
<ref id="ref15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Casper</surname><given-names>S.</given-names></name> <name><surname>Bailey</surname><given-names>L.</given-names></name> <name><surname>Hunter</surname><given-names>R.</given-names></name> <name><surname>Ezell</surname><given-names>C.</given-names></name> <name><surname>Cabal&#x00E9;</surname><given-names>E.</given-names></name> <name><surname>Gerovitch</surname><given-names>M.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>The AI agent index</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2502.01635</pub-id></citation></ref>
<ref id="ref16"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Chan</surname><given-names>A.</given-names></name> <name><surname>Salganik</surname><given-names>R.</given-names></name> <name><surname>Markelius</surname><given-names>A.</given-names></name> <name><surname>Pang</surname><given-names>C.</given-names></name> <name><surname>Rajkumar</surname><given-names>N.</given-names></name> <name><surname>Krasheninnikov</surname><given-names>D.</given-names></name> <etal/></person-group>. (<year>2023</year>). &#x201C;<article-title>Harms from increasingly agentic algorithmic systems</article-title>&#x201D; in <source>2023 ACM conference on fairness accountability and transparency</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>651</fpage>&#x2013;<lpage>666</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3593013.3594033</pub-id></citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>Q.</given-names></name> <name><surname>Hu</surname><given-names>Y.</given-names></name> <name><surname>Peng</surname><given-names>X.</given-names></name> <name><surname>Xie</surname><given-names>Q.</given-names></name> <name><surname>Jin</surname><given-names>Q.</given-names></name> <name><surname>Gilson</surname><given-names>A.</given-names></name> <etal/></person-group>. (<year>2025b</year>). <article-title>Benchmarking large language models for biomedical natural language processing applications and recommendations</article-title>. <source>Nat. Commun.</source> <volume>16</volume>:<fpage>3280</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41467-025-56989-2</pub-id>, PMID: <pub-id pub-id-type="pmid">40188094</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>B.</given-names></name> <name><surname>Zhang</surname><given-names>Z.</given-names></name> <name><surname>Langren&#x00E9;</surname><given-names>N.</given-names></name> <name><surname>Zhu</surname><given-names>S.</given-names></name></person-group> (<year>2025a</year>). <article-title>Unleashing the potential of prompt engineering for large language models</article-title>. <source>Patterns</source> <volume>6</volume>:<fpage>101260</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.patter.2025.101260</pub-id>, PMID: <pub-id pub-id-type="pmid">40575123</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chew</surname><given-names>B.-H.</given-names></name> <name><surname>Ngiam</surname><given-names>K. Y.</given-names></name></person-group> (<year>2025</year>). <article-title>Artificial intelligence tool development: what clinicians need to know?</article-title> <source>BMC Med.</source> <volume>23</volume>:<fpage>244</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12916-025-04076-0</pub-id>, PMID: <pub-id pub-id-type="pmid">40275334</pub-id></citation></ref>
<ref id="ref20"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Chojnacki</surname><given-names>B.</given-names></name></person-group> (<year>2025</year>). <article-title>The ultimate guide to selecting the right large language model for [Online]</article-title>. <source>dsstream</source>. Available online at: <ext-link xlink:href="https://www.dsstream.com/post/the-ultimate-guide-to-selecting-the-right-large-language-model-for" ext-link-type="uri">https://www.dsstream.com/post/the-ultimate-guide-to-selecting-the-right-large-language-model-for</ext-link> (Accessed May 11, 2025)</citation></ref>
<ref id="ref21"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Chung</surname><given-names>J.</given-names></name> <name><surname>Kamar</surname><given-names>E.</given-names></name> <name><surname>Amershi</surname><given-names>S.</given-names></name></person-group> (<year>2023</year>). &#x201C;<article-title>Increasing diversity while maintaining accuracy: text data generation with large language models and human interventions</article-title>&#x201D; in <source>Proceedings of the 61st annual meeting of the ACL (volume 1: Long papers)</source> (<publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>Association for Computational Linguistics (ACL)</publisher-name>), <fpage>575</fpage>&#x2013;<lpage>593</lpage>. doi: <pub-id pub-id-type="doi">10.18653/v1/2023.acl-long.34</pub-id></citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Coveney</surname><given-names>P. V.</given-names></name> <name><surname>Succi</surname><given-names>S.</given-names></name></person-group> (<year>2025</year>). <article-title>The wall confronting large language models</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2507.19703</pub-id></citation></ref>
<ref id="ref23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Das</surname><given-names>A. B.</given-names></name> <name><surname>Ahmed</surname><given-names>S.</given-names></name> <name><surname>Sakib</surname><given-names>S. K.</given-names></name></person-group> (<year>2025</year>). <article-title>Hallucinations and key information extraction in medical texts: a comprehensive assessment of open-source large language models</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2504.19061</pub-id></citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Daull</surname><given-names>X.</given-names></name> <name><surname>Bellot</surname><given-names>P.</given-names></name> <name><surname>Bruno</surname><given-names>E.</given-names></name> <name><surname>Martin</surname><given-names>V.</given-names></name> <name><surname>Murisasco</surname><given-names>E.</given-names></name></person-group> (<year>2023</year>). <article-title>Complex qa and language models hybrid architectures, survey</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2302.09051</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="other"><person-group person-group-type="author"><collab id="coll1">DeepMind</collab></person-group> (<year>2025</year>). <italic>Alphaevolve: A gemini-powered coding agent for designing advanced algorithms</italic> [Online]. Available online at: <ext-link xlink:href="https://deepmind.google/discover/blog/alphaevolve-a-gemini-powered-coding-agent-for-designing-advanced-algorithms/" ext-link-type="uri">https://deepmind.google/discover/blog/alphaevolve-a-gemini-powered-coding-agent-for-designing-advanced-algorithms/</ext-link> (Accessed May 29, 2025).</citation></ref>
<ref id="ref26"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Dibia</surname><given-names>V.</given-names></name> <name><surname>Chen</surname><given-names>J.</given-names></name> <name><surname>Bansal</surname><given-names>G.</given-names></name> <name><surname>Syed</surname><given-names>S.</given-names></name> <name><surname>Fourney</surname><given-names>A.</given-names></name> <name><surname>Zhu</surname><given-names>E.</given-names></name> <etal/></person-group> (<year>2024</year>). <article-title>Autogen studio: a no-code developer tool for building and debugging multi-agent systems</article-title>, in: <conf-name>Proceedings of the 2024 conference on empirical methods in natural language processing: System demonstrations</conf-name>, (<publisher-loc>Stroudsburg, PA</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>72</fpage>&#x2013;<lpage>79</lpage>. doi: <pub-id pub-id-type="doi">10.18653/v1/2024.emnlp-demo.8</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Doan</surname><given-names>S.</given-names></name> <name><surname>Conway</surname><given-names>M.</given-names></name> <name><surname>Phuong</surname><given-names>T. M.</given-names></name> <name><surname>Ohno-Machado</surname><given-names>L.</given-names></name></person-group> (<year>2014</year>). &#x201C;<article-title>Natural language processing in biomedicine; a unified system architecture overview</article-title>&#x201D; in <source>Clinical bioinformatics. Methods in molecular biology</source>. ed. <person-group person-group-type="editor"><name><surname>Trent</surname><given-names>R.</given-names></name></person-group> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Humana Press</publisher-name>), <fpage>275</fpage>&#x2013;<lpage>294</lpage>.</citation></ref>
<ref id="ref28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Drechsler</surname><given-names>J.</given-names></name> <name><surname>Haensch</surname><given-names>A.-C.</given-names></name></person-group> (<year>2024</year>). <article-title>30 years of synthetic data</article-title>. <source>Stat. Sci.</source> <volume>39</volume>, <fpage>221</fpage>&#x2013;<lpage>242</lpage>. doi: <pub-id pub-id-type="doi">10.1214/24-STS927</pub-id></citation></ref>
<ref id="ref29"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Driehuis</surname><given-names>F.</given-names></name> <name><surname>Woudenberg-Hulleman</surname><given-names>I.</given-names></name> <name><surname>Verhof-van Westing</surname><given-names>I. M.</given-names></name> <name><surname>Geurkink</surname><given-names>H.</given-names></name> <name><surname>Hartstra</surname><given-names>L.</given-names></name> <name><surname>Trouw</surname><given-names>M.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title><italic>Verantwoording en toelichting kngf-richtlijn fysiotherapeutische dossiervoering 2019</italic> [Online]</article-title>. <source>Amersfoort: Koninklijk Nederlands Genootschap voor Fysiotherapie (KNGF).</source> Available online at: <ext-link xlink:href="https://www.kngf.nl/app/uploads/2024/09/fysiotherapeutische-dossiervoering-2019-verantwoording-en-toelichting-versie-1.1.pdf" ext-link-type="uri">https://www.kngf.nl/app/uploads/2024/09/fysiotherapeutische-dossiervoering-2019-verantwoording-en-toelichting-versie-1.1.pdf</ext-link> (Accessed May 11, 2025).</citation></ref>
<ref id="ref30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Eigenschink</surname><given-names>P.</given-names></name> <name><surname>Reutterer</surname><given-names>T.</given-names></name> <name><surname>Vamosi</surname><given-names>S.</given-names></name> <name><surname>Vamosi</surname><given-names>R.</given-names></name> <name><surname>Sun</surname><given-names>C.</given-names></name> <name><surname>Kalcher</surname><given-names>K.</given-names></name></person-group> (<year>2023</year>). <article-title>Deep generative models for synthetic data: a survey</article-title>. <source>IEEE Access</source> <volume>11</volume>, <fpage>47304</fpage>&#x2013;<lpage>47320</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ACCESS.2023.3275134</pub-id></citation></ref>
<ref id="ref31"><citation citation-type="other"><person-group person-group-type="author"><collab id="coll2">European Parliament and Council</collab></person-group>. (<year>2024</year>). Regulation (EU) 2024/1689 of the European parliament and of the council of 13 June 2024 laying down harmonised rules on artificial intelligence and amending certain union legislative acts (artificial intelligence act) [online]. Brussels: Official Journal of the European Union, L, 202. Available online at: <ext-link xlink:href="https://eur-lex.europa.eu/eli/reg/2024/1689/oj/eng" ext-link-type="uri">https://eur-lex.europa.eu/eli/reg/2024/1689/oj/eng</ext-link> (Accessed May 15, 2025).</citation></ref>
<ref id="ref32"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Fu</surname><given-names>Y.</given-names></name> <name><surname>Mai</surname><given-names>L.</given-names></name> <name><surname>Ustiugov</surname><given-names>D.</given-names></name></person-group> (<year>2025</year>). <article-title><italic>AI goes serverless: are systems ready?</italic> [online]</article-title>. <source>Sigops.</source> Available online at: <ext-link xlink:href="https://www.sigops.org/2025/ai-goes-serverless-are-systems-ready/" ext-link-type="uri">https://www.sigops.org/2025/ai-goes-serverless-are-systems-ready/</ext-link> (Accessed May 11, 2025).</citation></ref>
<ref id="ref33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gallegos</surname><given-names>I. O.</given-names></name> <name><surname>Rossi</surname><given-names>R. A.</given-names></name> <name><surname>Barrow</surname><given-names>J.</given-names></name> <name><surname>Tanjim</surname><given-names>M. M.</given-names></name> <name><surname>Kim</surname><given-names>S.</given-names></name> <name><surname>Dernoncourt</surname><given-names>F.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Bias and fairness in large language models: a survey</article-title>. <source>Comput. Linguist.</source> <volume>50</volume>, <fpage>1097</fpage>&#x2013;<lpage>1179</lpage>. doi: <pub-id pub-id-type="doi">10.1162/coli_a_00524</pub-id></citation></ref>
<ref id="ref34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Garg</surname><given-names>M.</given-names></name> <name><surname>Raza</surname><given-names>S.</given-names></name> <name><surname>Rayana</surname><given-names>S.</given-names></name> <name><surname>Liu</surname><given-names>X.</given-names></name> <name><surname>Sohn</surname><given-names>S.</given-names></name></person-group> (<year>2025</year>). <article-title>The rise of small language models in healthcare: a comprehensive survey</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2504.17119</pub-id></citation></ref>
<ref id="ref35"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Goodfellow</surname><given-names>I.</given-names></name> <name><surname>Bengio</surname><given-names>Y.</given-names></name> <name><surname>Courville</surname><given-names>A.</given-names></name></person-group> (<year>2018</year>). <source>Deep learning</source>. <publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>MIT press</publisher-name>.</citation></ref>
<ref id="ref36"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Goyal</surname><given-names>M.</given-names></name> <name><surname>Mahmoud</surname><given-names>Q. H.</given-names></name></person-group> (<year>2024</year>). <article-title>A systematic review of synthetic data generation techniques using generative AI</article-title>. <source>Electronics</source> <volume>13</volume>:<fpage>3509</fpage>. Available online at: <ext-link xlink:href="https://www.mdpi.com/2079-9292/13/17/3509" ext-link-type="uri">https://www.mdpi.com/2079-9292/13/17/3509</ext-link></citation></ref>
<ref id="ref37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gridach</surname><given-names>M.</given-names></name> <name><surname>Nanavati</surname><given-names>J.</given-names></name> <name><surname>Abidine</surname><given-names>K. Z. E.</given-names></name> <name><surname>Mendes</surname><given-names>L.</given-names></name> <name><surname>Mack</surname><given-names>C.</given-names></name></person-group> (<year>2025</year>). <article-title>Agentic ai for scientific discovery: a survey of progress, challenges, and future directions</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2503.08979</pub-id></citation></ref>
<ref id="ref38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>G&#x00FC;nther</surname><given-names>M.</given-names></name> <name><surname>Mohr</surname><given-names>I.</given-names></name> <name><surname>Williams</surname><given-names>D. J.</given-names></name> <name><surname>Wang</surname><given-names>B.</given-names></name> <name><surname>Xiao</surname><given-names>H.</given-names></name></person-group> (<year>2024</year>). <article-title>Late chunking: contextual chunk embeddings using long-context embedding models</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2409.04701</pub-id></citation></ref>
<ref id="ref39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gupta</surname><given-names>S.</given-names></name></person-group> (<year>2025</year>). <article-title>The rise of serverless AI: transforming machine learning deployment</article-title>. <source>Eur. J. Comput. Sci. Inf. Technol.</source> <volume>13</volume>, <fpage>45</fpage>&#x2013;<lpage>67</lpage>. doi: <pub-id pub-id-type="doi">10.37745/ejcsit.2013/vol13n54567</pub-id></citation></ref>
<ref id="ref40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han</surname><given-names>T.</given-names></name> <name><surname>Adams</surname><given-names>L. C.</given-names></name> <name><surname>Papaioannou</surname><given-names>J.-M.</given-names></name> <name><surname>Grundmann</surname><given-names>P.</given-names></name> <name><surname>Oberhauser</surname><given-names>T.</given-names></name> <name><surname>L&#x00F6;ser</surname><given-names>A.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Medalpaca &#x2013; an open-source collection of medical conversational AI models and training data</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2304.08247</pub-id></citation></ref>
<ref id="ref41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Harnad</surname><given-names>S.</given-names></name></person-group> (<year>2025</year>). <article-title>Language writ large: LLMs, chatgpt, meaning, and understanding</article-title>. <source>Front. Artif. Intell.</source> <volume>7</volume>:<fpage>1490698</fpage>. doi: <pub-id pub-id-type="doi">10.3389/frai.2024.1490698</pub-id>, PMID: <pub-id pub-id-type="pmid">40013231</pub-id></citation></ref>
<ref id="ref42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hart</surname><given-names>E. M.</given-names></name> <name><surname>Barmby</surname><given-names>P.</given-names></name> <name><surname>LeBauer</surname><given-names>D.</given-names></name> <name><surname>Michonneau</surname><given-names>F.</given-names></name> <name><surname>Mount</surname><given-names>S.</given-names></name> <name><surname>Mulrooney</surname><given-names>P.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Ten simple rules for digital data storage</article-title>. <source>PLoS Comput. Biol.</source> <volume>12</volume>:<fpage>e1005097</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pcbi.1005097</pub-id>, PMID: <pub-id pub-id-type="pmid">27764088</pub-id></citation></ref>
<ref id="ref43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Haug</surname><given-names>C. J.</given-names></name></person-group> (<year>2018</year>). <article-title>Turning the tables &#x2013; the new european general data protection regulation</article-title>. <source>N. Engl. J. Med.</source> <volume>379</volume>, <fpage>207</fpage>&#x2013;<lpage>209</lpage>. doi: <pub-id pub-id-type="doi">10.1056/NEJMp1806637</pub-id>, PMID: <pub-id pub-id-type="pmid">29874143</pub-id></citation></ref>
<ref id="ref44"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Hechler</surname><given-names>E.</given-names></name> <name><surname>Weihrauch</surname><given-names>M.</given-names></name> <name><surname>Wu</surname><given-names>Y.</given-names></name></person-group> (<year>2023</year>). &#x201C;<article-title>Evolution of data architecture</article-title>&#x201D; in <source>Data fabric and data mesh approaches with AI</source> (<publisher-loc>Berkeley, CA</publisher-loc>: <publisher-name>Apress</publisher-name>), <fpage>3</fpage>&#x2013;<lpage>15</lpage>. doi: <pub-id pub-id-type="doi">10.1007/978-1-4842-9253-2_1</pub-id></citation></ref>
<ref id="ref45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hernandez</surname><given-names>M.</given-names></name> <name><surname>Osorio-Marulanda</surname><given-names>P. A.</given-names></name> <name><surname>Catalina</surname><given-names>M.</given-names></name> <name><surname>Loinaz</surname><given-names>L.</given-names></name> <name><surname>Epelde</surname><given-names>G.</given-names></name> <name><surname>Aginako</surname><given-names>N.</given-names></name></person-group> (<year>2025</year>). <article-title>Comprehensive evaluation framework for synthetic tabular data in health: Fidelity, utility and privacy analysis of generative models with and without privacy guarantees</article-title>. <source>Front. Digit. Health</source> <volume>7</volume>:<fpage>1576290</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fdgth.2025.1576290</pub-id>, PMID: <pub-id pub-id-type="pmid">40343213</pub-id></citation></ref>
<ref id="ref46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hettiarachchi</surname><given-names>I.</given-names></name></person-group> (<year>2025</year>). <article-title>Exploring generative AI agents: architecture, applications, and challenges</article-title>. <source>J. Artif. Intell. Gen. Sci.</source> <volume>8</volume>, <fpage>105</fpage>&#x2013;<lpage>127</lpage>. doi: <pub-id pub-id-type="doi">10.60087/jaigs.v8i1.350</pub-id></citation></ref>
<ref id="ref47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hoofnagle</surname><given-names>C. J.</given-names></name> <name><surname>Van Der Sloot</surname><given-names>B.</given-names></name> <name><surname>Borgesius</surname><given-names>F. Z.</given-names></name></person-group> (<year>2019</year>). <article-title>The European Union general data protection regulation: what it is and what it means</article-title>. <source>Inf. Commun. Technol. Law</source> <volume>28</volume>, <fpage>65</fpage>&#x2013;<lpage>98</lpage>. doi: <pub-id pub-id-type="doi">10.1080/13600834.2019.1573501</pub-id></citation></ref>
<ref id="ref48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname><given-names>Y.</given-names></name> <name><surname>Zhan</surname><given-names>R.</given-names></name> <name><surname>Wong</surname><given-names>D. F.</given-names></name> <name><surname>Chao</surname><given-names>L. S.</given-names></name> <name><surname>Tao</surname><given-names>A.</given-names></name></person-group> (<year>2025</year>). <article-title>Intrinsic model weaknesses: how priming attacks unveil vulnerabilities in large language models</article-title>. <source>Assoc. Comput. Ling.</source>, <fpage>1405</fpage>&#x2013;<lpage>1425</lpage>. doi: <pub-id pub-id-type="doi">10.18653/v1/2025.findings-naacl.77</pub-id></citation></ref>
<ref id="ref49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huerta</surname><given-names>E. A.</given-names></name> <name><surname>Blaiszik</surname><given-names>B.</given-names></name> <name><surname>Brinson</surname><given-names>L. C.</given-names></name> <name><surname>Bouchard</surname><given-names>K. E.</given-names></name> <name><surname>Diaz</surname><given-names>D.</given-names></name> <name><surname>Doglioni</surname><given-names>C.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Fair for AI: an interdisciplinary and international community building perspective</article-title>. <source>Sci Data</source> <volume>10</volume>:<fpage>487</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41597-023-02298-6</pub-id>, PMID: <pub-id pub-id-type="pmid">37495591</pub-id></citation></ref>
<ref id="ref50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ibrahim</surname><given-names>M.</given-names></name> <name><surname>Khalil</surname><given-names>Y. A.</given-names></name> <name><surname>Amirrajab</surname><given-names>S.</given-names></name> <name><surname>Sun</surname><given-names>C.</given-names></name> <name><surname>Breeuwer</surname><given-names>M.</given-names></name> <name><surname>Pluim</surname><given-names>J.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Generative AI for synthetic data across multiple medical modalities: a systematic review of recent developments and challenges</article-title>. <source>Comput. Biol. Med.</source> <volume>189</volume>:<fpage>109834</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compbiomed.2025.109834</pub-id>, PMID: <pub-id pub-id-type="pmid">40023073</pub-id></citation></ref>
<ref id="ref51"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Inoue</surname><given-names>K.</given-names></name></person-group> (<year>2024</year>). <article-title><italic>A practitioner&#x2019;s guide to selecting large language models for your business needs</italic> [Online]</article-title>. <source>Veritone</source>. Available online at: <ext-link xlink:href="https://www.veritone.com/blog/a-practitioners-guide-to-selecting-large-language-models-for-your-business-needs/" ext-link-type="uri">https://www.veritone.com/blog/a-practitioners-guide-to-selecting-large-language-models-for-your-business-needs/</ext-link> (Accessed May 11, 2025].</citation></ref>
<ref id="ref52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jeong</surname><given-names>C.</given-names></name></person-group> (<year>2025</year>). <article-title>Beyond text: implementing multimodal large language model-powered multi-agent systems using a no-code platform</article-title>. <source>J. Intell. Inf. Syst.</source> <volume>31</volume>, <fpage>191</fpage>&#x2013;<lpage>231</lpage>. doi: <pub-id pub-id-type="doi">10.13088/jiis.2025.31.1.191</pub-id></citation></ref>
<ref id="ref53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jin</surname><given-names>M.</given-names></name> <name><surname>Sang-Min</surname><given-names>C.</given-names></name> <name><surname>Gun-Woo</surname><given-names>K.</given-names></name></person-group> (<year>2025</year>). <article-title>Comcare: a collaborative ensemble framework for context-aware medical named entity recognition and relation extraction</article-title>. <source>Electronics</source> <volume>14</volume>:<fpage>328</fpage>. doi: <pub-id pub-id-type="doi">10.3390/electronics14020328</pub-id></citation></ref>
<ref id="ref54"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kaplan</surname><given-names>J.</given-names></name> <name><surname>McCandlish</surname><given-names>S.</given-names></name> <name><surname>Henighan</surname><given-names>T.</given-names></name> <name><surname>Brown</surname><given-names>T. B.</given-names></name> <name><surname>Chess</surname><given-names>B.</given-names></name> <name><surname>Child</surname><given-names>R.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Scaling laws for neural language models</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2001.08361</pub-id></citation></ref>
<ref id="ref55"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Karpathy</surname><given-names>A.</given-names></name></person-group> (<year>2025</year>). <italic>Vibe coding</italic> [Online]. Available online at: <ext-link xlink:href="https://x.com/karpathy/status/1886192184808149383" ext-link-type="uri">https://x.com/karpathy/status/1886192184808149383</ext-link> (Accessed May 12, 2025).</citation></ref>
<ref id="ref56"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname><given-names>H.</given-names></name> <name><surname>Hwang</surname><given-names>H.</given-names></name> <name><surname>Lee</surname><given-names>J.</given-names></name> <name><surname>Park</surname><given-names>S.</given-names></name> <name><surname>Kim</surname><given-names>D.</given-names></name> <name><surname>Lee</surname><given-names>T.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Small language models learn enhanced reasoning skills from medical textbooks</article-title>. <source>npj Digital Medicine</source> <volume>8</volume>:<fpage>240</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41746-025-01653-8</pub-id>, PMID: <pub-id pub-id-type="pmid">40316765</pub-id></citation></ref>
<ref id="ref57"><citation citation-type="other"><person-group person-group-type="author"><collab id="coll3">Knaw</collab><collab id="coll301">NFU, NWO, TO Federatie, Vereniging Hogescholen</collab><collab id="coll302">Vsnu</collab></person-group> (<year>2018</year>). Nederlandse gedragscode wetenschappelijke integriteit. V1. doi: <pub-id pub-id-type="doi">10.17026/dans-2cj-nvwu</pub-id></citation></ref>
<ref id="ref58"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Kochanowska</surname><given-names>M.</given-names></name> <name><surname>Gagliardi</surname><given-names>W. R.</given-names></name><collab id="coll4">with reference to Jonathan, B</collab></person-group> (<year>2022</year>). &#x201C;<article-title>The double diamond model: in pursuit of simplicity and flexibility</article-title>&#x201D; in <source>Perspectives on design ii</source>. eds. <person-group person-group-type="editor"><name><surname>Raposo</surname><given-names>D.</given-names></name> <name><surname>Neves</surname><given-names>J.</given-names></name> <name><surname>Silva</surname><given-names>J.</given-names></name></person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>19</fpage>&#x2013;<lpage>32</lpage>.</citation></ref>
<ref id="ref59"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krishnakumar</surname><given-names>A.</given-names></name> <name><surname>Ogras</surname><given-names>U.</given-names></name> <name><surname>Marculescu</surname><given-names>R.</given-names></name> <name><surname>Kishinevsky</surname><given-names>M.</given-names></name> <name><surname>Mudge</surname><given-names>T.</given-names></name></person-group> (<year>2023</year>). <article-title>Domain-specific architectures: research problems and promising approaches</article-title>. <source>ACM Trans. Embed. Comput. Syst.</source> <volume>22</volume>, <fpage>1</fpage>&#x2013;<lpage>26</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3563946</pub-id></citation></ref>
<ref id="ref60"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Kumar</surname><given-names>A.</given-names></name></person-group> (<year>2023</year>). 5 best Flowise alternatives in 2024 to build AI agents [online]. Available online at: <ext-link xlink:href="https://blog.fabrichq.ai/6-best-flowise-alternatives-in-2024-to-build-ai-agents-8af9cb572449" ext-link-type="uri">https://blog.fabrichq.ai/6-best-flowise-alternatives-in-2024-to-build-ai-agents-8af9cb572449</ext-link> (Accessed March 24, 2025).</citation></ref>
<ref id="ref61"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>LeCun</surname><given-names>Y.</given-names></name> <name><surname>Bengio</surname><given-names>Y.</given-names></name> <name><surname>Hinton</surname><given-names>G.</given-names></name></person-group> (<year>2015</year>). <article-title>Deep learning</article-title>. <source>Nature</source> <volume>521</volume>, <fpage>436</fpage>&#x2013;<lpage>444</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nature14539</pub-id>, PMID: <pub-id pub-id-type="pmid">26017442</pub-id></citation></ref>
<ref id="ref62"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname><given-names>J.</given-names></name> <name><surname>Yoon</surname><given-names>W.</given-names></name> <name><surname>Kim</surname><given-names>S.</given-names></name> <name><surname>Kim</surname><given-names>D.</given-names></name> <name><surname>Kim</surname><given-names>S.</given-names></name> <name><surname>So</surname><given-names>C. H.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Biobert: a pre-trained biomedical language representation model for biomedical text mining</article-title>. <source>Bioinformatics</source> <volume>36</volume>, <fpage>1234</fpage>&#x2013;<lpage>1240</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btz682</pub-id>, PMID: <pub-id pub-id-type="pmid">31501885</pub-id></citation></ref>
<ref id="ref63"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Li</surname><given-names>Z.</given-names></name> <name><surname>Zhu</surname><given-names>H.</given-names></name> <name><surname>Lu</surname><given-names>Z.</given-names></name> <name><surname>Yin</surname><given-names>M.</given-names></name></person-group> (<year>2023</year>). &#x201C;<article-title>Synthetic data generation with large language models for text classification: potential and limitations</article-title>&#x201D; in <source>Proceedings of the 2023 conference on empirical methods in natural language processing</source>. eds. <person-group person-group-type="editor"><name><surname>Bouamor</surname><given-names>H.</given-names></name> <name><surname>Pino</surname><given-names>J.</given-names></name> <name><surname>Bali</surname><given-names>K.</given-names></name></person-group> (<publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>Association for Computational Linguistics (ACL)</publisher-name>), <fpage>10443</fpage>&#x2013;<lpage>10461</lpage>. doi: <pub-id pub-id-type="doi">10.18653/v1/2023.emnlp-main.647</pub-id></citation></ref>
<ref id="ref64"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Little</surname><given-names>R. J.</given-names></name></person-group> (<year>1993</year>) <article-title>Statistical analysis of masked data</article-title>. <source>J. Off. Stat.</source>, <volume>9</volume>: <fpage>407</fpage>&#x2013;<lpage>426</lpage>. Available online at: <ext-link xlink:href="https://www.imstat.org/publications/sts/sts_39_2/sts_39_2.pdf" ext-link-type="uri">https://www.imstat.org/publications/sts/sts_39_2/sts_39_2.pdf</ext-link></citation></ref>
<ref id="ref65"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>Y.</given-names></name> <name><surname>Acharya</surname><given-names>U. R.</given-names></name> <name><surname>Tan</surname><given-names>J. H.</given-names></name></person-group> (<year>2025</year>). <article-title>Preserving privacy in healthcare: a systematic review of deep learning approaches for synthetic data generation</article-title>. <source>Comput. Methods Prog. Biomed.</source> <volume>260</volume>:<fpage>108571</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cmpb.2024.108571</pub-id>, PMID: <pub-id pub-id-type="pmid">39742693</pub-id></citation></ref>
<ref id="ref66"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Loni</surname><given-names>M.</given-names></name> <name><surname>Poursalim</surname><given-names>F.</given-names></name> <name><surname>Asadi</surname><given-names>M.</given-names></name> <name><surname>Gharehbaghi</surname><given-names>A.</given-names></name></person-group> (<year>2025</year>). <article-title>A review on generative AI models for synthetic medical text, time series, and longitudinal data</article-title>. <source>npj Digit. Med.</source> <volume>8</volume>:<fpage>281</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41746-024-01409-w</pub-id></citation></ref>
<ref id="ref67"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Mayo</surname><given-names>M.</given-names></name></person-group> (<year>2025</year>). <italic>Feel the vibe: Why AI-dependent coding isn&#x2019;t the enemy (or is it?)</italic> [Online]. Available online at <ext-link xlink:href="https://www.kdnuggets.com/feel-the-vibe-why-ai-dependent-coding-isnt-the-enemy-or-is-it" ext-link-type="uri">https://www.kdnuggets.com/feel-the-vibe-why-ai-dependent-coding-isnt-the-enemy-or-is-it</ext-link> (Accessed May 15, 2025).</citation></ref>
<ref id="ref68"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mei</surname><given-names>L.</given-names></name> <name><surname>Yao</surname><given-names>J.</given-names></name> <name><surname>Ge</surname><given-names>Y.</given-names></name> <name><surname>Wang</surname><given-names>Y.</given-names></name> <name><surname>Bi</surname><given-names>B.</given-names></name> <name><surname>Cai</surname><given-names>Y.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>A survey of context engineering for large language models</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2507.13334</pub-id></citation></ref>
<ref id="ref69"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Meinke</surname><given-names>A.</given-names></name> <name><surname>Schoen</surname><given-names>B.</given-names></name> <name><surname>Scheurer</surname><given-names>J.</given-names></name> <name><surname>Balesni</surname><given-names>M.</given-names></name> <name><surname>Shah</surname><given-names>R.</given-names></name> <name><surname>Hobbhahn</surname><given-names>M.</given-names></name></person-group> (<year>2024</year>). <article-title>Frontier models are capable of in-context scheming</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2412.04984</pub-id></citation></ref>
<ref id="ref70"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Meng</surname><given-names>X.-L.</given-names></name></person-group> (<year>2021</year>). <article-title>Building data science infrastructures and infrastructural data science</article-title>. <source>Harv. Data Sci. Rev.</source> <volume>3</volume>. doi: <pub-id pub-id-type="doi">10.1162/99608f92.abfa0e70</pub-id></citation></ref>
<ref id="ref71"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Meystre</surname><given-names>S. M.</given-names></name> <name><surname>Lovis</surname><given-names>C.</given-names></name> <name><surname>Burkle</surname><given-names>T.</given-names></name> <name><surname>Tognola</surname><given-names>G.</given-names></name> <name><surname>Budrionis</surname><given-names>A.</given-names></name> <name><surname>Lehmann</surname><given-names>C. U.</given-names></name></person-group> (<year>2017</year>). <article-title>Clinical data reuse or secondary use: current status and potential future progress</article-title>. <source>Yearb. Med. Inform.</source> <volume>26</volume>, <fpage>38</fpage>&#x2013;<lpage>52</lpage>. doi: <pub-id pub-id-type="doi">10.15265/IY-2017-007</pub-id>, PMID: <pub-id pub-id-type="pmid">28480475</pub-id></citation></ref>
<ref id="ref72"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mons</surname><given-names>B.</given-names></name> <name><surname>Neylon</surname><given-names>C.</given-names></name> <name><surname>Velterop</surname><given-names>J.</given-names></name> <name><surname>Dumontier</surname><given-names>M.</given-names></name> <name><surname>da Silva Santos</surname><given-names>L. O. B.</given-names></name> <name><surname>Wilkinson</surname><given-names>M. D.</given-names></name></person-group> (<year>2017</year>). <article-title>Cloudy, increasingly fair; revisiting the fair data guiding principles for the European open science cloud</article-title>. <source>Inf. Serv. Use</source> <volume>37</volume>, <fpage>49</fpage>&#x2013;<lpage>56</lpage>. doi: <pub-id pub-id-type="doi">10.3233/isu-170824</pub-id></citation></ref>
<ref id="ref73"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Morris</surname><given-names>J. X.</given-names></name> <name><surname>Sitawarin</surname><given-names>C.</given-names></name> <name><surname>Guo</surname><given-names>C.</given-names></name> <name><surname>Kokhlikyan</surname><given-names>N.</given-names></name> <name><surname>Suh</surname><given-names>G. E.</given-names></name> <name><surname>Rush</surname><given-names>A. M.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>How much do language models memorize?</article-title> <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2505.24832</pub-id></citation></ref>
<ref id="ref74"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Murtaza</surname><given-names>H.</given-names></name> <name><surname>Ahmed</surname><given-names>M.</given-names></name> <name><surname>Khan</surname><given-names>N. F.</given-names></name> <name><surname>Murtaza</surname><given-names>G.</given-names></name> <name><surname>Zafar</surname><given-names>S.</given-names></name> <name><surname>Bano</surname><given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>Synthetic data generation: state of the art in health care domain</article-title>. <source>Comput Sci Rev</source> <volume>48</volume>:<fpage>100546</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cosrev.2023.100546</pub-id></citation></ref>
<ref id="ref75"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Negro-Calduch</surname><given-names>E.</given-names></name> <name><surname>Azzopardi-Muscat</surname><given-names>N.</given-names></name> <name><surname>Krishnamurthy</surname><given-names>R. S.</given-names></name> <name><surname>Novillo-Ortiz</surname><given-names>D.</given-names></name></person-group> (<year>2021</year>). <article-title>Technological progress in electronic health record system optimization: systematic review of systematic literature reviews</article-title>. <source>Int. J. Med. Inform.</source> <volume>152</volume>:<fpage>104507</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2021.104507</pub-id>, PMID: <pub-id pub-id-type="pmid">34049051</pub-id></citation></ref>
<ref id="ref76"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nijor</surname><given-names>S.</given-names></name> <name><surname>Rallis</surname><given-names>G.</given-names></name> <name><surname>Lad</surname><given-names>N.</given-names></name> <name><surname>Gokcen</surname><given-names>E.</given-names></name></person-group> (<year>2022</year>). <article-title>Patient safety issues from information overload in electronic medical records</article-title>. <source>J. Patient Saf.</source> <volume>18</volume>, <fpage>e999</fpage>&#x2013;<lpage>e1003</lpage>. doi: <pub-id pub-id-type="doi">10.1097/PTS.0000000000001002</pub-id>, PMID: <pub-id pub-id-type="pmid">35985047</pub-id></citation></ref>
<ref id="ref77"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Novikov</surname><given-names>A.</given-names></name> <name><surname>Vu</surname><given-names>N.</given-names></name> <name><surname>Eisenberger</surname><given-names>M.</given-names></name> <name><surname>Dupont</surname><given-names>E.</given-names></name> <name><surname>Huang</surname><given-names>P.</given-names></name> <name><surname>Wagner</surname><given-names>A.Z.</given-names></name> <etal/></person-group> (<year>2025</year>). Alphaevolve: A coding agent for scientific and algorithmic discovery. [online] [Preprint]. Available online at: <ext-link xlink:href="https://storage.googleapis.com/deepmind-media/DeepMind.com/Blog/alphaevolve-a-gemini-powered-coding-agent-for-designing-advanced-algorithms/AlphaEvolve.pdf" ext-link-type="uri">https://storage.googleapis.com/deepmind-media/DeepMind.com/Blog/alphaevolve-a-gemini-powered-coding-agent-for-designing-advanced-algorithms/AlphaEvolve.pdf</ext-link> (Accessed May 09, 2025).</citation></ref>
<ref id="ref78"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ohse</surname><given-names>J.</given-names></name> <name><surname>Had&#x017E;i&#x0107;</surname><given-names>B.</given-names></name> <name><surname>Mohammed</surname><given-names>P.</given-names></name> <name><surname>Peperkorn</surname><given-names>N.</given-names></name> <name><surname>Danner</surname><given-names>M.</given-names></name> <name><surname>Yorita</surname><given-names>A.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Zero-shot strike: testing the generalisation capabilities of out-of-the-box llm models for depression detection</article-title>. <source>Comput. Speech Lang.</source> <volume>88</volume>:<fpage>101663</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.csl.2024.101663</pub-id></citation></ref>
<ref id="ref79"><citation citation-type="other"><person-group person-group-type="author"><collab id="coll5">OpenAI</collab></person-group>. (<year>2025</year>). Introducing gpt-4.1 in the api [Online]. Available online at: <ext-link xlink:href="https://openai.com/index/gpt-4-1/" ext-link-type="uri">https://openai.com/index/gpt-4-1/</ext-link> (Accessed May 13, 2025).</citation></ref>
<ref id="ref80"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Orru</surname><given-names>G.</given-names></name> <name><surname>Piarulli</surname><given-names>A.</given-names></name> <name><surname>Conversano</surname><given-names>C.</given-names></name> <name><surname>Gemignani</surname><given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>Human-like problem-solving abilities in large language models using chatgpt</article-title>. <source>Front Artif Intell</source> <volume>6</volume>:<fpage>1199350</fpage>. doi: <pub-id pub-id-type="doi">10.3389/frai.2023.1199350</pub-id>, PMID: <pub-id pub-id-type="pmid">37293238</pub-id></citation></ref>
<ref id="ref81"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Park</surname><given-names>J. S.</given-names></name> <name><surname>O'Brien</surname><given-names>J.</given-names></name> <name><surname>Cai</surname><given-names>C. J.</given-names></name> <name><surname>Morris</surname><given-names>M. R.</given-names></name> <name><surname>Liang</surname><given-names>P.</given-names></name> <name><surname>Bernstein</surname><given-names>M. S.</given-names></name></person-group> (<year>2023</year>). <article-title>Generative agents: interactive simulacra of human behavior</article-title>, in: <conf-name>Proceedings of the 36th annual ACM symposium on user Interface software and technology (UIST '23)</conf-name>, <publisher-loc>San Francisco, CA</publisher-loc>: <publisher-name>ACM</publisher-name>, <fpage>1</fpage>&#x2013;<lpage>22</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3586183.3606763</pub-id></citation></ref>
<ref id="ref82"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Peeperkorn</surname><given-names>M.</given-names></name> <name><surname>Kouwenhoven</surname><given-names>T.</given-names></name> <name><surname>Brown</surname><given-names>D.</given-names></name> <name><surname>Jordanous</surname><given-names>A.</given-names></name></person-group> (<year>2024</year>). <article-title>Is temperature the creativity parameter of large language models?</article-title>, in: <conf-name>Proceedings of the 15th international conference on computational creativity (ICCC&#x2019;24), (Coimbra, Portugal: Association for Computational Creativity)</conf-name>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2405.00492</pub-id></citation></ref>
<ref id="ref83"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pezoulas</surname><given-names>V. C.</given-names></name> <name><surname>Zaridis</surname><given-names>D. I.</given-names></name> <name><surname>Mylona</surname><given-names>E.</given-names></name> <name><surname>Androutsos</surname><given-names>C.</given-names></name> <name><surname>Apostolidis</surname><given-names>K.</given-names></name> <name><surname>Tachos</surname><given-names>N. S.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Synthetic data generation methods in healthcare: a review on open-source tools and methods</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>23</volume>, <fpage>2892</fpage>&#x2013;<lpage>2910</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.csbj.2024.07.005</pub-id>, PMID: <pub-id pub-id-type="pmid">39108677</pub-id></citation></ref>
<ref id="ref84"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Piccialli</surname><given-names>F.</given-names></name> <name><surname>Chiaro</surname><given-names>D.</given-names></name> <name><surname>Sarwar</surname><given-names>S.</given-names></name> <name><surname>Cerciello</surname><given-names>D.</given-names></name> <name><surname>Qi</surname><given-names>P.</given-names></name> <name><surname>Mele</surname><given-names>V.</given-names></name></person-group> (<year>2025</year>). <article-title>Agentai: a comprehensive survey on autonomous agents in distributed AI for industry 4.0</article-title>. <source>Expert Syst. Appl.</source> <volume>291</volume>:<fpage>128404</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.eswa.2025.128404</pub-id></citation></ref>
<ref id="ref85"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Post</surname><given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>A call for clarity in reporting bleu scores</article-title>, in: <conf-name>Proceedings of the 3rd conference on machine translation: Research papers</conf-name>, (<publisher-loc>Stroudsburg, PA</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>186</fpage>&#x2013;<lpage>191</lpage>. doi: <pub-id pub-id-type="doi">10.18653/v1/W18-6319</pub-id></citation></ref>
<ref id="ref86"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Priebe</surname><given-names>T.</given-names></name> <name><surname>Neumaier</surname><given-names>S.</given-names></name> <name><surname>Markus</surname><given-names>S.</given-names></name></person-group> (<year>2021</year>). "<article-title>Finding your way through the jungle of big data architectures</article-title>", in: <conf-name>2021 IEEE international conference on big data (big data)</conf-name>, (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>5994</fpage>&#x2013;<lpage>5996</lpage>. doi: <pub-id pub-id-type="doi">10.1109/BigData52589.2021.9671862</pub-id></citation></ref>
<ref id="ref87"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qiu</surname><given-names>X.</given-names></name> <name><surname>Wang</surname><given-names>H.</given-names></name> <name><surname>Tan</surname><given-names>X.</given-names></name> <name><surname>Qu</surname><given-names>C.</given-names></name> <name><surname>Xiong</surname><given-names>Y.</given-names></name> <name><surname>Cheng</surname><given-names>Y.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Towards collaborative intelligence: propagating intentions and reasoning for multi-agent coordination with large language models</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2407.12532</pub-id></citation></ref>
<ref id="ref88"><citation citation-type="other"><person-group person-group-type="author"><collab id="coll6">QuantSpark</collab></person-group>. (<year>2023</year>). Choosing the right llm: A guide for decision-makers. Available online at: <ext-link xlink:href="https://quantspark.ai/blogs/2023/10/31/ai-comparison-guide" ext-link-type="uri">https://quantspark.ai/blogs/2023/10/31/ai-comparison-guide</ext-link> (Accessed May 11, 2025).</citation></ref>
<ref id="ref89"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Reddy</surname><given-names>S.</given-names></name></person-group> (<year>2024</year>). <article-title>Generative AI in healthcare: an implementation science informed translational path on application, integration and governance</article-title>. <source>Implement. Sci.</source> <volume>19</volume>:<fpage>27</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13012-024-01357-9</pub-id>, PMID: <pub-id pub-id-type="pmid">38491544</pub-id></citation></ref>
<ref id="ref90"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Rubin</surname><given-names>D. B.</given-names></name></person-group> (<year>1993</year>). <article-title>Statistical disclosure limitation</article-title>. <source>J. Off. Stat.</source> <volume>9</volume>, <fpage>461</fpage>&#x2013;<lpage>468</lpage>. Available online at: <ext-link xlink:href="https://ecommons.cornell.edu/server/api/core/bitstreams/dd0b63ff-4494-4491-96ba-a69811563dee/content" ext-link-type="uri">https://ecommons.cornell.edu/server/api/core/bitstreams/dd0b63ff-4494-4491-96ba-a69811563dee/content</ext-link></citation></ref>
<ref id="ref91"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Ruczynski</surname><given-names>K.</given-names></name></person-group> (<year>2024</year>). <article-title>Compare llms: a guide to finding the best large language models [online]</article-title>. <source>Word</source>. Available online at: <ext-link xlink:href="https://www.wordware.ai/blog/compare-llms-a-guide-to-finding-the-best-large-language-models" ext-link-type="uri">https://www.wordware.ai/blog/compare-llms-a-guide-to-finding-the-best-large-language-models</ext-link> (Accessed May 11, 2025).</citation></ref>
<ref id="ref92"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rujas</surname><given-names>M.</given-names></name> <name><surname>Gomez</surname><given-names>M.</given-names></name> <name><surname>Del Moral Herranz</surname><given-names>R.</given-names></name> <name><surname>Fico</surname><given-names>G.</given-names></name> <name><surname>Merino-Barbancho</surname><given-names>B.</given-names></name></person-group> (<year>2025</year>). <article-title>Synthetic data generation in healthcare: a scoping review of reviews on domains, motivations, and future applications</article-title>. <source>Int. J. Med. Inform.</source> <volume>195</volume>:<fpage>105763</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ijmedinf.2024.105763</pub-id>, PMID: <pub-id pub-id-type="pmid">39719743</pub-id></citation></ref>
<ref id="ref93"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sai</surname><given-names>S.</given-names></name> <name><surname>Gaur</surname><given-names>A.</given-names></name> <name><surname>Sai</surname><given-names>R.</given-names></name> <name><surname>Chamola</surname><given-names>V.</given-names></name> <name><surname>Guizani</surname><given-names>M.</given-names></name> <name><surname>Rodrigues</surname><given-names>J. J. P. C.</given-names></name></person-group> (<year>2024</year>). <article-title>Generative AI for transformative healthcare: a comprehensive study of emerging models, applications, case studies, and limitations</article-title>. <source>IEEE Access</source> <volume>12</volume>, <fpage>31078</fpage>&#x2013;<lpage>31106</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ACCESS.2024.3367715</pub-id></citation></ref>
<ref id="ref94"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Scherr</surname><given-names>R.</given-names></name> <name><surname>Spina</surname><given-names>A.</given-names></name> <name><surname>Dao</surname><given-names>A.</given-names></name> <name><surname>Andalib</surname><given-names>S.</given-names></name> <name><surname>Halaseh</surname><given-names>F. F.</given-names></name> <name><surname>Blair</surname><given-names>S.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Novel evaluation metric and quantified performance of chatgpt-4 patient management simulations for early clinical education: experimental study</article-title>. <source>JMIR Form. Res.</source> <volume>9</volume>:<fpage>e66478</fpage>. doi: <pub-id pub-id-type="doi">10.2196/66478</pub-id>, PMID: <pub-id pub-id-type="pmid">40013991</pub-id></citation></ref>
<ref id="ref95"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Schick</surname><given-names>T.</given-names></name> <name><surname>Sch&#x00FC;tze</surname><given-names>H.</given-names></name></person-group> (<year>2021</year>). "<article-title>It&#x2019;s not just size that matters: small language models are also few-shot learners</article-title>", in: <conf-name>Proceedings of the 2021 conference of the north American chapter of the Association for Computational Linguistics (NAACL): Human language technologies</conf-name>, (<publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>2339</fpage>&#x2013;<lpage>2352</lpage>. doi: <pub-id pub-id-type="doi">10.18653/v1/2021.naacl-main.185</pub-id></citation></ref>
<ref id="ref96"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schneider</surname><given-names>J.</given-names></name></person-group> (<year>2025</year>). <article-title>Generative to agentic AI: survey, conceptualization, and challenges</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2504.18875</pub-id></citation></ref>
<ref id="ref97"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schut</surname><given-names>M. C.</given-names></name> <name><surname>Luik</surname><given-names>T. T.</given-names></name> <name><surname>Vagliano</surname><given-names>I.</given-names></name> <name><surname>Rios</surname><given-names>M.</given-names></name> <name><surname>Helsper</surname><given-names>C. W.</given-names></name> <name><surname>van Asselt</surname><given-names>K. M.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Artificial intelligence for early detection of lung cancer in gps' clinical notes: a retrospective observational cohort study</article-title>. <source>Br. J. Gen. Pract.</source> <volume>75</volume>, <fpage>e316</fpage>&#x2013;<lpage>e322</lpage>. doi: <pub-id pub-id-type="doi">10.3399/BJGP.2023.0489</pub-id>, PMID: <pub-id pub-id-type="pmid">40044183</pub-id></citation></ref>
<ref id="ref98"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Seinen</surname><given-names>T. M.</given-names></name> <name><surname>Kors</surname><given-names>J. A.</given-names></name> <name><surname>van mulligen</surname><given-names>E. M.</given-names></name> <name><surname>Rijnbeek</surname><given-names>P. R.</given-names></name></person-group> (<year>2024</year>). <article-title>Structured codes and free-text notes: measuring information complementarity in electronic health records</article-title>. <source>medRxiv [Preprint].</source> doi: <pub-id pub-id-type="doi">10.1101/2024.10.28.24316294</pub-id></citation></ref>
<ref id="ref99"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sengupta</surname><given-names>A.</given-names></name> <name><surname>Goel</surname><given-names>Y.</given-names></name> <name><surname>Chakraborty</surname><given-names>T.</given-names></name></person-group> (<year>2025</year>). <article-title>How to upscale neural networks with scaling law? A survey and practical guidelines</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2502.12051</pub-id></citation></ref>
<ref id="ref100"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shannon</surname><given-names>C. E.</given-names></name></person-group> (<year>1948</year>). <article-title>A mathematical theory of communication</article-title>. <source>Bcl. Syst. Tech. J.</source> <volume>27</volume>, <fpage>379</fpage>&#x2013;<lpage>423</lpage>. doi: <pub-id pub-id-type="doi">10.1002/j.1538-7305.1948.tb01338.x</pub-id></citation></ref>
<ref id="ref101"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shojaee</surname><given-names>P.</given-names></name> <name><surname>Mirzadeh</surname><given-names>I.</given-names></name> <name><surname>Alizadeh</surname><given-names>K.</given-names></name> <name><surname>Horton</surname><given-names>M.</given-names></name> <name><surname>Bengio</surname><given-names>S.</given-names></name> <name><surname>Farajtabar</surname><given-names>M.</given-names></name></person-group> (<year>2025</year>). <article-title>The illusion of thinking: understanding the strengths and limitations of reasoning models via the lens of problem complexity</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2506.06941</pub-id></citation></ref>
<ref id="ref102"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Singhal</surname><given-names>K.</given-names></name> <name><surname>Azizi</surname><given-names>S.</given-names></name> <name><surname>Tu</surname><given-names>T.</given-names></name> <name><surname>Mahdavi</surname><given-names>S. S.</given-names></name> <name><surname>Wei</surname><given-names>J.</given-names></name> <name><surname>Chung</surname><given-names>H. W.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Large language models encode clinical knowledge</article-title>. <source>Nature</source> <volume>620</volume>, <fpage>172</fpage>&#x2013;<lpage>180</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41586-023-06291-2</pub-id>, PMID: <pub-id pub-id-type="pmid">37438534</pub-id></citation></ref>
<ref id="ref103"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smolyak</surname><given-names>D.</given-names></name> <name><surname>Bjarnad&#x00F3;ttir</surname><given-names>M. V.</given-names></name> <name><surname>Crowley</surname><given-names>K.</given-names></name> <name><surname>Agarwal</surname><given-names>R.</given-names></name></person-group> (<year>2024</year>). <article-title>Large language models and synthetic health data: progress and prospects</article-title>. <source>JAMIA Open</source> <volume>7</volume>:<fpage>ooae114</fpage>. doi: <pub-id pub-id-type="doi">10.1093/jamiaopen/ooae114</pub-id>, PMID: <pub-id pub-id-type="pmid">39464796</pub-id></citation></ref>
<ref id="ref104"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Steinkamp</surname><given-names>J.</given-names></name> <name><surname>Kantrowitz</surname><given-names>J. J.</given-names></name> <name><surname>Airan-Javia</surname><given-names>S.</given-names></name></person-group> (<year>2022</year>). <article-title>Prevalence and sources of duplicate information in the electronic medical record</article-title>. <source>JAMA Netw. Open</source> <volume>5</volume>:<fpage>e2233348</fpage>. doi: <pub-id pub-id-type="doi">10.1001/jamanetworkopen.2022.33348</pub-id>, PMID: <pub-id pub-id-type="pmid">36156143</pub-id></citation></ref>
<ref id="ref105"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Suominen</surname><given-names>H.</given-names></name> <name><surname>Salanter&#x00E4;</surname><given-names>S.</given-names></name> <name><surname>Velupillai</surname><given-names>S.</given-names></name> <name><surname>Chapman</surname><given-names>W. W.</given-names></name> <name><surname>Savova</surname><given-names>G.</given-names></name> <name><surname>Elhadad</surname><given-names>N.</given-names></name> <etal/></person-group> (<year>2013</year>) <article-title>Overview of the share/clef ehealth evaluation lab 2013</article-title> <conf-name>Information access evaluation. Multilinguality, multimodality, and visualization: 4th international conference of the CLEF initiative</conf-name> <publisher-loc>Heidelberg, Germany</publisher-loc> <publisher-name>Springer</publisher-name>, <fpage>212</fpage>&#x2013;<lpage>231</lpage>, doi: <pub-id pub-id-type="doi">10.1007/978-3-642-40802-1_17</pub-id></citation></ref>
<ref id="ref106"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Swart</surname><given-names>N. M.</given-names></name> <name><surname>Apeldoorn</surname><given-names>A. T.</given-names></name> <name><surname>Conijn</surname><given-names>D.</given-names></name> <name><surname>Meerhoff</surname><given-names>G. A.</given-names></name> <name><surname>Ostelo</surname><given-names>R. W. J. G.</given-names></name></person-group> (<year>2021</year>). <article-title><italic>Kngf-richtlijn lage rugpijn en lumbosacraal radiculair syndroom</italic> [Online]</article-title>. <source>Amersfoort/ Utrecht: Koninklijk Nederlands Genootschap voor Fysiotherapie (KNGF) &#x0026; Vereniging van Oefentherapeuten Cesar en Mensendieck</source>. Available online at: <ext-link xlink:href="https://vvocm.nl/Portals/2/kngf_richtlijn_lage_rugpijn_en_lrs_2021_verantwoording.pdf" ext-link-type="uri">https://vvocm.nl/Portals/2/kngf_richtlijn_lage_rugpijn_en_lrs_2021_verantwoording.pdf</ext-link> (Accessed May 15, 2025).</citation></ref>
<ref id="ref107"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thandla</surname><given-names>S. R.</given-names></name> <name><surname>Armstrong</surname><given-names>G. Q.</given-names></name> <name><surname>Menon</surname><given-names>A.</given-names></name> <name><surname>Shah</surname><given-names>A.</given-names></name> <name><surname>Gueye</surname><given-names>D. L.</given-names></name> <name><surname>Harb</surname><given-names>C.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Comparing new tools of artificial intelligence to the authentic intelligence of our global health students</article-title>. <source>BioData Min.</source> <volume>17</volume>:<fpage>58</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13040-024-00408-7</pub-id>, PMID: <pub-id pub-id-type="pmid">39696442</pub-id></citation></ref>
<ref id="ref108"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tischendorf</surname><given-names>T.</given-names></name> <name><surname>Hinsche</surname><given-names>L.</given-names></name> <name><surname>Hasseler</surname><given-names>M.</given-names></name> <name><surname>Schaal</surname><given-names>T.</given-names></name></person-group> (<year>2025</year>). <article-title>Genai in nursing and clinical practice: a rapid review of applications and challenges</article-title>. <source>J. Public Health (Berl.).</source> doi: <pub-id pub-id-type="doi">10.1007/s10389-025-02523-z</pub-id></citation></ref>
<ref id="ref109"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Tuulos</surname><given-names>V.</given-names></name></person-group> (<year>2022</year>). <source>Effective data science infrastructure: How to make data scientists productive</source>. <publisher-loc>Shelter Island, NY</publisher-loc>: <publisher-name>Manning</publisher-name>.</citation></ref>
<ref id="ref110"><citation citation-type="other"><person-group person-group-type="author"><collab id="coll7">U.S.-National-Library-of-Medicine</collab></person-group>. (<year>2024</year>). Unified medical language system (UMLS) [Online]. Available online at: <ext-link xlink:href="https://www.nlm.nih.gov/research/umls/index.html" ext-link-type="uri">https://www.nlm.nih.gov/research/umls/index.html</ext-link> (Accessed May 15, 2025).</citation></ref>
<ref id="ref111"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Ulfsnes</surname><given-names>R.</given-names></name> <name><surname>Moe</surname><given-names>N. B.</given-names></name> <name><surname>Stray</surname><given-names>V.</given-names></name> <name><surname>Skarpen</surname><given-names>M.</given-names></name></person-group> (<year>2024</year>). &#x201C;<article-title>Transforming software development with generative AI: empirical insights on collaboration and workflow</article-title>&#x201D; in <source>Generative AI for effective software development</source>. eds. <person-group person-group-type="editor"><name><surname>Nguyen-Duc</surname><given-names>A.</given-names></name> <name><surname>Abrahamsson</surname><given-names>P.</given-names></name> <name><surname>Khomh</surname><given-names>F.</given-names></name></person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer Nature Switzerland</publisher-name>), <fpage>219</fpage>&#x2013;<lpage>234</lpage>.</citation></ref>
<ref id="ref112"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>van der Willigen</surname><given-names>R. F.</given-names></name> <name><surname>Versnel</surname><given-names>H.</given-names></name> <name><surname>van Opstal</surname><given-names>A. J.</given-names></name></person-group> (<year>2024</year>). <article-title>Spectral-temporal processing of naturalistic sounds in monkeys and humans</article-title>. <source>J. Neurophysiol.</source> <volume>131</volume>, <fpage>38</fpage>&#x2013;<lpage>63</lpage>. doi: <pub-id pub-id-type="doi">10.1152/jn.00129.2023</pub-id>, PMID: <pub-id pub-id-type="pmid">37965933</pub-id></citation></ref>
<ref id="ref113"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>van Velzen</surname><given-names>M.</given-names></name> <name><surname>de Graaf-Waar</surname><given-names>H. I.</given-names></name> <name><surname>Ubert</surname><given-names>T.</given-names></name> <name><surname>van der Willigen</surname><given-names>R. F.</given-names></name> <name><surname>Muilwijk</surname><given-names>L.</given-names></name> <name><surname>Schmitt</surname><given-names>M. A.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>21st century (clinical) decision support in nursing and allied healthcare. Developing a learning health system: a reasoned design of a theoretical framework</article-title>. <source>BMC Med. Inform. Decis. Mak.</source> <volume>23</volume>:<fpage>279</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12911-023-02372-4</pub-id></citation></ref>
<ref id="ref114"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Vasireddy</surname><given-names>I.</given-names></name> <name><surname>Sriveni</surname><given-names>B.</given-names></name> <name><surname>Prathyusha</surname><given-names>K.</given-names></name> <name><surname>Kandi</surname><given-names>A.</given-names></name></person-group> (<year>2024</year>). <article-title>Sentiment analysis of web media using hybrid model with t5 and gpt-4 models</article-title>, in: <conf-name>2nd international conference on recent trends in microelectronics, automation, computing and communications systems (ICMACC)</conf-name>, (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>634</fpage>&#x2013;<lpage>639</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ICMACC62921.2024.10894040</pub-id></citation></ref>
<ref id="ref115"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Vaswani</surname><given-names>A.</given-names></name> <name><surname>Shazeer</surname><given-names>N.</given-names></name> <name><surname>Parmar</surname><given-names>N.</given-names></name> <name><surname>Uszkoreit</surname><given-names>J.</given-names></name> <name><surname>Jones</surname><given-names>L.</given-names></name> <name><surname>Gomez</surname><given-names>A. N.</given-names></name> <etal/></person-group>. (<year>2017</year>). &#x201C;<article-title>Attention is all you need</article-title>&#x201D;, in: <source>Advances in Neural Information Processing Systems 30 (31st NeurIPS, 2017)</source>, eds. <person-group person-group-type="editor"><name><surname>Guyon</surname><given-names>I.</given-names></name> <name><surname>Luxburg</surname><given-names>U. V.</given-names></name> <name><surname>Bengio</surname><given-names>S.</given-names></name> <name><surname>Wallach</surname><given-names>H.</given-names></name> <name><surname>Fergus</surname><given-names>R.</given-names></name> <name><surname>Vishwanathan</surname><given-names>S.</given-names></name> <etal/></person-group>: <publisher-name>Curran Associates, Inc.</publisher-name>), <fpage>15</fpage> pages. Available: <ext-link xlink:href="https://proceedings.neurips.cc/paper_files/paper/2017/file/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf" ext-link-type="uri">https://proceedings.neurips.cc/paper_files/paper/2017/file/3f5ee243547dee91fbd053c1c4a845aa-Paper.pdf</ext-link></citation></ref>
<ref id="ref116"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Verhulst</surname><given-names>S.</given-names></name> <name><surname>Zahuranec</surname><given-names>A. J.</given-names></name> <name><surname>Chafetz</surname><given-names>H.</given-names></name></person-group> (<year>2025</year>). Moving toward the fair-r principles: Advancing AI-ready data. Available online at: SSRN.</citation></ref>
<ref id="ref117"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Walturn</surname></name></person-group> (<year>2025</year>). Gpt-4.1 and the frontier of AI: Capabilities, improvements, and comparison to claude 3, gemini, mistral, and llama [Online]. Available online at: <ext-link xlink:href="https://www.walturn.com/insights/gpt-4-1-and-the-frontier-of-ai-capabilities-improvements-and-comparison-to-claude-3-gemini-mistral-and-llama" ext-link-type="uri">https://www.walturn.com/insights/gpt-4-1-and-the-frontier-of-ai-capabilities-improvements-and-comparison-to-claude-3-gemini-mistral-and-llama</ext-link> (Accessed May 13, 2025).</citation></ref>
<ref id="ref118"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Woisetschl&#x00E4;ger</surname><given-names>H.</given-names></name> <name><surname>Erben</surname><given-names>A.</given-names></name> <name><surname>Marino</surname><given-names>B.</given-names></name> <name><surname>Wang</surname><given-names>S.</given-names></name> <name><surname>Lane</surname><given-names>N. D.</given-names></name> <name><surname>Mayer</surname><given-names>R.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Federated learning priorities under the European Union artificial intelligence act</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2402.05968</pub-id></citation></ref>
<ref id="ref119"><citation citation-type="book"><person-group person-group-type="author"><collab id="coll8">World Health Organization</collab></person-group> (<year>2001</year>). <source>Icf: International classification of functioning, disability and health / world health organization</source>. <publisher-loc>Geneva</publisher-loc>: <publisher-name>World Health Organization</publisher-name>.</citation></ref>
<ref id="ref120"><citation citation-type="journal"><person-group person-group-type="author"><collab id="coll9">World Medical Association</collab></person-group> (<year>2025</year>). <article-title>World medical association declaration of Helsinki: ethical principles for medical research involving human participants</article-title>. <source>JAMA</source> <volume>333</volume>, <fpage>71</fpage>&#x2013;<lpage>74</lpage>. doi: <pub-id pub-id-type="doi">10.1001/jama.2024.21972</pub-id>, PMID: <pub-id pub-id-type="pmid">39425955</pub-id></citation></ref>
<ref id="ref121"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xie</surname><given-names>C.</given-names></name> <name><surname>Cai</surname><given-names>S.</given-names></name> <name><surname>Wang</surname><given-names>W.</given-names></name> <name><surname>Li</surname><given-names>P.</given-names></name> <name><surname>Sang</surname><given-names>Z.</given-names></name> <name><surname>Yang</surname><given-names>K.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Infir: crafting effective small language models and multimodal small language models in reasoning</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2502.11573</pub-id></citation></ref>
<ref id="ref122"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xie</surname><given-names>J.</given-names></name> <name><surname>Chen</surname><given-names>Z.</given-names></name> <name><surname>Zhang</surname><given-names>R.</given-names></name> <name><surname>Wan</surname><given-names>X.</given-names></name> <name><surname>Li</surname><given-names>G.</given-names></name></person-group> (<year>2024</year>). <article-title>Large multimodal agents: a survey</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2402.15116</pub-id></citation></ref>
<ref id="ref123"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname><given-names>X.</given-names></name> <name><surname>Xiao</surname><given-names>Y.</given-names></name> <name><surname>Liu</surname><given-names>D.</given-names></name> <name><surname>Deng</surname><given-names>H.</given-names></name> <name><surname>Huang</surname><given-names>J.</given-names></name> <name><surname>Zhou</surname><given-names>Y.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Cross language transformation of free text into structured lobectomy surgical records from a multi center study</article-title>. <source>Sci. Rep.</source> <volume>15</volume>:<fpage>15417</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-025-97500-7</pub-id>, PMID: <pub-id pub-id-type="pmid">40316625</pub-id></citation></ref>
<ref id="ref124"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>T.</given-names></name> <name><surname>Kishore</surname><given-names>V.</given-names></name> <name><surname>Wu</surname><given-names>F.</given-names></name> <name><surname>Weinberger</surname><given-names>K. Q.</given-names></name> <name><surname>Artzi</surname><given-names>Y.</given-names></name></person-group> (<year>2019</year>). <article-title>Bertscore: evaluating text generation with bert</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1904.09675</pub-id></citation></ref>
<ref id="ref125"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Zhao</surname><given-names>Z.</given-names></name> <name><surname>Wallace</surname><given-names>E.</given-names></name> <name><surname>Feng</surname><given-names>S.</given-names></name> <name><surname>Klein</surname><given-names>D.</given-names></name> <name><surname>Singh</surname><given-names>S.</given-names></name></person-group> (<year>2021</year>). "<article-title>Calibrate before use: improving few-shot performance of language models</article-title>", in: <conf-name>Proceedings of the 38th international conference on machine learning</conf-name>, eds. <person-group person-group-type="editor"><name><surname>Marina</surname><given-names>M.</given-names></name> <name><surname>Tong</surname><given-names>Z.</given-names></name></person-group>: <publisher-name>PMLR</publisher-name>, <fpage>12697</fpage>&#x2013;&#x2013;<lpage>12706</lpage>. Available online at: <ext-link xlink:href="https://proceedings.mlr.press/v139/zhao21c.html" ext-link-type="uri">https://proceedings.mlr.press/v139/zhao21c.html</ext-link></citation></ref>
<ref id="ref126"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname><given-names>Y.</given-names></name> <name><surname>Li</surname><given-names>T.</given-names></name> <name><surname>Huang</surname><given-names>H.</given-names></name> <name><surname>Zeng</surname><given-names>T.</given-names></name> <name><surname>Lu</surname><given-names>J.</given-names></name> <name><surname>Chu</surname><given-names>C.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Are all prompt components value-neutral? Understanding the heterogeneous adversarial robustness of dissected prompt in large language models</article-title>. <source>arXiv [Preprint]</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2508.01554</pub-id></citation></ref>
</ref-list>
</back>
</article>