<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="brief-report">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Vet. Sci.</journal-id>
<journal-title>Frontiers in Veterinary Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Vet. Sci.</abbrev-journal-title>
<issn pub-type="epub">2297-1769</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fvets.2024.1352726</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Veterinary Science</subject>
<subj-group>
<subject>Perspective</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Text mining for disease surveillance in veterinary clinical data: part two, training computers to identify features in clinical text</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Davies</surname> <given-names>Heather</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2585031/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Nenadic</surname> <given-names>Goran</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/893106/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Alfattni</surname> <given-names>Ghada</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn003"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/945848/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Arguello Casteleiro</surname> <given-names>Mercedes</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn004"><sup>&#x02020;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Al Moubayed</surname> <given-names>Noura</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/393587/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Farrell</surname> <given-names>Sean</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2599534/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Radford</surname> <given-names>Alan D.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Noble</surname> <given-names>P.-J. M.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2132355/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Institute of Infection, Veterinary and Ecological Sciences, University of Liverpool</institution>, <addr-line>Liverpool</addr-line>, <country>United Kingdom</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Computer Science, Manchester University</institution>, <addr-line>Manchester</addr-line>, <country>United Kingdom</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Computer Science, Durham University</institution>, <addr-line>Durham</addr-line>, <country>United Kingdom</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Moh A. Alkhamis, Kuwait University, Kuwait</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Heinzpeter Schwermer, Federal Food Safety and Veterinary Office (FSVO), Switzerland</p>
<p>Severino Segato, University of Padua, Italy</p></fn>
<corresp id="c001">&#x0002A;Correspondence: P.-J. M. Noble <email>rtnorle&#x00040;liverpool.ac.uk</email></corresp>
<fn fn-type="present-address" id="fn002"><p>&#x02020;Present addresses: Heather Davies, Department of Population Medicine, Ontario Veterinary College, University of Guelph, Guelph, ON, Canada</p></fn>
<fn fn-type="present-address" id="fn003"><p>Ghada Alfattni, Department of Computer Science, Jamoum University College, Umm Al-Qura University, Makkah, Saudi Arabia</p></fn>
<fn fn-type="present-address" id="fn004"><p>Mercedes Arguello Casteleiro, Electronics and Computer Science, University of Southampton, Southampton, United Kingdom</p></fn></author-notes>
<pub-date pub-type="epub">
<day>22</day>
<month>08</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>11</volume>
<elocation-id>1352726</elocation-id>
<history>
<date date-type="received">
<day>08</day>
<month>12</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>17</day>
<month>07</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Davies, Nenadic, Alfattni, Arguello Casteleiro, Al Moubayed, Farrell, Radford and Noble.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Davies, Nenadic, Alfattni, Arguello Casteleiro, Al Moubayed, Farrell, Radford and Noble</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>In part two of this mini-series, we evaluate the range of machine-learning tools now available for application to veterinary clinical text-mining. These tools will be vital to automate extraction of information from large datasets of veterinary clinical narratives curated by projects such as the Small Animal Veterinary Surveillance Network (SAVSNET) and VetCompass, where volumes of millions of records preclude reading records and the complexities of clinical notes limit usefulness of more &#x0201C;traditional&#x0201D; text-mining approaches. We discuss the application of various machine learning techniques ranging from simple models for identifying words and phrases with similar meanings to expand lexicons for keyword searching, to the use of more complex language models. Specifically, we describe the use of language models for record annotation, unsupervised approaches for identifying topics within large datasets, and discuss more recent developments in the area of generative models (such as ChatGPT). As these models become increasingly complex it is pertinent that researchers and clinicians work together to ensure that the outputs of these models are explainable in order to instill confidence in any conclusions drawn from them.</p></abstract>
<kwd-group>
<kwd>big data</kwd>
<kwd>machine learning</kwd>
<kwd>neural language modeling</kwd>
<kwd>clinical records</kwd>
<kwd>companion animals</kwd>
</kwd-group>
<counts>
<fig-count count="2"/>
<table-count count="0"/>
<equation-count count="0"/>
<ref-count count="46"/>
<page-count count="7"/>
<word-count count="5320"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Veterinary Epidemiology and Economics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Natural Language Processing (NLP) is a rapidly growing branch of machine learning aimed at understanding unstructured text using computational methodologies. NLP provides frameworks for computer systems to understand, interpret, and generate human language, which has important implications for applications such as machine translation, text classification, named entity recognition and summarization (<xref ref-type="bibr" rid="B1">1</xref>). NLP involves the development of algorithms and computational models to help achieve these challenging goals. However, language does not always follow defined rules but is complex, ambiguous and context-dependent with complications that include regional and dialect differences, emergence of new words, phrases, abbreviations, and colloquialisms.</p>
<p>Within healthcare, NLP is increasingly used to analyze and extract useful information from large volumes of unstructured clinical text data such as electronic health records (EHRs) and clinical notes in an automated process, enhancing the speed and efficiency of clinical decision-making, supporting the detection of disease outbreaks and the ongoing monitoring of disease incidence and prevalence. One study used EHRs from nine hospitals to support an understanding of the outbreak&#x00027;s transmission routes (<xref ref-type="bibr" rid="B2">2</xref>). Another area in which NLP has been applied is pharmacovigilance, to capture adverse drug events being reported within clinical narratives to gauge the prevalence and severity and can support the understanding of multi-drug interactions (<xref ref-type="bibr" rid="B3">3</xref>, <xref ref-type="bibr" rid="B4">4</xref>).</p>
<p>With the availability of large volumes of veterinary EHRs through frameworks such as SAVSNET and VetCompass, there is a growing need to develop tools and methodologies to fully utilize the rich source of disease information that lies within them (<xref ref-type="bibr" rid="B5">5</xref>, <xref ref-type="bibr" rid="B6">6</xref>). As we will describe, machine learning models have the potential to reveal disease syndromic signals within complex textual inputs and have become increasingly accessible even to researchers on modest budgets. This democratization of access to machine learning methods with the attendant potential to screen clinical records at scale has the potential to enhance our understanding of disease patterns in veterinary medicine profoundly.</p>
<p>In the second part of this mini-series, we will discuss the applications and potentials of machine learning methodology to extract valuable insights from unstructured clinical records. We explore how such tools are the building blocks for improving the capabilities of downstream applications such as disease epidemiology and outbreak surveillance. We examine the role of language models, such as bidirectional encoder representations through transformers (BERT) (<xref ref-type="bibr" rid="B7">7</xref>) and generative pretrained transformers such as chatGPT (<xref ref-type="bibr" rid="B8">8</xref>), Llama (<xref ref-type="bibr" rid="B9">9</xref>) (and there are now many more of these to choose from), to extract word meanings to understand the nuances of language and spelling variations within the corpora to better adapt fixed rule-based systems before evaluating them as independent classification tools. By providing an overview of the field&#x00027;s current state, we aim to highlight the pivotal role of machine learning-based text mining in enhancing companion animal care and disease surveillance in veterinary medicine.</p></sec>
<sec id="s2">
<title>2 Text-mining veterinary clinical notes using machine learning</title>
<sec>
<title>2.1 Machine-learning for word meaning</title>
<p>Machine-learning (ML) is becoming increasingly important for text analysis. In many cases ML relies on neural networks. These are computational representations or software simulations of biological neural networks wherein virtual neurons (or nodes) in multiple layers, are interconnected. Each node aggregates the value of connections from nodes in the layer above, the mathematical weighting of these connections adjusts how much any given connection contributes to the activation of a node. In the simplest neural network these weightings are adjusted by evaluating training data over many iterations and at each iteration adjusting these weightings until the input to the network leads to the &#x0201C;correct&#x0201D; output. The number of weightings (connections) is sometimes referred to as the number of parameters (<xref ref-type="bibr" rid="B10">10</xref>).</p>
<p>In part one of this mini-series we discussed the utility of keyword searches for identifying features of EHRs. Veterinary free-text invariably includes both technical scientific and colloquial language, including non-standard abbreviations and misspellings. As it is difficult to curate a complete list of the different ways in which veterinary professionals will describe the same observation, dictionary development can be time consuming and the use of standardized ontologies may result in a loss of recall. Furthermore, even the most complete dictionaries will require updating due to the ever changing nature of nature language. However, there are some simple machine learning approaches which can be implemented in order to augment these dictionaries. An example of such approaches is word embeddings.</p>
<p>Word embedding involves creating vector representations of words (coordinates for a word in a multidimensional word-space) which encapsulate word meaning, therefore allowing mathematical analysis of text. An efficient method for creating embeddings, word2vec, was developed by Mikolov et al., training a neural network to predict words in sentences based on the words surrounding them using a large corpus (hundreds of thousands to millions) of documents (<xref ref-type="bibr" rid="B11">11</xref>). The resultant neural network weightings are effectively a vector (usually 200&#x02013;300 numbers) representing the words embedding. Words, spellings and abbreviations with similar meanings can be identified due to the mathematical similarity of the vectors representing them. A similar procedure can be used to &#x0201C;embed&#x0201D; sentences and passages of text giving these numerical representations of their overall meaning (<xref ref-type="bibr" rid="B12">12</xref>).</p>
<p>This approach has been used to expand a dictionary of dietary supplements as described in clinical notes (<xref ref-type="bibr" rid="B13">13</xref>), with the model identifying between one and 12 variants for each supplement, resulting in retrieval of 8.39% more clinical notes on average. Similarly, word2vec was used to identify misspellings of pharmaceutical words in clinical notes (<xref ref-type="bibr" rid="B14">14</xref>) and resulted in identification of 150 new terms which were used to create an extended lexicon.</p>
<p>The specific advantage of this approach is that the model is trained on data taken from the target corpus, allowing development of embeddings more representative of the language used within that corpus. However, word2vec models do not capture remote relationships between words and attribute a single vector for words that might have multiple meanings. This limitation is avoided in later approaches such as ELMO (<xref ref-type="bibr" rid="B15">15</xref>) and transformer architecture (as described below).</p></sec>
<sec>
<title>2.2 Language models as tools for record annotation</title>
<p>Language models can accurately capture semantic and syntactic structures, which is critical to leverage the rich sources of information that unstructured clinical narratives have within them. Such language models permit flexibility in understanding by capturing patterns and relationships across large volumes of data rather than pre-defined rules; the dynamic nature of language requires an equally malleable system to capture the many ways clinicians can articulate their notes. Rule-based systems are limited to defined inputs where subtle variants in language can add significant complexities to their design and, therefore, can only practically be used for searching for one item at a time such as for single disease scope studies. Rule-based systems also rely heavily on the developers&#x00027; domain-specific knowledge and manual readings of the records to produce. Neural network-based architectures present in language models, such as convolutional neural networks (CNNs) and recurrent neural networks (RNNs), were a significant innovation away from statistical and rule-based frameworks and introduced the concept of word embeddings, a movement toward capturing rich contextual word representations (<xref ref-type="bibr" rid="B16">16</xref>).</p>
<p>Recently, the transformer architecture, which capitalizes on the concept of attention (the relationship between words in phrases/sentences sometimes separated by some distance), has enabled new state-of-the-art performances across many NLP tasks (<xref ref-type="bibr" rid="B17">17</xref>). Transformers are the core architecture behind Bidirectional Encoder Representation for Transformers (BERT) (<xref ref-type="bibr" rid="B7">7</xref>), Generative Pretraining (GPT) (<xref ref-type="bibr" rid="B18">18</xref>), and Language Models for Dialog Applications (LaMDA) (<xref ref-type="bibr" rid="B19">19</xref>). A key difference to word2vec embeddings is that these models allow for context-specific representations of words to allow different meanings to be coded differently. For example the word &#x0201C;discharge&#x0201D; may mean fluid leaking from a wound or releasing a patient from hospital and would have the same embedding in word2vec models but is ultimately treated as different entities in BERT models depending on context (<xref ref-type="bibr" rid="B7">7</xref>).</p>
<p>Disease coding frameworks, such as the International Classification of Disease (ICD), provide a robust methodology for understanding mortality and morbidity information for research and epidemiology (<xref ref-type="bibr" rid="B20">20</xref>). However, disease coding of unstructured clinical notes is challenging and is inherently time-consuming, expensive, and prone to errors (<xref ref-type="bibr" rid="B21">21</xref>&#x02013;<xref ref-type="bibr" rid="B24">24</xref>). For these reasons, the concept of automating such a process with language models has been well-explored, with previous research exploring the application of RNNs, CNNS and, more recently, the incorporation of attention mechanisms and the transformer architectures (<xref ref-type="bibr" rid="B25">25</xref>&#x02013;<xref ref-type="bibr" rid="B29">29</xref>). The desire for automated disease annotation frameworks to exist within veterinary medicine is no different. Previous works have aimed to apply SNOMED-CT diagnosis labels using bidirectional long-short-term memory networks (BLSTMs), a variant of RNNs, using 112,558 expert annotations from a tertiary referral centre showing promising results (<xref ref-type="bibr" rid="B30">30</xref>). Further works capitalized from this integrating a transformer architecture allowing for a hierarchical organization of automated disease codings (<xref ref-type="bibr" rid="B31">31</xref>). Language models have also been used to understand veterinarians reasoning behind an antimicrobial administration; here, a BERT model was additionally trained on 15 million clinical notes from the VetCompass Australia corpus (<xref ref-type="bibr" rid="B32">32</xref>).</p>
<p>In summary, language models present promise for automating annotations within unstructured EHRs. Their capacity to analyze extensive datasets and discern intricate linguistic relationships empowers these models to enhance annotation precision and speed, facilitating more comprehensive data analysis in disease epidemiology. An exemplification of this potential lies in the development of PetBERT, a substantial language model trained on a corpus exceeding 500 million tokens sourced from the SAVSNET dataset (<xref ref-type="bibr" rid="B33">33</xref>). This dataset comprises clinical narratives from diverse veterinary practices across the UK. Through fine-tuning, PetBERT was transformed into a multi-label classifier proficient in automatically coding veterinary clinical EHRs using the International Classification of Diseases 11 framework. Impressively, it achieved F1 scores surpassing 83% across 20 disease codings with minimal annotation requirements. Moreover, they serve as foundational structures for bolstering disease outbreak detection capabilities. Employing this syndromic labeling system, we identified a documented disease outbreak. Comparative analysis between PetBERT&#x00027;s automated identification and the previously employed clinician-assigned point-of-care labeling strategy revealed PetBERT&#x00027;s capability to identify the outbreak up to three weeks earlier. The demonstrated proficiency of PetBERT in automating coding processes within veterinary clinical narratives underscores the transformative potential of language models in augmenting disease surveillance and timely outbreak detection within veterinary medicine.</p></sec>
<sec>
<title>2.3 Unsupervised machine learning</title>
<p>The machine-learning approaches described above often rely on identifying groups of records for study and in the case of using neural language models will often entail training models using gold-standard annotations made by experts. These are often referred to as &#x00027;supervised&#x00027; systems where the researcher/developer directs what the system learns. Unsupervised systems involve presenting a dataset to a model without stipulating an underlying structure to be detected and to identify the underlying structure within the data in the hope that this will expose relevant clinical features. One such approach is topic modeling. A key approach to this was published by Blei et al. (<xref ref-type="bibr" rid="B34">34</xref>) using a statistical approach to text (latent dirichlet allocation) which modeled the assumption that documents contain a distribution of topics and that topics are made up of a specific distribution of words. Reverse engineering this allowed identification of the word distributions present in a set of topics which were inferred from the text itself rather than a prior assumption of what topics were present in the data. This method is made readily available through accessible programming interfaces (<xref ref-type="bibr" rid="B35">35</xref>). This approach has been applied to veterinary clinical data allowing clinically relevant topics to be identified in veterinary text to the extent that a topic identified through this method displayed a clear temporal pattern matching a known national outbreak of gastroenteric disease in dogs (<xref ref-type="bibr" rid="B36">36</xref>). Topic modeling has been recognized as valuable tool for bioinformatic research (<xref ref-type="bibr" rid="B37">37</xref>) and as a tool for clinical research (<xref ref-type="bibr" rid="B38">38</xref>, <xref ref-type="bibr" rid="B39">39</xref>).A key feature of topic modeling is that topics discovered in documents are easily interpreted due to each topic having a list of words (and probabilities for those words) that make them up, such that in the example above the outbreak identified above could be clearly seen to be gastroenteric disease given that it comprised words like &#x0201C;diarrhoea,&#x0201D; &#x0201C;vomit,&#x0201D; and &#x0201C;food.&#x0201D; Furthermore, topic-modeling methods allow for evaluation of the weighting of words that comprise the topic with time (or across other categories such as breed, age, and date) highlighting evolution of themes with time that, in the case of disease phenotype, might illustrate emergence of new syndromes (<xref ref-type="bibr" rid="B36">36</xref>, <xref ref-type="bibr" rid="B40">40</xref>). More recently, transformer-based models that create whole document embeddings i.e., representation of clinical notes as 768 dimensional arrays based on the meaning of words/tokens in the documents provide another route to topic modeling. Following a reduction of dimensionality in these arrays, a clustering algorithm can be used to cluster documents. An analysis of term-frequency-inverse document frequency (tf-idf) identifies key-words reflected in the these clusters. These words then indicate themes or topics in those documents. This approach is encapsulated in the BERTopic Package (<xref ref-type="bibr" rid="B41">41</xref>) and when applied to 1,000,000 SAVSNET records, we produced a topic model comprising 200 different topics each with keywords produced by the model without external prompting or intervention. Each topic is characterized by the keywords present and these usually have a clear clinical correlate for instance words describing ear disease (&#x0201C;ear,&#x0201D; &#x0201C;canal,&#x0201D; &#x0201C;left ear,&#x0201D; &#x0201C;wax,&#x0201D; &#x0201C;right ear,&#x0201D; &#x0201C;otitis,&#x0201D; &#x0201C;discharge,&#x0201D; &#x0201C;left,&#x0201D; &#x0201C;drops,&#x0201D; &#x0201C;both,&#x0201D; &#x0201C;osurnia&#x0201D;) and words describing wound management (&#x0201C;collar,&#x0201D; &#x0201C;wound,&#x0201D; &#x0201C;buster collar,&#x0201D; &#x0201C;poc,&#x0201D; &#x0201C;keep,&#x0201D; &#x0201C;healing,&#x0201D; &#x0201C;looks,&#x0201D; &#x0201C;wound looks,&#x0201D; &#x0201C;healed,&#x0201D; &#x0201C;post,&#x0201D; &#x0201C;off,&#x0201D; &#x0201C;licking,&#x0201D; and &#x0201C;well&#x0201D;). The probability distribution for a given topic could be calculated across breeds. An example of how a topic relating to seizures (with keywords such as seizure, bloods, diazepam epiphen, and epilepsy) created using BERTopic in this way is distributed in records from dogs of different breeds is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. This data was very similar to published breed-related data on seizuring (<xref ref-type="bibr" rid="B42">42</xref>). While topics may not be precise, they can allow rapid and comprehensive screening of large volumes of records for a huge variety of disease phenotypes in a single study.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Distribution of odds ratios with 95% confidence interval for a given topic to occur in records from specific breeds. The model comprised 200 topics, generated entirely by the computer without external &#x0201C;hints,&#x0201D; many having clear clinical relevance. Here, odds ratio for records with topic 23 as their most probable topic are calculated against the odds of records from crossbreeds having that topic as the most probable one. Data for topic 23 with key-words: seizure, seizures, bloods, diazepam, epiphen, phenobarb, epilepsy, pexion, lasted, episode.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fvets-11-1352726-g0001.tif"/>
</fig></sec>


<sec>
<title>2.4 Generative models</title>
<p>The neural network models described so far have ranged from tens of thousands of connections (parameters) through to hundreds of millions of parameters (in BERT Language models). More recently models have been developed that have billions to trillions of parameters. As a comparison, the human neocortex is estimated to have in the order of 200 trillion synapses (<xref ref-type="bibr" rid="B43">43</xref>). These generative models are often trained using tasks such as text completion (particularly next word prediction) and question/answer using massive training sets from diverse sources. Examples include GPT-3, ChatGPT, created by OpenAI (<xref ref-type="bibr" rid="B44">44</xref>) and OPT and Llama from Meta (<xref ref-type="bibr" rid="B45">45</xref>). The complexity of the internal wiring of these models combined with the extensive training data set leads to behavior that is uncannily human. The main form of interaction is to provide a prompt to which the model generates an answer (this is after all the task the model is trained on). Thus prompting ChatGPT (<xref ref-type="bibr" rid="B8">8</xref>) with the text <italic>&#x0201C;tell me in 30 words, why dogs vomit&#x0201D;</italic> returns the text <italic>&#x0201C;Dogs can vomit due to various reasons such as eating too fast, consuming something toxic, having an underlying medical condition, or experiencing motion sickness.&#x0201D;</italic> On the face of it, this conversational dialog appears challenging to extract structured data from but the model will respond with more useful text if prompted in a more structured manner. So called &#x0201C;prompt engineering&#x0201D; allows the prompter to coerce the model to return structured outputs and can be used to classify text from clinical records. For instance when trying to extract an important clinical index of health such as heart rate data from free-text narratives, a suitably engineered prompt can do this (<xref ref-type="fig" rid="F2">Figure 2</xref>). This technique would allow evaluation of a population at scale where previously extensive manual reading or arcane rule-based classification of records might be needed covering only a small subset of available records.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Example of prompt engineering to extract selected data (in this case heart rate) from clinical narratives.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fvets-11-1352726-g0002.tif"/>
</fig>


<p>The ability to run prompts across numerous texts is made available through a programming interface which, in theory, could allow for screening of extremely large numbers of records but there are several drawbacks: Firstly, for large volumes, record screening can have a cost per record which can be come substantial for very large datasets (millions of records) with multiple prompts; Secondly, some large language models capture the prompt text for further model training which can lead to some of the potentially sensitive material appearing in outputted text for other users. In order to run such large language models on a local machine (i.e., a copy which will not train on the text or be prompted by external users) requires substantial computing resource. Additionally, when using more complex prompts looking for complicated responses, further issues arise: given the very varied nature of the training data, these models can output opinion that can be misleading, completely fictitious (often referred to as &#x0201C;hallucination&#x0201D;) and even prejudiced/unwholesome. In the case of ChatGPT, beyond the prompt-response training, further fine tuning has been implemented using reinforcement learning from human feedback among other tools to try to forestall misleading or offensive content (<xref ref-type="bibr" rid="B44">44</xref>). In our preliminary experiments utilizing ChatGPT to identify overweight animals and body condition scores based on notes written by clinicians during consultations, ChatGPT&#x00027;s performance compared very favorably with a rule-based classifier used for the same task (<xref ref-type="bibr" rid="B46">46</xref>). Manual reading of records remains the gold standard against which such approaches are validated.</p></sec></sec>
<sec sec-type="conclusions" id="s3">
<title>3 Conclusion</title>
<p>Machine learning and artificial intelligence are revolutionizing our ability to automate the generation of signals relating to disease phenotypes in veterinary patients using EHRs. The underlying machinery of the tools is becoming less and less explainable as models with massive numbers of parameters are trained on vast datasets. The impact of large language models will become very significant in the coming years and it will be important that users understand the provenance of data from these models and that researchers work to ensure that the outputs from these models are explainable. While the methods discussed in this second part of the review series have clear benefits in screening huge volumes of data sometimes without requiring any stipulation of the nature of signals to detect, there will remain a role for manual review of records identified by these tools to maintain confidence that valid conclusions are drawn from them.</p></sec>
<sec sec-type="data-availability" id="s4">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s5">
<title>Author contributions</title>
<p>HD: Writing - review &#x00026; editing, Writing - original draft. GN: Writing - review &#x00026; editing. GA: Writing - review &#x00026; editing. MA: Writing - review &#x00026; editing. NA: Writing - review &#x00026; editing, Writing - original draft. SF: Writing - review &#x00026; editing, Writing - original draft. AR: Writing - review &#x00026; editing. P-JN: Writing - review &#x00026; editing, Writing - original draft.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="s6">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. Work for this review carried out by GA, GN, and MA at Manchester University was funded by CWG grant from Dogs Trust (SAVSNET Agile) and Healtex: UK Healthcare Text Analytics Research Network (EP/N027280/1, EPSRC). HD was funded by University of Liverpool. P-JN was funded by CWG grant from Dogs Trust (SAVSNET Agile). SF was supervised by NA on a BBSRC funded Ph. D. studentship. This work was partly supported by Dogs Trust through the SAVSNET Agile award.</p>
</sec>
<ack><p>This work would not have been possible without the contribution of practicing vets across the UK contributing to the SAVSNET project and without the help and support of the SAVSNET core team including Bethaney Brant, Steve Smyth, and Gina Pinchbeck.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s7">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cambria</surname> <given-names>E</given-names></name> <name><surname>White</surname> <given-names>B</given-names></name></person-group>. <article-title>Jumping NLP curves: a review of natural language processing research</article-title>. <source>IEEE Comput Intell Mag</source>. (<year>2014</year>) <volume>9</volume>:<fpage>48</fpage>&#x02013;<lpage>57</lpage>. <pub-id pub-id-type="doi">10.1109/mci.2014.2307227</pub-id></citation>
</ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sundermann</surname> <given-names>AJ</given-names></name> <name><surname>Miller</surname> <given-names>JK</given-names></name> <name><surname>Marsh</surname> <given-names>JW</given-names></name> <name><surname>Saul</surname> <given-names>MI</given-names></name> <name><surname>Shutt</surname> <given-names>KA</given-names></name> <name><surname>Pacey</surname> <given-names>M</given-names></name> <etal/></person-group>. <article-title>Automated Data Mining of the electronic health record for investigation of healthcare-associated outbreaks</article-title>. <source>Infect Contr Hospit Epidemiol</source>. (<year>2019</year>) <volume>40</volume>:<fpage>314</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1017/ice.2018.343</pub-id><pub-id pub-id-type="pmid">30773168</pub-id></citation></ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>Y</given-names></name> <name><surname>Thompson</surname> <given-names>WK</given-names></name> <name><surname>Herr</surname> <given-names>TM</given-names></name> <name><surname>Zeng</surname> <given-names>Z</given-names></name> <name><surname>Berendsen</surname> <given-names>MA</given-names></name> <name><surname>Jonnalagadda</surname> <given-names>SR</given-names></name> <etal/></person-group>. <article-title>Natural language processing for EHR-based pharmacovigilance: a structured review</article-title>. <source>Drug Saf</source> . (<year>2017</year>) <volume>40</volume>:<fpage>1075</fpage>&#x02013;<lpage>89</lpage>. <pub-id pub-id-type="doi">10.1007/s40264-017-0558-6</pub-id><pub-id pub-id-type="pmid">28643174</pub-id></citation></ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>F</given-names></name> <name><surname>Jagannatha</surname> <given-names>A</given-names></name> <name><surname>Yu</surname> <given-names>H</given-names></name></person-group>. <article-title>Towards Drug Safety Surveillance and pharmacovigilance: current progress in detecting medication and adverse drug events from Electronic Health Records</article-title>. <source>Drug Saf</source> . (<year>2019</year>) <volume>42</volume>:<fpage>95</fpage>&#x02013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1007/s40264-018-0766-8</pub-id><pub-id pub-id-type="pmid">30649734</pub-id></citation></ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Radford</surname> <given-names>A</given-names></name> <name><surname>Tierney</surname></name> <name><surname>Coyne</surname> <given-names>KP</given-names></name> <name><surname>Gaskell</surname> <given-names>RM</given-names></name> <name><surname>Noble</surname> <given-names>PJ</given-names></name> <name><surname>Dawson</surname> <given-names>S</given-names></name> <etal/></person-group>. <article-title>Developing a network for small animal disease surveillance</article-title>. <source>Vet Rec</source>. (<year>2010</year>) <volume>167</volume>:<fpage>472</fpage>&#x02013;<lpage>4</lpage>. <pub-id pub-id-type="doi">10.1136/vr.c5180</pub-id><pub-id pub-id-type="pmid">20871079</pub-id></citation></ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>McGreevy</surname> <given-names>P</given-names></name> <name><surname>Thomson</surname> <given-names>P</given-names></name> <name><surname>Dhand</surname> <given-names>NK</given-names></name> <name><surname>Raubenheimer</surname> <given-names>D</given-names></name> <name><surname>Masters</surname> <given-names>S</given-names></name> <name><surname>Mansfield</surname> <given-names>CS</given-names></name> <etal/></person-group>. <article-title>VetCompass Australia: a national big data collection system for veterinary science</article-title>. <source>Animals</source>. (<year>2017</year>) <volume>7</volume>:<fpage>74</fpage>. <pub-id pub-id-type="doi">10.3390/ani7100074</pub-id><pub-id pub-id-type="pmid">28954419</pub-id></citation></ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Devlin</surname> <given-names>J</given-names></name> <name><surname>Chang</surname> <given-names>MW</given-names></name> <name><surname>Lee</surname> <given-names>K</given-names></name> <name><surname>Toutanova</surname> <given-names>K</given-names></name></person-group>. <article-title>BERT: pre-training of deep bidirectional transformers for language understanding</article-title>. In: <source>NAACL HLT 2019&#x02013;2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies&#x02014;Proceedings of the Conference</source> (<year>2018</year>). p. <fpage>4171</fpage>&#x02013;<lpage>86</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1810.04805">http://arxiv.org/abs/1810.04805</ext-link> (accessed November 23, 2023).</citation>
</ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab>OpenAI</collab></person-group>. <source>ChatGPT (Version 3.5)</source> (<year>2023</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.openai.com">https://www.openai.com</ext-link> (accessed November 23, 2023).</citation>
</ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Touvron</surname> <given-names>H</given-names></name> <name><surname>Lavril</surname> <given-names>T</given-names></name> <name><surname>Izacard</surname> <given-names>G</given-names></name> <name><surname>Martinet</surname> <given-names>X</given-names></name> <name><surname>Lachaux</surname> <given-names>MA</given-names></name> <name><surname>Lacroix</surname> <given-names>T</given-names></name> <etal/></person-group>. <article-title>LLaMA: open and efficient foundation language models</article-title>. ArXiv [preprint] arXiv: abs/2302.13971. (<year>2023</year>). <pub-id pub-id-type="doi">10.48550/arXiv.2302.13971</pub-id></citation>
</ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sarker</surname> <given-names>IH</given-names></name></person-group>. <article-title>Deep learning: a comprehensive overview on techniques, taxonomy, applications and research directions</article-title>. <source>SN Comput Sci</source>. (<year>2021</year>) <volume>2</volume>:<fpage>1</fpage>. <pub-id pub-id-type="doi">10.1007/s42979-021-00815-1</pub-id><pub-id pub-id-type="pmid">34426802</pub-id></citation></ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Mikolov</surname> <given-names>T</given-names></name> <name><surname>Chen</surname> <given-names>K</given-names></name> <name><surname>Corrado</surname> <given-names>G</given-names></name> <name><surname>Dean</surname> <given-names>J</given-names></name></person-group>. <article-title>Efficient estimation of word representations in vector space</article-title>. In: <source>1st International Conference on Learning Representations, Workshop Track Proceedings</source>. <publisher-loc>Scottsdale, AZ</publisher-loc>: <publisher-name>ICLR 2013</publisher-name> (<year>2013</year>).<pub-id pub-id-type="pmid">31752376</pub-id></citation></ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>L</given-names></name> <name><surname>Yen</surname> <given-names>IEH</given-names></name> <name><surname>Xu</surname> <given-names>K</given-names></name> <name><surname>Xu</surname> <given-names>F</given-names></name> <name><surname>Balakrishnan</surname> <given-names>A</given-names></name> <name><surname>Chen</surname> <given-names>PY</given-names></name> <etal/></person-group>. <article-title>Word mover&#x00027;s embedding: from Word2Vec to document embedding</article-title>. In: <source>Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing</source>. <publisher-loc>Brussels</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name> (<year>2018</year>). p. <fpage>4524</fpage>&#x02013;<lpage>34</lpage>.</citation>
</ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fan</surname> <given-names>Y</given-names></name> <name><surname>Pakhomov</surname> <given-names>S</given-names></name> <name><surname>McEwan</surname> <given-names>R</given-names></name> <name><surname>Zhao</surname> <given-names>W</given-names></name> <name><surname>Lindemann</surname> <given-names>E</given-names></name> <name><surname>Zhang</surname> <given-names>R</given-names></name></person-group>. <article-title>Using word embeddings to expand terminology of dietary supplements on clinical notes</article-title>. <source>JAMIA Open</source>. (<year>2019</year>) <volume>2</volume>:<fpage>246</fpage>&#x02013;<lpage>53</lpage>. <pub-id pub-id-type="doi">10.1093/jamiaopen/ooz007</pub-id><pub-id pub-id-type="pmid">31825016</pub-id></citation></ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Workman</surname> <given-names>ET</given-names></name> <name><surname>Divita</surname> <given-names>G</given-names></name> <name><surname>Shao</surname> <given-names>Y</given-names></name> <name><surname>Zeng-Treitler</surname> <given-names>Q</given-names></name></person-group>. <article-title>A proficient spelling analysis method applied to a pharmacovigilance task</article-title>. <source>Stud Health Technol Informat</source>. (<year>2019</year>) <volume>264</volume>:<fpage>452</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.3233/SHTI190262</pub-id><pub-id pub-id-type="pmid">31437964</pub-id></citation></ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Peters</surname> <given-names>ME</given-names></name> <name><surname>Neumann</surname> <given-names>M</given-names></name> <name><surname>Iyyer</surname> <given-names>M</given-names></name> <name><surname>Gardner</surname> <given-names>M</given-names></name> <name><surname>Clark</surname> <given-names>C</given-names></name> <name><surname>Lee</surname> <given-names>K</given-names></name> <etal/></person-group>. <article-title>Deep contextualized word representations</article-title>. In: <source>Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long Papers)</source>. <publisher-loc>New Orleans, LA</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name> (<year>2018</year>). p. <fpage>2227</fpage>&#x02013;<lpage>37</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/N18-1202">https://aclanthology.org/N18-1202</ext-link> (accessed November 23, 2023).</citation>
</ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Collobert</surname> <given-names>R</given-names></name> <name><surname>Weston</surname> <given-names>J</given-names></name></person-group>. <article-title>A unified architecture for natural language processing: deep neural networks with multitask learning</article-title>. In: <source>Proceedings of the 25th International Conference on Machine Learning. ICML &#x00027;08</source>. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name> (<year>2008</year>). p. <fpage>160</fpage>&#x02013;<lpage>7</lpage>.<pub-id pub-id-type="pmid">22461885</pub-id></citation></ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaswani</surname> <given-names>A</given-names></name> <name><surname>Shazeer</surname> <given-names>N</given-names></name> <name><surname>Parmar</surname> <given-names>N</given-names></name> <name><surname>Uszkoreit</surname> <given-names>J</given-names></name> <name><surname>Jones</surname> <given-names>L</given-names></name> <name><surname>Gomez</surname> <given-names>AN</given-names></name> <etal/></person-group>. <article-title>Attention is all you need</article-title>. arXiv [preprint] arXiv: abs/1706.03762. (<year>2017</year>). <pub-id pub-id-type="doi">10.48550/arXiv.1706.03762</pub-id></citation>
</ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname> <given-names>TB</given-names></name> <name><surname>Mann</surname> <given-names>B</given-names></name> <name><surname>Ryder</surname> <given-names>N</given-names></name> <name><surname>Subbiah</surname> <given-names>M</given-names></name> <name><surname>Kaplan</surname> <given-names>J</given-names></name> <name><surname>Dhariwal</surname> <given-names>P</given-names></name> <etal/></person-group>. <article-title>Language models are few-shot learners</article-title>. arXiv [preprint] arXiv:abs/2005.14165. (<year>2020</year>). <pub-id pub-id-type="doi">10.48550/arXiv.2005.14165</pub-id></citation>
</ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thoppilan</surname> <given-names>R</given-names></name> <name><surname>Freitas</surname> <given-names>DD</given-names></name> <name><surname>Hall</surname> <given-names>J</given-names></name> <name><surname>Shazeer</surname> <given-names>N</given-names></name> <name><surname>Kulshreshtha</surname> <given-names>A</given-names></name> <name><surname>Cheng</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>LaMDA: language models for dialog applications</article-title>. arXiv [preprint] arXiv: abs/2201.08239. (<year>2022</year>). <pub-id pub-id-type="doi">10.48550/arXiv.2201.08239</pub-id></citation>
</ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Harrison</surname> <given-names>JE</given-names></name> <name><surname>Weber</surname> <given-names>S</given-names></name> <name><surname>Jakob</surname> <given-names>R</given-names></name> <name><surname>Chute</surname> <given-names>CG</given-names></name></person-group>. <article-title>ICD-11: an international classification of diseases for the twenty-first century</article-title>. <source>BMC Med Informat Decision Mak</source>. (<year>2021</year>) <volume>21</volume>:<fpage>6</fpage>. <pub-id pub-id-type="doi">10.1186/s12911-021-01534-6</pub-id><pub-id pub-id-type="pmid">34753471</pub-id></citation></ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lloyd</surname> <given-names>SS</given-names></name> <name><surname>Rissing</surname> <given-names>JP</given-names></name></person-group>. <article-title>Physician and coding errors in patient records</article-title>. <source>J Am Med Assoc</source>. (<year>1985</year>) <volume>254</volume>:<fpage>1330</fpage>&#x02013;<lpage>6</lpage>.</citation>
</ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hasan</surname> <given-names>M</given-names></name> <name><surname>Meara</surname> <given-names>RJ</given-names></name> <name><surname>Bhowhick</surname> <given-names>BK</given-names></name></person-group>. <article-title>The quality of diagnostic coding in cerebrovascular disease</article-title>. <source>Int J Qual Health Care</source>. (<year>1995</year>) <volume>7</volume>:<fpage>407</fpage>&#x02013;<lpage>10</lpage>.</citation>
</ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Farzandipour</surname> <given-names>M</given-names></name> <name><surname>Sheikhtaheri</surname> <given-names>A</given-names></name> <name><surname>Sadoughi</surname> <given-names>F</given-names></name></person-group>. <article-title>Effective factors on accuracy of principal diagnosis coding based on International Classification of Diseases, the 10th revision (ICD-10)</article-title>. <source>Int J Inform Manag</source>. (<year>2010</year>) <volume>30</volume>:<fpage>78</fpage>&#x02013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijinfomgt.2009.07.002</pub-id></citation>
</ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>O&#x00027;Malley</surname> <given-names>KJ</given-names></name> <name><surname>Cook</surname> <given-names>KF</given-names></name> <name><surname>Price</surname> <given-names>MD</given-names></name> <name><surname>Wildes</surname> <given-names>KR</given-names></name> <name><surname>Hurdle</surname> <given-names>JF</given-names></name> <name><surname>Ashton</surname> <given-names>CM</given-names></name></person-group>. <article-title>Measuring diagnoses: ICD code accuracy</article-title>. <source>Health Serv Res</source>. (<year>2005</year>) <volume>40</volume>:<fpage>1620</fpage>&#x02013;<lpage>39</lpage>. <pub-id pub-id-type="doi">10.1111/j.1475-6773.2005.00444.x</pub-id><pub-id pub-id-type="pmid">16178999</pub-id></citation></ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname> <given-names>H</given-names></name> <name><surname>Xie</surname> <given-names>P</given-names></name> <name><surname>Hu</surname> <given-names>Z</given-names></name> <name><surname>Zhang</surname> <given-names>M</given-names></name> <name><surname>Xing</surname> <given-names>EP</given-names></name></person-group>. <article-title>Towards automated ICD coding using deep learning</article-title>. arXiv [preprint] arXiv: abs/1711.04075. (<year>2017</year>). <pub-id pub-id-type="doi">10.48550/arXiv.1711.04075</pub-id><pub-id pub-id-type="pmid">30385092</pub-id></citation></ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>F</given-names></name> <name><surname>Yu</surname> <given-names>H</given-names></name></person-group>. <article-title>ICD coding from clinical text using multi-filter residual convolutional neural network</article-title>. arXiv [preprint] arXiv: abs/1912.00862. (<year>2019</year>). <pub-id pub-id-type="doi">10.48550/arXiv.1912.00862</pub-id><pub-id pub-id-type="pmid">34322282</pub-id></citation></ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Cao</surname> <given-names>P</given-names></name> <name><surname>Chen</surname> <given-names>Y</given-names></name> <name><surname>Liu</surname> <given-names>K</given-names></name> <name><surname>Zhao</surname> <given-names>J</given-names></name> <name><surname>Liu</surname> <given-names>S</given-names></name> <name><surname>Chong</surname> <given-names>W</given-names></name></person-group>. <article-title>HyperCore: hyperbolic and co-graph representation for automatic ICD coding</article-title>. In: <source>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics</source>. <publisher-loc>Online</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name> (<year>2020</year>). p. <fpage>3105</fpage>&#x02013;<lpage>14</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2020.acl-main.282">https://aclanthology.org/2020.acl-main.282</ext-link> (accessed November 23, 2023).</citation>
</ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Z</given-names></name> <name><surname>Liu</surname> <given-names>J</given-names></name> <name><surname>Razavian</surname> <given-names>N</given-names></name></person-group>. <article-title>BERT-XML: large scale automated ICD coding using BERT pretraining</article-title>. In: <source>Proceedings of the 3rd Clinical Natural Language Processing Workshop</source>. <publisher-loc>Online</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name> (<year>2020</year>). p. <fpage>24</fpage>&#x02013;<lpage>34</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2020.clinicalnlp-1.3">https://aclanthology.org/2020.clinicalnlp-1.3</ext-link> (accessed November 23, 2023).</citation>
</ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Pascual</surname> <given-names>D</given-names></name> <name><surname>Luck</surname> <given-names>S</given-names></name> <name><surname>Wattenhofer</surname> <given-names>R</given-names></name></person-group>. <article-title>Towards BERT-based automatic ICD coding: limitations and opportunities</article-title>. In: <source>Proceedings of the 20th Workshop on Biomedical Language Processing</source>. <publisher-loc>Online</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name> (<year>2021</year>). p. <fpage>54</fpage>&#x02013;<lpage>63</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2021.bionlp-1.6">https://aclanthology.org/2021.bionlp-1.6</ext-link> (accessed November 23, 2023).</citation>
</ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nie</surname> <given-names>A</given-names></name> <name><surname>Zehnder</surname> <given-names>A</given-names></name> <name><surname>Page</surname> <given-names>RL</given-names></name> <name><surname>Zhang</surname> <given-names>Y</given-names></name> <name><surname>Pineda</surname> <given-names>AL</given-names></name> <name><surname>Rivas</surname> <given-names>MA</given-names></name> <etal/></person-group>. <article-title>DeepTag: inferring diagnoses from veterinary clinical notes</article-title>. <source>NPJ Digit Med</source>. (<year>2018</year>) <volume>1</volume>:<fpage>8</fpage>. <pub-id pub-id-type="doi">10.1038/s41746-018-0067-8</pub-id><pub-id pub-id-type="pmid">31304339</pub-id></citation></ref>
<ref id="B31">
<label>31.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y</given-names></name> <name><surname>Nie</surname> <given-names>A</given-names></name> <name><surname>Zehnder</surname> <given-names>A</given-names></name> <name><surname>Page</surname> <given-names>RL</given-names></name> <name><surname>Zou</surname> <given-names>J</given-names></name></person-group>. <article-title>VetTag: improving automated veterinary diagnosis coding via large-scale language modeling</article-title>. <source>NPJ Digit Med</source>. (<year>2019</year>) <volume>2</volume>:<fpage>1</fpage>. <pub-id pub-id-type="doi">10.1038/s41746-019-0113-1</pub-id><pub-id pub-id-type="pmid">31304381</pub-id></citation></ref>
<ref id="B32">
<label>32.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Hur</surname> <given-names>B</given-names></name> <name><surname>Baldwin</surname> <given-names>T</given-names></name> <name><surname>Verspoor</surname> <given-names>K</given-names></name> <name><surname>Hardefeldt</surname> <given-names>L</given-names></name> <name><surname>Gilkerson</surname> <given-names>J</given-names></name></person-group>. <article-title>Domain adaptation and instance selection for disease syndrome classification over veterinary clinical notes</article-title>. In: <source>Proceedings of the 19th SIGBioMed Workshop on Biomedical Language Processing</source>. <publisher-loc>Online</publisher-loc>: <publisher-name>Association for Computational Linguistics</publisher-name> (<year>2020</year>). p. <fpage>156</fpage>&#x02013;<lpage>66</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2020.bionlp-1.17">https://aclanthology.org/2020.bionlp-1.17</ext-link> (accessed November 23, 2023).</citation>
</ref>
<ref id="B33">
<label>33.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Farrell</surname> <given-names>S</given-names></name> <name><surname>Appleton</surname> <given-names>C</given-names></name> <name><surname>Noble</surname> <given-names>PJM</given-names></name> <name><surname>Al Moubayed</surname> <given-names>N</given-names></name></person-group>. <article-title>PetBERT: automated ICD-11 syndromic disease coding for outbreak detection in first opinion veterinary electronic health records</article-title>. <source>Sci Rep</source>. (<year>2023</year>) <volume>13</volume>:<fpage>18015</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-023-45155-7</pub-id><pub-id pub-id-type="pmid">37865683</pub-id></citation></ref>
<ref id="B34">
<label>34.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Blei</surname> <given-names>DM</given-names></name> <name><surname>Ng</surname> <given-names>AY</given-names></name> <name><surname>Jordan</surname> <given-names>MI</given-names></name></person-group>. <article-title>Latent dirichlet allocation</article-title>. <source>J Machine Learn Res</source>. (<year>2003</year>) <volume>3</volume>:<fpage>993</fpage>&#x02013;<lpage>1022</lpage>.</citation>
</ref>
<ref id="B35">
<label>35.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Rehurek</surname> <given-names>R</given-names></name> <name><surname>Sojka</surname> <given-names>P</given-names></name></person-group>. <source>Gensim&#x02013;Python Framework for Vector Space Modelling</source>. <publisher-loc>Brno</publisher-loc>: <publisher-name>NLP Centre, Faculty of Informatics, Masaryk University</publisher-name> (<year>2011</year>).</citation>
</ref>
<ref id="B36">
<label>36.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Noble</surname> <given-names>PJM</given-names></name> <name><surname>Appleton</surname> <given-names>C</given-names></name> <name><surname>Radford</surname> <given-names>AD</given-names></name> <name><surname>Nenadic</surname> <given-names>G</given-names></name></person-group>. <article-title>Using topic modelling for unsupervised annotation of electronic health records to identify an outbreak of disease in UK dogs</article-title>. <source>PLoS ONE</source>. (<year>2021</year>) <volume>16</volume>:<fpage>e0260402</fpage>. <pub-id pub-id-type="doi">10.1371/JOURNAL.PONE.0260402</pub-id><pub-id pub-id-type="pmid">34882714</pub-id></citation></ref>
<ref id="B37">
<label>37.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>L</given-names></name> <name><surname>Tang</surname> <given-names>L</given-names></name> <name><surname>Dong</surname> <given-names>W</given-names></name> <name><surname>Yao</surname> <given-names>S</given-names></name> <name><surname>Zhou</surname> <given-names>W</given-names></name></person-group>. <article-title>An overview of topic modeling and its current applications in bioinformatics</article-title>. <source>Springerplus</source>. (<year>2016</year>) <volume>5</volume>:<fpage>1608</fpage>. <pub-id pub-id-type="doi">10.1186/s40064-016-3252-8</pub-id><pub-id pub-id-type="pmid">27652181</pub-id></citation></ref>
<ref id="B38">
<label>38.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>P&#x000E9;rez</surname> <given-names>J</given-names></name> <name><surname>P&#x000E9;rez</surname> <given-names>A</given-names></name> <name><surname>Casillas</surname> <given-names>A</given-names></name> <name><surname>Gojenola</surname> <given-names>K</given-names></name></person-group>. <article-title>Cardiology record multi-label classification using latent Dirichlet allocation</article-title>. <source>Comput Methods Progr Biomed</source>. (<year>2018</year>) <volume>164</volume>:<fpage>111</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1016/j.cmpb.2018.07.002</pub-id><pub-id pub-id-type="pmid">30195419</pub-id></citation></ref>
<ref id="B39">
<label>39.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ghosh</surname> <given-names>S</given-names></name> <name><surname>Chakraborty</surname> <given-names>P</given-names></name> <name><surname>Nsoesie</surname> <given-names>EO</given-names></name> <name><surname>Cohn</surname> <given-names>E</given-names></name> <name><surname>Mekaru</surname> <given-names>SR</given-names></name> <name><surname>Brownstein</surname> <given-names>JS</given-names></name> <etal/></person-group>. <article-title>Temporal topic modeling to assess associations between news trends and infectious disease outbreaks</article-title>. <source>Sci Rep</source>. (<year>2017</year>) <volume>7</volume>:<fpage>40841</fpage>. <pub-id pub-id-type="doi">10.1038/srep40841</pub-id><pub-id pub-id-type="pmid">28102319</pub-id></citation></ref>
<ref id="B40">
<label>40.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Blei</surname> <given-names>DM</given-names></name> <name><surname>Lafferty</surname> <given-names>JD</given-names></name></person-group>. <article-title>Dynamic topic models</article-title>. In: <source>Proceedings of the 23rd International Conference on Machine Learning. ICML &#x00027;06</source>. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name> (<year>2006</year>). p. <fpage>113</fpage>&#x02013;<lpage>20</lpage>.</citation>
</ref>
<ref id="B41">
<label>41.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grootendorst</surname> <given-names>M</given-names></name></person-group>. <article-title>BERTopic: neural topic modeling with a class-based TF-IDF procedure</article-title>. arXiv [preprint] arXiv:220305794. (<year>2022</year>). <pub-id pub-id-type="doi">10.48550/arXiv.2203.05794</pub-id></citation>
</ref>
<ref id="B42">
<label>42.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Erlen</surname> <given-names>A</given-names></name> <name><surname>Potschka</surname> <given-names>H</given-names></name> <name><surname>Volk</surname> <given-names>HA</given-names></name> <name><surname>Sauter-Louis</surname> <given-names>C</given-names></name> <name><surname>O&#x00027;Neill</surname> <given-names>DG</given-names></name></person-group>. <article-title>Seizure occurrence in dogs under primary veterinary care in the UK: prevalence and risk factors</article-title>. <source>J Vet Intern Med</source>. (<year>2018</year>) <volume>32</volume>:<fpage>1665</fpage>&#x02013;<lpage>76</lpage>. <pub-id pub-id-type="doi">10.1111/jvim.15290</pub-id><pub-id pub-id-type="pmid">30216557</pub-id></citation></ref>
<ref id="B43">
<label>43.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nguyen</surname> <given-names>T</given-names></name></person-group>. <article-title>Total number of synapses in the adult human neocortex</article-title>. <source>Undergrad J Math Model</source>. (<year>2010</year>) <volume>3</volume>:<fpage>26</fpage>. <pub-id pub-id-type="doi">10.5038/2326-3652.3.1.26</pub-id></citation>
</ref>
<ref id="B44">
<label>44.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ouyang</surname> <given-names>L</given-names></name> <name><surname>Wu</surname> <given-names>J</given-names></name> <name><surname>Jiang</surname> <given-names>X</given-names></name> <name><surname>Almeida</surname> <given-names>D</given-names></name> <name><surname>Wainwright</surname> <given-names>CL</given-names></name> <name><surname>Mishkin</surname> <given-names>P</given-names></name> <etal/></person-group>. <article-title>Training language models to follow instructions with human feedback</article-title>. <source>arXiv preprint</source>. (<year>2022</year>). <pub-id pub-id-type="doi">10.48550/arXiv.2203.02155</pub-id></citation>
</ref>
<ref id="B45">
<label>45.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>S</given-names></name> <name><surname>Roller</surname> <given-names>S</given-names></name> <name><surname>Goyal</surname> <given-names>N</given-names></name> <name><surname>Artetxe</surname> <given-names>M</given-names></name> <name><surname>Chen</surname> <given-names>M</given-names></name> <name><surname>Chen</surname> <given-names>S</given-names></name> <etal/></person-group>. <article-title>OPT: open pre-trained transformer language models</article-title>. <source>arXiv preprint</source>. (<year>2022</year>). <pub-id pub-id-type="doi">10.48550/arXiv.2205.01068</pub-id><pub-id pub-id-type="pmid">38271138</pub-id></citation></ref>
<ref id="B46">
<label>46.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fins</surname> <given-names>IS</given-names></name> <name><surname>Davies</surname> <given-names>H</given-names></name> <name><surname>Farrell</surname> <given-names>S</given-names></name> <name><surname>Torres</surname> <given-names>JR</given-names></name> <name><surname>Pinchbeck</surname> <given-names>G</given-names></name> <name><surname>Radford</surname> <given-names>AD</given-names></name> <etal/></person-group>. <article-title>Evaluating ChatGPT text-mining of clinical records for obesity monitoring</article-title>. <source>Vet Record</source>. (<year>2023</year>) <volume>2023</volume>:<fpage>e3669</fpage>. <pub-id pub-id-type="doi">10.1002/vetr.3669</pub-id><pub-id pub-id-type="pmid">38058223</pub-id></citation></ref>
</ref-list>
</back>
</article>