<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Vet. Sci.</journal-id>
<journal-title>Frontiers in Veterinary Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Vet. Sci.</abbrev-journal-title>
<issn pub-type="epub">2297-1769</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fvets.2025.1660745</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Veterinary Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>An intelligent diagnostic method for porcine gastrointestinal infectious diseases based on multimodal AI and large language model</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wen</surname>
<given-names>Haiyan</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3124164/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Shi</surname>
<given-names>Hongtao</given-names>
</name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yu</surname>
<given-names>Jiashang</given-names>
</name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Fan</surname>
<given-names>Zhaobin</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dai</surname>
<given-names>Haicheng</given-names>
</name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jiang</surname>
<given-names>Lili</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Song</surname>
<given-names>Qinye</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1839628/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>College of Veterinary Medicine, Hebei Agricultural University</institution>, <addr-line>Baoding</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>College of Pharmacy, Heze University</institution>, <addr-line>Heze</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>School of Science and Information Science, Qingdao Agricultural University</institution>, <addr-line>Qingdao</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>College of Mathematics and Statistics, Heze University</institution>, <addr-line>Heze</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>Rizhao Jiacheng Animal Health Products Co., Ltd</institution>, <addr-line>Rizhao</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0001">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1631397/overview">Mengmeng Zhao</ext-link>, Foshan University, China</p>
</fn>
<fn fn-type="edited-by" id="fn0002">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1052398/overview">Yuanzhi Pan</ext-link>, Zhenjiang Hongxiang Automation Technology Co., Ltd, China</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3140531/overview">Haopu Li</ext-link>, Shanxi Agricultural University, China</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Qinye Song, <email>songqinye@126.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>05</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1660745</elocation-id>
<history>
<date date-type="received">
<day>06</day>
<month>07</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>25</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Wen, Shi, Yu, Fan, Dai, Jiang and Song.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wen, Shi, Yu, Fan, Dai, Jiang and Song</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The swine farming industry, a key pillar of Chinese animal husbandry, faces significant challenges due to frequent outbreaks of porcine gastrointestinal infectious diseases (PGID). Traditional diagnostic methods reliant on human expertise suffer from low efficiency, high subjectivity, and poor accuracy. To address these issues, this paper proposes a multimodal diagnostic method based on artificial intelligence (AI) and large language model (LLM) for six common types of PGID. In this method, ChatGPT and image augmentation techniques were first used to expand the dataset. Next, the Multi-scale TextCNN (MS-TextCNN) model was employed to capture multi-granularity semantic features from text. Subsequently, an improved Mask R-CNN model was applied to segment small intestine lesion regions, after which seven convolutional neural network (CNN) models were used to classify the segmented images. Finally, five machine learning models were utilized for multimodal classification diagnosis. Experimental results demonstrate that the multimodal diagnostic model can accurately identify six common types of PGID. This study provides an efficient and accurate intelligent solution for diagnosing PGID and demonstrates superior performance compared with single-modality methods.</p>
</abstract>
<kwd-group>
<kwd>porcine gastrointestinal infectious diseases</kwd>
<kwd>multimodal</kwd>
<kwd>artificial intelligence</kwd>
<kwd>large language model</kwd>
<kwd>Mask R-CNN</kwd>
<kwd>machine learning</kwd>
</kwd-group>
<counts>
<fig-count count="8"/>
<table-count count="9"/>
<equation-count count="5"/>
<ref-count count="49"/>
<page-count count="15"/>
<word-count count="9471"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Veterinary Infectious Diseases</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>The swine farming industry is a crucial component of Chinese animal husbandry. In 2024, China&#x2019;s pork production reached 57.06 million tons, accounting for 59.05% of the total output of pork, beef, mutton, and poultry (<xref ref-type="bibr" rid="ref1">1</xref>). The swine farming industry not only plays a vital role in ensuring the safe supply of meat but is also a significant industry related to national economic and social welfare, holding a pivotal position in Chinese agricultural production. During the swine farming process, various diseases frequently occur, with digestive tract infectious diseases being the most common and severe, representing one of the primary causes of piglet mortality (<xref ref-type="bibr" rid="ref2">2</xref>). Economic losses in swine farms due to digestive tract infectious diseases exceed 10 billion yuan annually, causing substantial financial damage to the swine farming industry (<xref ref-type="bibr" rid="ref3">3</xref>, <xref ref-type="bibr" rid="ref4">4</xref>). Early, rapid, and accurate diagnosis is critical for disease prevention and control in swine.</p>
<p>Traditional clinical diagnostic methods for PGID primarily rely on observing clinical symptoms, pathological changes, and epidemiological data to make preliminary diagnoses, or make confirmed diagnoses on some diseases that present typical and characteristic clinical signs. These methods heavily depend on the expertise and experience of frontline veterinarians, suffering from strong subjectivity, low efficiency, and poor accuracy. Additionally, the specialized skills and experience of veterinary experts are difficult to replicate quickly, leading to a shortage of qualified frontline veterinarians. In contrast, laboratory diagnostic methods utilize advanced detection technologies and sophisticated equipment to enable early diagnosis with high accuracy and robustness, making them one of the most commonly used and effective approaches for diagnosing swine diseases. Among these, multiplex quantitative PCR (qPCR) and enzyme-linked immunosorbent assay (ELISA) are widely applied in diagnosing PGID. Chen et al. developed a triplex qPCR for detecting porcine transmissible gastroenteritis virus, porcine epidemic diarrhea virus, and porcine delta coronavirus, achieving a clinical sample detection concordance rate of approximately 95% (<xref ref-type="bibr" rid="ref5">5</xref>). Yang et al. established an indirect ELISA using the COE protein of porcine epidemic diarrhea virus expressed in Pichia pastoris as the coating antigen to detect antibodies against porcine epidemic diarrhea in serum, with a detection concordance rate of up to 99.4% (<xref ref-type="bibr" rid="ref6">6</xref>). Although PCR and ELISA can achieve high diagnostic rates, these methods are complex, time-consuming, costly, and require specialized equipment and trained personnel to perform.</p>
<p>In recent years, with the rapid development of AI and image processing technologies, image recognition techniques based on deep learning have been widely applied in animal disease diagnosis (<xref ref-type="bibr" rid="ref7 ref8 ref9">7&#x2013;9</xref>). Kittichai et al. proposed an automated tool based on deep neural networks and image retrieval procedures for identifying Anaplasmosis, a common livestock disease, in microscopic images (<xref ref-type="bibr" rid="ref10">10</xref>). This method, utilizing the ResNeXt-50 model combined with the Triplet-Margin loss function, achieved an accuracy of 91.30% and a specificity of 92.83%. Muhammad Saqib et al. introduced a deep learning approach using the MobileNetV2 model and RMSprop optimizer for diagnosing lumpy skin disease in cattle (<xref ref-type="bibr" rid="ref11">11</xref>). This method demonstrated an accuracy of up to 95%, surpassing existing benchmark methods by 4&#x2013;10%. Yu et al. developed a deep learning model based on the YOLOv8 detection algorithm, utilizing kidney ultrasound images to classify the International Renal Interest Society (IRIS) stages of chronic kidney disease in dogs (<xref ref-type="bibr" rid="ref12">12</xref>). This model performed best in distinguishing IRIS stage 3 and above in canine chronic kidney disease, achieving an accuracy of 85%, significantly outperforming the 48&#x2013;62% accuracy of veterinary imaging experts. Buric et al. employed a U-Net architecture combined with backbone networks such as VGG, ResNet, Inception, and EfficientNet for diagnosing various canine ophthalmic diseases (<xref ref-type="bibr" rid="ref13">13</xref>). This model exhibited strong reliability, with an Intersection over Union score exceeding 80%, demonstrating high accuracy in the segmentation and diagnosis of canine eye diseases. Although these methods have shown significant success in animal disease diagnosis, their direct application to diagnosing porcine digestive tract remains challenging. The diagnosis of PGID is highly complex, requiring not only the identification of small intestine lesion characteristics from anatomical images but also the integration of textual case information for comprehensive analysis, involving the collaborative processing of multimodal image and text data. Currently, research on multimodal diagnostic techniques is primarily focused on human diseases (<xref ref-type="bibr" rid="ref14 ref15 ref16 ref17">14&#x2013;17</xref>), with no related academic reports in the field of PGID diagnosis.</p>
<p>Moreover, due to the sensitive nature of swine disease case information, which involves the interests of farms and the stability of the industry, collecting cases of PGID is challenging, leading to insufficient sample sizes for disease case data. This, in turn, causes issues such as low accuracy and overfitting in diagnostic models. Data augmentation is a key technology for addressing this problem, but traditional text augmentation methods (e.g., synonym replacement, word embeddings) have limited effectiveness. As an LLM, ChatGPT, with its powerful text generation and semantic understanding capabilities, excels in data augmentation, effectively tackling the challenges of insufficient training data and limited diversity. Dai et al. proposed the AugGPT method, which prompts ChatGPT to perform multiple rewrites of sentences, generating semantically similar but diversely expressed samples, significantly improving the accuracy and sample distribution diversity in text classification tasks (<xref ref-type="bibr" rid="ref18">18</xref>). Fang et al. utilized ChatGPT to generate synthetic text, enhancing the compositional generalization ability of open-intent detection models and improving their capability to handle unseen data (<xref ref-type="bibr" rid="ref19">19</xref>). Han et al. employed ChatGPT to generate synthetic data to reduce model bias, designing two strategies: targeted prompts and general prompts. The former is more effective but requires predefined bias types, while the latter is more broadly applicable (<xref ref-type="bibr" rid="ref20">20</xref>). These methods provide viable pathways for text data augmentation in the diagnosis of PGID.</p>
<p>Here, a multimodal AI and LLM-based method was proposed for diagnosing PGID. The method first employs LLMs and image augmentation techniques to enhance case sample data, then uses a MS-TextCNN to extract text features from case reports and adopts an improved Mask R-CNN combined with a CNN classification model to identify small intestine lesion features, and finally utilizes a machine learning model to perform disease classification and diagnosis based on multimodal text and image features, which will effectively address the issue of insufficient text data, and achieve high-precision disease diagnosis.</p>
</sec>
<sec sec-type="materials|methods" id="sec2">
<label>2</label>
<title>Materials and methods</title>
<p>The experimental workflow for diagnosing PGID using multimodal AI in this study is illustrated in <xref ref-type="fig" rid="fig1">Figure 1</xref>. Step A: Constructed a text dataset of PGID case information. This involved using text to describe the onset details of PGID cases, including age at onset, season of onset, disease progression, clinical signs, appetite status, and fecal characteristics. Step B: Constructed and annotated a dataset of swine anatomical images. Domain experts manually annotated the small intestine lesion regions in the anatomical images and assigned classification labels based on eight distinct small intestine lesion characteristics. Step C: Augmented the text dataset of PGID case information. Using ChatGPT-4, each text description was augmented to generate five new text samples that were semantically consistent but vary in expression style. Step D: Augmented the swine anatomical image dataset. New swine anatomical images were generated using rotation (90&#x00B0;, 180&#x00B0;, and 270&#x00B0;) and mirroring (horizontal and vertical), producing five new images per original image. Step E: Implemented disease prediction based on multimodal feature fusion for PGID. The augmented text and image datasets were merged as multi-source data, and each case was labeled with a disease tag based on laboratory disease detection results. MS-TextCNN and Mask R-CNN&#x202F;+&#x202F;CNN branch networks were constructed to extract text and image features, respectively. The extracted features are concatenated, feature-level fusion is performed, machine learning classification is performed, and the classification results are evaluated.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Flow chart of porcine gastrointestinal infectious diseases diagnosis by multimodal AI and LLM.</p>
</caption>
<graphic xlink:href="fvets-12-1660745-g001.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Flowchart illustrating the collection and diagnostic workflow of porcine gastrointestinal infectious disease cases. The process begins with case collection and information recording, followed by lesion image annotation and laboratory testing. Dissection images of pigs display labeled lesion sites. Case text information is expanded using a GPT-4 model, after which image enhancement is performed. Datasets are merged with disease labels added. An MS-TextCNN and a Mask R-CNN+CNN network are constructed to extract text features and image features, respectively. The extracted features are fused for machine learning classification and evaluation.</alt-text>
</graphic>
</fig>
<sec id="sec3">
<label>2.1</label>
<title>Data collection and preprocessing</title>
<sec id="sec4">
<label>2.1.1</label>
<title>Dataset</title>
<p>The swine disease data used in this study were sourced from the Animal Disease Research Institute of Heze University, collected from July to December 2023. The dataset comprised 106 confirmed cases of swine disease, covering 6 common types of PGID: porcine epidemic diarrhea (PED), transmissible gastroenteritis of pigs (TGE), porcine proliferative enteropathy (PPE), yellow scour of newborn piglets (YSNP), white scour of piglets (WSP), and clostridial enteritis of piglets (CEP). Details are provided in <xref ref-type="table" rid="tab1">Table 1</xref>.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Dataset details of PGID.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Disease type</th>
<th align="left" valign="top">Disease description</th>
<th align="center" valign="top">Number</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">PED</td>
<td align="left" valign="middle">Porcine epidemic diarrhea, caused by <italic>porcine epidemic diarrhea virus</italic>, characterized by watery diarrhea and vomiting (<xref ref-type="bibr" rid="ref46">46</xref>)</td>
<td align="center" valign="middle">46</td>
</tr>
<tr>
<td align="left" valign="middle">TGE</td>
<td align="left" valign="middle">Transmissible gastroenteritis of pigs, caused by <italic>transmissible gastroenteritis virus</italic>, characterized by vomiting, severe diarrhea, and high mortality in 2&#x2013;3-week-old piglets (<xref ref-type="bibr" rid="ref47">47</xref>)</td>
<td align="center" valign="middle">27</td>
</tr>
<tr>
<td align="left" valign="middle">PPE</td>
<td align="left" valign="middle">Porcine proliferative enteropathy, caused by <italic>Lawsonia intracellularis</italic>, characterized by proliferation of crypt epithelial cells in the ileum and colon, leading to thickened intestinal mucosa (<xref ref-type="bibr" rid="ref48">48</xref>)</td>
<td align="center" valign="middle">8</td>
</tr>
<tr>
<td align="left" valign="middle">YSNP</td>
<td align="left" valign="middle">Yellow scour of newborn piglets, caused by pathogenic <italic>Escherichia coli</italic>, characterized by severe diarrhea, yellow watery feces, and rapid death (<xref ref-type="bibr" rid="ref49">49</xref>)</td>
<td align="center" valign="middle">12</td>
</tr>
<tr>
<td align="left" valign="middle">WSP</td>
<td align="left" valign="middle">White scour of piglets, also caused by pathogenic <italic>Escherichia coli</italic>, characterized by milky-white or grayish pasty feces (<xref ref-type="bibr" rid="ref49">49</xref>)</td>
<td align="center" valign="middle">9</td>
</tr>
<tr>
<td align="left" valign="middle">CEP</td>
<td align="left" valign="middle">Clostridial enteritis of piglets, caused by <italic>Clostridium perfringens</italic> type C, characterized by red feces, small intestine mucosal hemorrhage and necrosis, with rapid onset, short disease course, and high mortality (<xref ref-type="bibr" rid="ref4">4</xref>)</td>
<td align="center" valign="middle">4</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Each case included data in two different formats: a textual description of the swine&#x2019;s disease onset information and anatomical images of the swine containing small intestine lesion regions. The textual description of the onset information covered details such as age at onset, season of onset, disease progression, clinical signs, appetite status, and fecal characteristics. The swine anatomical images included the lesion regions of the small intestine, with lesion characteristics primarily classified into 8 categories: small intestinal serous membrane congestion (SISMC), small intestinal wall bleeding (SIWB), hemorrhage and thinning of small intestinal wall (HTSIW), thinning of small intestinal wall (TSIW), small intestinal wall hyperplasia (SIWH), small intestinal mucosal hyperplasia (SIMHp), hemorrhage and hyperplasia of small intestinal mucosa (HHSIM), and small intestinal mucosal hemorrhage (SIMHh) (see <xref ref-type="fig" rid="fig2">Figure 2</xref> for details).</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Lesion features in porcine small intestine. SISMC, the serosal layer shows dilated and engorged capillaries, appearing bright red with prominent texture; SIWB, Damaged mucosal capillaries cause blood to enter the intestinal lumen, with affected segments appearing dark red; TSIW, the intestinal wall becomes thin, even transparent, losing its original toughness and elasticity; HTSIW, the intestinal wall appears red, becomes thin and transparent, and loses its toughness; SIWH, the intestinal wall thickens, resembling a soft hose, with a rough or granular surface; SIMHp, the mucosal surface thickens, exhibiting longitudinal and transverse wrinkles with an uneven surface; HHSIM, the mucosa appears red or dark red, with thickened layers, a rough surface, and accompanying wrinkles; SIMHh, the mucosal surface shows bright red or dark red dotted, patchy, or diffuse hemorrhages.</p>
</caption>
<graphic xlink:href="fvets-12-1660745-g002.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Porcine dissection image highlighting the small intestine area. Lesions of the small intestine are categorized into eight types: SISMC, SIWB, TSIW, HTSIW, SIWH, SIMHp, HHSIM, and SIMHh.</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec5">
<label>2.1.2</label>
<title>Data preprocessing</title>
<p>To mitigate the issues of classification instability and overfitting caused by insufficient data samples, this section applied data augmentation to both the textual descriptions of swine disease case information and the anatomical images of swine. For text augmentation, ChatGPT was utilized for natural language generation. ChatGPT, built on the GPT-4 architecture, is an autoregressive language model with a core Transformer decoder structure. Trained on large-scale corpora through unsupervised pre-training, it possesses robust semantic modeling and language generation capabilities. Without requiring fine-tuning, ChatGPT can generate high-quality text samples that maintain semantic consistency but vary in expression style through input prompts. The conditional probability formula of ChatGPT&#x2019;s language model is presented in <xref ref-type="disp-formula" rid="EQ1">Equation 1</xref>:</p>
<disp-formula id="EQ1">
<label>(1)</label>
<mml:math id="M1">
<mml:mi>P</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mo>&#x220F;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mi>P</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2223;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mn>...</mml:mn>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math id="M2">
<mml:mi>x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</inline-formula> represents the generated text sequence, and <inline-formula>
<mml:math id="M3">
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> is the i-th word or token. By maximizing this conditional probability distribution, ChatGPT generates complete sentences, achieving both semantic preservation and diversity in expression. For each original clinical case description, a prompt (&#x201C;Based on the text above, generate 5 sentences that have the same meaning but different expressions.&#x201D;) was constructed and inputted into ChatGPT to generate five semantically consistent but stylistically diverse text samples, thereby expanding the dataset. Through this method, the original text samples were expanded from 106 to 636, significantly enhancing corpus richness and linguistic variability, thus improving the model&#x2019;s ability to recognize and understand different expression styles.</p>
<p>Image augmentation employs various geometric transformation operations, including rotation by 90&#x00B0;, 180&#x00B0;, and 270&#x00B0;, as well as horizontal and vertical mirroring. These methods significantly increased the diversity of image samples while ensuring that the semantic characteristics of small intestine lesions remained unchanged. The augmented image dataset was also expanded to 6 times the original size, improving the model&#x2019; generalization ability across different image angles, orientations, and visual perturbations. Image augmentation not only increased the scale of training data but also provided a more comprehensive feature representation space during training, enhancing the robust identification of lesion regions.</p>
<p>Sequencely, the Labelme software was used to manually annotate the augmented images, accurately delineating the boundaries of small intestinal lesion areas and explicitly labeling their lesion types. During the model testing phase, 5-fold cross-validation was employed for objective evaluation to comprehensively assess the model&#x2019;s generalization ability and stability.</p>
</sec>
</sec>
<sec id="sec6">
<label>2.2</label>
<title>Multi-scale TextCNN</title>
<p>TextCNN is a CNN-based text classification model widely used in natural language processing tasks (<xref ref-type="bibr" rid="ref21">21</xref>). Its core principle involves representing text as a word embedding matrix, capturing local semantic features (e.g., n-gram patterns) through convolutional filters, retaining prominent features via max-pooling, and outputting classification results through fully connected layers and a softmax layer (<xref ref-type="bibr" rid="ref22">22</xref>). TextCNN is renowned for its efficient feature extraction and robustness in handling short texts, making it suitable for classifying case information descriptions in porcine digestive tract infectious disease diagnosis.</p>
<p>The MS-TextCNN proposed in this paper (shown in <xref ref-type="fig" rid="fig3">Figure 3</xref>) enhanced the modeling capability for complex texts by introducing convolutional kernels of multiple sizes to extract semantic features of varying lengths in parallel. In the context of porcine gastrointestinal infectious disease diagnosis, the model took 100-dimensional word embeddings as input, employed 128 multi-scale filters, and integrated batch normalization, Rectified Linear Unit (ReLU) activation, Dropout, and global max-pooling to output a 6-class classification.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Disease diagnosis based on MS-TextCNN.</p>
</caption>
<graphic xlink:href="fvets-12-1660745-g003.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Diagram illustrating the use of a multi-scale TextCNN to analyze disease cases in piglets. Symptom text is tokenized into a vocabulary of 819 words, converted into word embeddings, and then transformed into integer sequences defined by sample size and sequence length. These sequences are fed into a TextCNN model comprising convolution layers, BatchNorm, ReLU, Dropout, MaxPooling, a fully connected layer, and a softmax layer, which outputs the classification results.</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec7">
<label>2.3</label>
<title>Improved Mask R-CNN</title>
<p>In practical diagnosis, small intestine lesion regions typically occupy only a small portion of anatomical images (as shown in <xref ref-type="fig" rid="fig2">Figure 2</xref>). Directly using a CNN model for whole-image classification is susceptible to interference from irrelevant background, which affects recognition accuracy. Therefore, it is necessary to first use an image detection model (e.g., Mask R-CNN) to locate and extract lesion regions before performing image classification to enhance the model&#x2019;s recognition performance. Compared to single-stage detection algorithms like YOLO, Mask R-CNN employs a two-stage detection mechanism, making it superior in small target detection and high-precision tasks.</p>
<p>To improve the recognition and segmentation accuracy of small intestine lesion regions, this study optimized the original Mask R-CNN model by incorporating the High-Resolution Network (HRNet) as the backbone network and embedding the Convolutional Block Attention Module (CBAM) attention mechanism during the feature extraction stage. HRNet can extract rich semantic features while maintaining spatial resolution, effectively preserving detailed information of lesion regions. CBAM, through its channel and spatial attention mechanisms, guides the model to focus on more discriminative feature regions. This improved structure enhanced the model&#x2019;s perception and segmentation accuracy for small target lesion regions while retaining Mask R-CNN&#x2019;s multi-task detection and segmentation capabilities.</p>
<p>The improved model, as shown in <xref ref-type="fig" rid="fig4">Figure 4</xref>, replaces the original ResNet-FPN backbone with HRNet and incorporates CBAM modules in key output layers to enhance feature representation. After image input, HRNet first extracts multi-scale high-resolution features, which are then enhanced by the CBAM attention module in both channel and spatial dimensions. Subsequently, the Region Proposal Network (RPN) generates candidate bounding boxes on the feature map for preliminary target localization. The candidate regions are aligned using ROIAlign and fed into three branches: a fully connected layer for classification, a regression layer for bounding box regression, and a fully convolutional network (FCN) for mask prediction. The final output includes the category, bounding box, and corresponding pixel-level segmentation mask for each candidate region.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Structure of the improved Mask R-CNN.</p>
</caption>
<graphic xlink:href="fvets-12-1660745-g004.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Diagram of a computer vision architecture combining HRNet, CBAM, and RPN modules. HRNet involves downsampling and upsampling layers with connections forming a network. CBAM includes input and output features processed by channel and spatial attention modules. RPN processes features through convolution for classification and regression. ROI Align integrates these into classification, bounding box, and mask heads, producing class labels, bounding boxes, and segmentation masks.</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec8">
<label>2.4</label>
<title>CNN classification models</title>
<p>This study selected 7 classic CNN classification models for classifying segmented images. Alex Krizhevsky&#x2019;s Convolutional Neural Network (AlexNet) employs large convolutional kernels and overlapping pooling layers, combined with ReLU activation and Dropout techniques, to effectively enhance image feature extraction capabilities, achieving groundbreaking results in the 2012 ImageNet challenge (<xref ref-type="bibr" rid="ref23">23</xref>). Visual Geometry Group Network (VGGNet) progressively deepens the network using multiple 3&#x202F;&#x00D7;&#x202F;3 convolutional layers and pooling layers, capturing rich features from low to high levels, and demonstrates outstanding performance in various image recognition tasks (<xref ref-type="bibr" rid="ref24">24</xref>). GoogleNet utilizes Inception modules to apply convolutional kernels of different sizes in parallel, capturing multi-scale features, while replacing fully connected layers with global average pooling to reduce computational complexity and improve efficiency (<xref ref-type="bibr" rid="ref25">25</xref>). Residual Network (ResNet) introduces residual learning and skip connections, enabling effective training of deep networks, addressing the vanishing gradient problem, and excelling in image recognition tasks (<xref ref-type="bibr" rid="ref26">26</xref>). Dense Convolutional Network (DenseNet) employs dense connections, allowing each layer to receive feature maps from all preceding layers, significantly improving information flow, reducing the vanishing gradient issue, and enhancing model generalization and efficiency (<xref ref-type="bibr" rid="ref27">27</xref>). EfficientNet is a convolutional neural network architecture that simultaneously balances the depth, width, and resolution of the network through composite coefficients, significantly reducing model parameters and computational complexity while achieving higher accuracy (<xref ref-type="bibr" rid="ref28">28</xref>). Vision Transformer divides images into patches and models them using a pure Transformer encoder, excelling at capturing global features and demonstrating image recognition performance comparable to or even superior to CNNs when trained on a large scale (<xref ref-type="bibr" rid="ref29">29</xref>).</p>
</sec>
<sec id="sec9">
<label>2.5</label>
<title>Machine learning models</title>
<p>This study selected 5 classic machine learning classification algorithms for the final disease classification. Naive Bayes (NB), based on Bayes&#x2019; theorem, calculates class probabilities through feature independence assumptions, making it suitable for high-dimensional sparse data with advantages of efficient computation and ease of implementation (<xref ref-type="bibr" rid="ref30">30</xref>). K-Nearest Neighbors (KNN) performs classification through majority voting based on sample distances, offering an intuitive and interpretable decision process, particularly suitable for small-scale datasets (<xref ref-type="bibr" rid="ref31">31</xref>). Support Vector Machine (SVM) constructs a hyperplane by maximizing the classification margin, effectively handling high-dimensional and nonlinear problems, and performs exceptionally well with small sample datasets (<xref ref-type="bibr" rid="ref32">32</xref>). Random Forest (RF) integrates multiple decision trees, enabling automatic feature selection and reducing overfitting, with strong robustness suitable for complex datasets (<xref ref-type="bibr" rid="ref33">33</xref>). eXtreme Gradient Boosting (XGBoost) employs a gradient boosting framework with regularization and optimized feature splitting, significantly improving accuracy and efficiency in modeling large-scale datasets and nonlinear relationships (<xref ref-type="bibr" rid="ref34">34</xref>).</p>
</sec>
<sec id="sec10">
<label>2.6</label>
<title>Evaluation metrics</title>
<p>This study adopted accuracy, precision, recall, and F1 score as evaluation metrics for model performance. Accuracy measures the overall classification prediction performance of the model, calculated as shown in <xref ref-type="disp-formula" rid="EQ2">Equation 2</xref>:</p>
<disp-formula id="EQ2">
<label>(2)</label>
<mml:math id="M4">
<mml:mtext mathvariant="italic">Accuracy</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="italic">TP</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="italic">TN</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">TP</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="italic">TN</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="italic">FP</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="italic">FN</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula>
<p>Where TP represents true positives (correctly predicted positive samples), TN represents true negatives (correctly predicted negative samples), FP represents false positives (negative samples incorrectly predicted as positive), and FN represents false negatives (positive samples incorrectly predicted as negative). Accuracy reflects the overall correctness of predictions across all sample categories.</p>
<p>Precision focuses on the proportion of true positives among samples predicted as positive, calculated as shown in <xref ref-type="disp-formula" rid="EQ3">Equation 3</xref>:</p>
<disp-formula id="EQ3">
<label>(3)</label>
<mml:math id="M5">
<mml:mtext mathvariant="italic">Precision</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mi mathvariant="italic">TP</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">TP</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="italic">FP</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula>
<p>A high precision indicates a low false positive rate.</p>
<p>Recall measures the proportion of actual positive samples correctly identified, calculated as shown in <xref ref-type="disp-formula" rid="EQ4">Equation 4</xref>:</p>
<disp-formula id="EQ4">
<label>(4)</label>
<mml:math id="M6">
<mml:mtext mathvariant="italic">Recall</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mi mathvariant="italic">TP</mml:mi>
<mml:mrow>
<mml:mi mathvariant="italic">TP</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="italic">FN</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula>
<p>A high recall indicates a low false negative rate.</p>
<p>The F1 score is the weighted harmonic mean of precision and recall, calculated as shown in <xref ref-type="disp-formula" rid="EQ5">Equation 5</xref>:</p>
<disp-formula id="EQ5">
<label>(5)</label>
<mml:math id="M7">
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>=</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext mathvariant="italic">Precision</mml:mtext>
<mml:mo>&#x00D7;</mml:mo>
<mml:mtext mathvariant="italic">Recall</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">Precision</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext mathvariant="italic">Recall</mml:mtext>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula>
<p>Its value ranges from 0 to 1, with a higher value indicating a better balance between the two types of errors.</p>
</sec>
<sec id="sec11">
<label>2.7</label>
<title>Experimental environment setup</title>
<p>The computer used in this study is equipped with an Intel Core i5-12400F CPU, 32GB RAM, and an Nvidia RTX 4060 GPU. During training, the Adam optimizer was used with an initial learning rate of 0.0001, a cross-entropy loss function, a batch size of 32, 200 training epochs, and early stopping set to 30. Experiments confirmed that this was sufficient for effective training.</p>
</sec>
</sec>
<sec id="sec12">
<label>3</label>
<title>Results and analysis</title>
<sec id="sec13">
<label>3.1</label>
<title>Mask R-CNN detection results</title>
<p>This study employed an improved Mask R-CNN network to detect and segment lesion regions in small intestine anatomical images. To optimize the detection performance of Mask R-CNN, three classification methods were used to annotate lesion regions, as shown in <xref ref-type="fig" rid="fig5">Figure 5</xref>. The first method was based on eight true lesion characteristics provided by domain experts. The second method merged these into six categories. The third method further consolidated the six categories into three. The choice of classification labels directly impacts Mask R-CNN&#x2019;s detection performance, and merging similar categories can reduce inter-class uncertainty, thereby improving the model&#x2019;s recognition performance and generalization ability.</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>Classification relationship of porcine small intestine lesion image.</p>
</caption>
<graphic xlink:href="fvets-12-1660745-g005.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Diagram showing small intestinal lesion images categorized into three hierarchical levels: eight categories (SISMC, SIWB, HTSIW, TSIW, SIWH, SIMHp, HHSIM, SIMHh), condensed into six categories (SISMC, SIWB, TSIW, SIWH, SIMHp, SIMHh), and further condensed into three categories (NDSIWL,SIMHp, SIMHh). Arrows link the categories, indicating the hierarchical relationship from the highest to the lowest level.</alt-text>
</graphic>
</fig>
<p><xref ref-type="table" rid="tab2">Table 2</xref> presents the detection performance of Mask R-CNN under different classification methods. In the 8-class scenario, except for &#x201C;TSIW&#x201D; and &#x201C;SIWH,&#x201D; the Precision, Recall, and F1 scores for all other categories were 0, with an overall accuracy (OA) of only 0.0845. This suggests that overly fine-grained category divisions may hinder the model&#x2019;s ability to effectively identify lesion types. In the 6-class scenario, detection performance improved for some categories, with &#x201C;SIWB&#x201D; achieving a Recall of 0.8125 and an F1 score of 0.5226, though the OA remained low at 0.287. In the 3-class scenario, detection performance significantly improved across all categories, particularly for &#x201C;Non-dissected small intestinal wall lesions (NDSIWL),&#x201D; where Precision, Recall, and F1 scores reached 0.9277, and OA increased to 0.8202. This indicates that merging similar categories enhances Mask R-CNN&#x2019;s detection performance and improves the model&#x2019;s generalization ability.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Detection results of the improved Mask R-CNN under different classification methods.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Category number</th>
<th align="center" valign="top">Metric</th>
<th align="left" valign="top">SISMC /NDSIWL</th>
<th align="left" valign="top">SIMHh</th>
<th align="left" valign="top">SIMHp</th>
<th align="left" valign="top">TSIW</th>
<th align="left" valign="top">SIWH</th>
<th align="left" valign="top">SIWB</th>
<th align="left" valign="top">HTSIW</th>
<th align="left" valign="top">HHSIM</th>
<th align="left" valign="top">OA</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" rowspan="3">8</td>
<td align="center" valign="top">Precision</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0.2632</td>
<td align="center" valign="top">0.2727</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0</td>
<td align="center" valign="middle" rowspan="3">0.0845</td>
</tr>
<tr>
<td align="center" valign="top">Recall</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0.3191</td>
<td align="center" valign="top">0.3000</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0</td>
</tr>
<tr>
<td align="center" valign="top">F1</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0.2885</td>
<td align="center" valign="top">0.2857</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="3">6</td>
<td align="center" valign="top">Precision</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0.75</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0.3852</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle" rowspan="3">0.2870</td>
</tr>
<tr>
<td align="center" valign="top">Recall</td>
<td align="center" valign="top">0.1282</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0.075</td>
<td align="center" valign="top">0.0426</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0.8125</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
</tr>
<tr>
<td align="center" valign="top">F1</td>
<td align="center" valign="top">0.2273</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0.1364</td>
<td align="center" valign="top">0.0816</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0.5226</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="3">3</td>
<td align="center" valign="top">Precision</td>
<td align="center" valign="top">0.9277</td>
<td align="center" valign="top">0.5238</td>
<td align="center" valign="top">0.7333</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle" rowspan="3">0.8202</td>
</tr>
<tr>
<td align="center" valign="top">Recall</td>
<td align="center" valign="top">0.9277</td>
<td align="center" valign="top">0.5500</td>
<td align="center" valign="top">0.4681</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
</tr>
<tr>
<td align="center" valign="top">F1</td>
<td align="center" valign="top">0.9277</td>
<td align="center" valign="top">0.5366</td>
<td align="center" valign="top">0.5714</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To further analyze the recognition performance of Mask R-CNN, <xref ref-type="table" rid="tab3">Table 3</xref> presents the detection rates under different classification methods to evaluate whether Mask R-CNN successfully detects lesion regions even when it fails to classify them correctly. The results show that in the 8-class scenario, the detection rates for all categories were low, with an overall detection rate of only 0.3521. In the 6-class scenario, the overall detection rate improved to 0.6620, with &#x201C;SIWB&#x201D; reaching a detection rate of 0.8281, and other categories also showing improved detection rates, indicating that merging similar categories enhances the ability to detect target regions. In the 3-class scenario, the detection rates for all categories significantly improved, particularly for &#x201C;NDSIWL&#x201D; and &#x201C;SIMHh,&#x201D; which achieved detection rates of 0.9814 and 0.9000, respectively. The overall detection rate increased to 0.9430, demonstrating that broader category merging significantly enhances the detection capability for lesion regions.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Detection rate of the improved Mask R-CNN under different classification methods.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Category Number</th>
<th align="center" valign="top">SISMC /NDSIWL</th>
<th align="center" valign="top">SIMHh</th>
<th align="center" valign="top">SIMHp</th>
<th align="center" valign="top">TSIW</th>
<th align="center" valign="top">SIWH</th>
<th align="center" valign="top">SIWB</th>
<th align="center" valign="top">HTSIW</th>
<th align="center" valign="top">HHSIM</th>
<th align="center" valign="top">OA</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">8</td>
<td align="center" valign="middle">0.2778</td>
<td align="center" valign="middle">0.1875</td>
<td align="center" valign="middle">0.2273</td>
<td align="center" valign="middle">0.4894</td>
<td align="center" valign="middle">0.3000</td>
<td align="center" valign="middle">0.4000</td>
<td align="center" valign="middle">0.5714</td>
<td align="center" valign="middle">0.1667</td>
<td align="center" valign="middle">0.3521</td>
</tr>
<tr>
<td align="left" valign="middle">6</td>
<td align="center" valign="middle">0.7436</td>
<td align="center" valign="middle">0.1875</td>
<td align="center" valign="middle">0.4250</td>
<td align="center" valign="middle">0.7234</td>
<td align="center" valign="middle">0.7000</td>
<td align="center" valign="middle">0.8281</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">0.6620</td>
</tr>
<tr>
<td align="left" valign="middle">3</td>
<td align="center" valign="middle">0.9814</td>
<td align="center" valign="middle">0.9000</td>
<td align="center" valign="middle">0.8298</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">0.9430</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Combining the results from <xref ref-type="table" rid="tab2">Tables 2</xref>, <xref ref-type="table" rid="tab3">3</xref>, it is evident that in the 3-class scenario, Mask R-CNN performs best in terms of both recognition accuracy and detection rate. However, the 3-class approach cannot fully describe the true lesion characteristics of the affected regions. Therefore, it is necessary to introduce a CNN network for secondary classification of the segmented images.</p>
</sec>
<sec id="sec14">
<label>3.2</label>
<title>CNN classification results</title>
<p>To investigate the impact of different classification granularities, the experiment was conducted with two approaches: 8-class and 6-class classifications, while the 3-classification method was solely used for image segmentation of lesion regions. Classic CNN models described in section 2.4 were trained and tested on the segmented images. The experimental results, as shown in <xref ref-type="table" rid="tab4">Table 4</xref>, indicate that all CNN models achieved higher classification accuracy on the pixel-level segmented images, with the OA of the 6-class approach generally outperforming the 8-class approach. Although the 8-class approach can more finely characterize lesion features, it also increases classification difficulty, resulting in slightly lower accuracy. For example, GoogleNet achieved an accuracy of 96.74% in the 8-class scenario, which further improved to 97.67% in the 6-class scenario. Similarly, DenseNet&#x2019;s accuracy increased from 95.81 to 96.74%, Vision Transformer&#x2019;s accuracy increased from 94.88 to 96.74%. This suggests that for fine-grained lesion classification tasks, moderately merging similar categories can reduce the model&#x2019;s learning difficulty, thereby achieving better classification performance.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>Classification results of improved Mask R-CNN-segmented images by different CNN models.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top" rowspan="2">Models</th>
<th align="center" valign="top" colspan="4">8 Categories</th>
<th align="center" valign="top" colspan="4">6 Categories</th>
</tr>
<tr>
<th align="center" valign="top">Acc</th>
<th align="center" valign="top">P</th>
<th align="center" valign="top">R</th>
<th align="center" valign="top">F1</th>
<th align="center" valign="top">Acc</th>
<th align="center" valign="top">P</th>
<th align="center" valign="top">R</th>
<th align="center" valign="top">F1</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">AlexNet</td>
<td align="center" valign="middle">0.8326</td>
<td align="center" valign="top">0.8000</td>
<td align="center" valign="top">0.8189</td>
<td align="center" valign="top">0.8038</td>
<td align="center" valign="middle">0.9023</td>
<td align="center" valign="top">0.8867</td>
<td align="center" valign="top">0.8983</td>
<td align="center" valign="top">0.8883</td>
</tr>
<tr>
<td align="left" valign="middle">DenseNet</td>
<td align="center" valign="middle">0.9581</td>
<td align="center" valign="top">0.9438</td>
<td align="center" valign="top">0.9575</td>
<td align="center" valign="top">0.9500</td>
<td align="center" valign="middle">0.9674</td>
<td align="center" valign="top">0.9683</td>
<td align="center" valign="top">0.9617</td>
<td align="center" valign="top">0.9683</td>
</tr>
<tr>
<td align="left" valign="middle">GoogleNet</td>
<td align="center" valign="middle">0.9674</td>
<td align="center" valign="top">0.9613</td>
<td align="center" valign="top">0.9675</td>
<td align="center" valign="top">0.9613</td>
<td align="center" valign="middle">0.9767</td>
<td align="center" valign="top">0.9717</td>
<td align="center" valign="top">0.9750</td>
<td align="center" valign="top">0.9733</td>
</tr>
<tr>
<td align="left" valign="middle">ResNet</td>
<td align="center" valign="middle">0.8047</td>
<td align="center" valign="top">0.7950</td>
<td align="center" valign="top">0.7750</td>
<td align="center" valign="top">0.7700</td>
<td align="center" valign="middle">0.8791</td>
<td align="center" valign="top">0.8441</td>
<td align="center" valign="top">0.8800</td>
<td align="center" valign="top">0.8617</td>
</tr>
<tr>
<td align="left" valign="middle">VGGNet</td>
<td align="center" valign="middle">0.8279</td>
<td align="center" valign="top">0.8463</td>
<td align="center" valign="top">0.7600</td>
<td align="center" valign="top">0.7500</td>
<td align="center" valign="middle">0.8465</td>
<td align="center" valign="top">0.8400</td>
<td align="center" valign="top">0.8250</td>
<td align="center" valign="top">0.8217</td>
</tr>
<tr>
<td align="left" valign="middle">EfficientNet</td>
<td align="center" valign="middle">0.9209</td>
<td align="center" valign="top">0.9239</td>
<td align="center" valign="top">0.8937</td>
<td align="center" valign="top">0.9052</td>
<td align="center" valign="middle">0.9256</td>
<td align="center" valign="top">0.9170</td>
<td align="center" valign="top">0.8966</td>
<td align="center" valign="top">0.9053</td>
</tr>
<tr>
<td align="left" valign="middle">Vision Transformer</td>
<td align="center" valign="middle">0.9488</td>
<td align="center" valign="top">0.9342</td>
<td align="center" valign="top">0.9571</td>
<td align="center" valign="top">0.9423</td>
<td align="center" valign="middle">0.9674</td>
<td align="center" valign="top">0.9659</td>
<td align="center" valign="top">0.9419</td>
<td align="center" valign="top">0.9524</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec15">
<label>3.3</label>
<title>Multimodal disease recognition</title>
<p>This study employed five representative machine learning models described in Section 2.5 to evaluate the classification performance of different feature modalities for PGID diagnosis. The classifiers were first applied to the text features (10 dimensions) extracted by MS-TextCNN, with results summarized in <xref ref-type="table" rid="tab5">Table 5</xref>. Among all models, RF achieved the best overall performance, reaching the highest accuracy 0.8479, precision 0.9198, recall 0.8564, and F1 score 0.8870. In contrast, NB performed the worst, with an accuracy of only 0.6261, despite showing relatively high precision 0.7821. The remaining models demonstrated generally good performance but were slightly less accurate and stable compared with RF. These results indicate that text-based clinical features are highly discriminative.</p>
<table-wrap position="float" id="tab5">
<label>Table 5</label>
<caption>
<p>Classification results of text features by different machine learning models.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th>Models</th>
<th align="center" valign="top">Acc</th>
<th align="center" valign="top">
<italic>P</italic>
</th>
<th align="center" valign="top">
<italic>R</italic>
</th>
<th align="center" valign="top">F1</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">NB</td>
<td align="center" valign="top">0.6261</td>
<td align="center" valign="top">0.7821</td>
<td align="center" valign="top">0.7369</td>
<td align="center" valign="top">0.7588</td>
</tr>
<tr>
<td align="left" valign="top">KNN</td>
<td align="center" valign="top">0.8132</td>
<td align="center" valign="top">0.8668</td>
<td align="center" valign="top">0.8213</td>
<td align="center" valign="top">0.8434</td>
</tr>
<tr>
<td align="left" valign="top">SVM</td>
<td align="center" valign="top">0.8399</td>
<td align="center" valign="top">0.9073</td>
<td align="center" valign="top">0.8481</td>
<td align="center" valign="top">0.8767</td>
</tr>
<tr>
<td align="left" valign="top">RF</td>
<td align="center" valign="top">0.8479</td>
<td align="center" valign="top">0.9198</td>
<td align="center" valign="top">0.8564</td>
<td align="center" valign="top">0.8870</td>
</tr>
<tr>
<td align="left" valign="top">XGBoost</td>
<td align="center" valign="top">0.8399</td>
<td align="center" valign="top">0.8798</td>
<td align="center" valign="top">0.8279</td>
<td align="center" valign="top">0.8531</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Then, the classifiers were applied to the image features (4 dimensions) extracted by Mask R-CNN&#x202F;+&#x202F;CNN under both 8-class and 6-class encodings, and the performance of all classifiers declined considerably, as shown in <xref ref-type="table" rid="tab6">Table 6</xref>. The best results again came from RF, which achieved the highest accuracy 0.5526 for 8-class and 0.5476 for 6-class, with F1 scores around 0.58. At the other extreme, NB remained the weakest, with accuracies of 0.3987 for 8-class and 0.3949 for 6-class, and F1 scores near 0.41. The other models exhibited intermediate performance between the best and worst results, revealing certain limitations relative to the RF model. Overall, the results confirm that compared with text features, image features alone are less discriminative and less reliable for PGID classification.</p>
<table-wrap position="float" id="tab6">
<label>Table 6</label>
<caption>
<p>Classification results of image features by different machine learning models.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th rowspan="2">Models</th>
<th align="center" valign="top" colspan="4">8-Class image features</th>
<th align="center" valign="top" colspan="4">6-Class image features</th>
</tr>
<tr>
<th align="center" valign="top">Acc</th>
<th align="center" valign="top">
<italic>P</italic>
</th>
<th align="center" valign="top">
<italic>R</italic>
</th>
<th align="center" valign="top">F1</th>
<th align="center" valign="top">Acc</th>
<th align="center" valign="top">
<italic>P</italic>
</th>
<th align="center" valign="top">
<italic>R</italic>
</th>
<th align="center" valign="top">F1</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">NB</td>
<td align="center" valign="top">0.3987</td>
<td align="center" valign="top">0.4165</td>
<td align="center" valign="top">0.4052</td>
<td align="center" valign="top">0.4108</td>
<td align="center" valign="top">0.3949</td>
<td align="center" valign="top">0.4168</td>
<td align="center" valign="top">0.3977</td>
<td align="center" valign="top">0.4080</td>
</tr>
<tr>
<td align="left" valign="top">KNN</td>
<td align="center" valign="top">0.4543</td>
<td align="center" valign="top">0.9126</td>
<td align="center" valign="top">0.8730</td>
<td align="center" valign="top">0.8924</td>
<td align="center" valign="top">0.4568</td>
<td align="center" valign="top">0.9168</td>
<td align="center" valign="top">0.8720</td>
<td align="center" valign="top">0.8826</td>
</tr>
<tr>
<td align="left" valign="top">SVM</td>
<td align="center" valign="top">0.4697</td>
<td align="center" valign="top">0.5176</td>
<td align="center" valign="top">0.4647</td>
<td align="center" valign="top">0.4897</td>
<td align="center" valign="top">0.4476</td>
<td align="center" valign="top">05008</td>
<td align="center" valign="top">0.4570</td>
<td align="center" valign="top">0.4769</td>
</tr>
<tr>
<td align="left" valign="top">RF</td>
<td align="center" valign="top">0.5526</td>
<td align="center" valign="top">0.5871</td>
<td align="center" valign="top">0.5731</td>
<td align="center" valign="top">0.5800</td>
<td align="center" valign="top">0.5476</td>
<td align="center" valign="top">05866</td>
<td align="center" valign="top">0.5687</td>
<td align="center" valign="top">0.5765</td>
</tr>
<tr>
<td align="left" valign="top">XGBoost</td>
<td align="center" valign="top">0.5149</td>
<td align="center" valign="top">0.5357</td>
<td align="center" valign="top">0.5047</td>
<td align="center" valign="top">0.5197</td>
<td align="center" valign="top">0.5079</td>
<td align="center" valign="top">0.5122</td>
<td align="center" valign="top">0.4976</td>
<td align="center" valign="top">0.5121</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><xref ref-type="fig" rid="fig6">Figure 6</xref> presents the visual results of text and image features. The LDA projection of text features (<xref ref-type="fig" rid="fig6">Figure 6A</xref>) reveals substantial overlap between PED and TGE, with blurred boundaries among other categories, which explains the residual misclassifications observed in <xref ref-type="table" rid="tab5">Table 5</xref> despite the strong performance of text-only models. In contrast, the Sankey diagram of small-intestine lesion features (<xref ref-type="fig" rid="fig6">Figure 6B</xref>) illustrates a many-to-many correspondence between lesion traits and PGID classes, thereby clarifying why the image-only models reported in <xref ref-type="table" rid="tab6">Table 6</xref> performed poorly.</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>Visualization of single-modal classification effects. <bold>(A)</bold> Visualization of text classification effects. <bold>(B)</bold> Visualization of the mapping between small intestine lesion characteristics and porcine gastrointestinal infectious diseases.</p>
</caption>
<graphic xlink:href="fvets-12-1660745-g006.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Scatter plot and Sankey diagram labeled A and B. A: Scatter plot with dots in different colors representing groups PED, TGE, PPE, YSNP, WSP, and CEP, distributed across two dimensions. B: Sankey diagram showing mapping relationships between groups SISMC, SIWB, TSIW, HTSIW, SIWH, SIMHp, HHSIM, SIMHh and groups PED, YSNP, WSP, CEP, TGE, and PPE.</alt-text>
</graphic>
</fig>
<p>To further improve the recognition performance, this study performed feature-level fusion, combining the text features from MS-TextCNN with the image features from Mask R-CNN&#x202F;+&#x202F;CNN under both 8-class and 6-class encodings. The results are shown in <xref ref-type="table" rid="tab7">Table 7</xref>. Among them, RF achieved the best performance, with accuracies of 87.58% (text + 8-class image features) and 86.47% (text + 6-class image features), outperforming all single-modality baselines. KNN, SVM, and XGBoost also demonstrated high overall accuracies (all &#x003E;83%), validating the robustness of multimodal fusion, while NB remained the weakest (66.03 and 62.78%). Additionally, most models performed slightly better with the text + 8-class scheme, suggesting that finer-grained image encoding provides richer complementary cues to bagging classification.</p>
<table-wrap position="float" id="tab7">
<label>Table 7</label>
<caption>
<p>Classification results of combined features by different machine learning models.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th rowspan="2">Models</th>
<th align="center" valign="top" colspan="4">Text features + 8-class image features</th>
<th align="center" valign="top" colspan="4">Text features + 6-class image features</th>
</tr>
<tr>
<th align="center" valign="top">Acc</th>
<th align="center" valign="top">
<italic>P</italic>
</th>
<th align="center" valign="top">
<italic>R</italic>
</th>
<th align="center" valign="top">F1</th>
<th align="center" valign="top">Acc</th>
<th align="center" valign="top">
<italic>P</italic>
</th>
<th align="center" valign="top">
<italic>R</italic>
</th>
<th align="center" valign="top">F1</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">NB</td>
<td align="center" valign="top">0.6603</td>
<td align="center" valign="top">0.8305</td>
<td align="center" valign="top">0.8152</td>
<td align="center" valign="top">0.8228</td>
<td align="center" valign="top">0.6278</td>
<td align="center" valign="top">0.7623</td>
<td align="center" valign="top">0.7749</td>
<td align="center" valign="top">0.7685</td>
</tr>
<tr>
<td align="left" valign="top">KNN</td>
<td align="center" valign="top">0.8569</td>
<td align="center" valign="top">0.9126</td>
<td align="center" valign="top">0.8730</td>
<td align="center" valign="top">0.8924</td>
<td align="center" valign="top">0.8569</td>
<td align="center" valign="top">0.9168</td>
<td align="center" valign="top">0.8720</td>
<td align="center" valign="top">0.8938</td>
</tr>
<tr>
<td align="left" valign="top">SVM</td>
<td align="center" valign="top">0.8443</td>
<td align="center" valign="top">0.9158</td>
<td align="center" valign="top">0.8564</td>
<td align="center" valign="top">0.8851</td>
<td align="center" valign="top">0.8455</td>
<td align="center" valign="top">0.9198</td>
<td align="center" valign="top">0.8564</td>
<td align="center" valign="top">0.8870</td>
</tr>
<tr>
<td align="left" valign="top">RF</td>
<td align="center" valign="top">0.8758</td>
<td align="center" valign="top">0.9301</td>
<td align="center" valign="top">0.8754</td>
<td align="center" valign="top">0.9019</td>
<td align="center" valign="top">0.8647</td>
<td align="center" valign="top">0.9249</td>
<td align="center" valign="top">0.8677</td>
<td align="center" valign="top">0.8954</td>
</tr>
<tr>
<td align="left" valign="top">XGBoost</td>
<td align="center" valign="top">0.8381</td>
<td align="center" valign="top">0.9093</td>
<td align="center" valign="top">0.8474</td>
<td align="center" valign="top">0.8773</td>
<td align="center" valign="top">0.8496</td>
<td align="center" valign="top">0.8898</td>
<td align="center" valign="top">0.8376</td>
<td align="center" valign="top">0.8629</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To more intuitively analyze and compare the recognition performance for each disease, confusion matrices for the classification results of each model under the 8-class image classification scenario were plotted, as shown in <xref ref-type="fig" rid="fig7">Figure 7</xref>. The confusion matrices indicate that RF model performed best, with only 21 PED cases misclassified as TGE, achieving a recognition accuracy of 87.58%. In contrast, NB model exhibited the poorest performance, with significant errors in distinguishing PED and TGE, resulting in lower model accuracy. KNN, SVM, and XGBoost models showed improved performance over NB model but did not match the superior accuracy of the RF model. Furthermore, the confusion matrix revealed a severe class imbalance issue, where the rare class CEP showed the lowest recognition performance across most models. As shown in <xref ref-type="fig" rid="fig7">Figure 7</xref>, the RF model achieved a recall rate of only 62.5% for CEP. In contrast, the NB model yielded the highest recall for CEP at 87.5%. We further explored common imbalance-handling strategies such as SMOTE (<xref ref-type="bibr" rid="ref35">35</xref>), ADASYN (<xref ref-type="bibr" rid="ref36">36</xref>), and AdaBoost (<xref ref-type="bibr" rid="ref37">37</xref>), but none of them yielded notable improvements for the minority class CEP.</p>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>Confusion matrices of different machine learning models. <bold>(A)</bold> NB. <bold>(B)</bold> KNN. <bold>(C)</bold> SVM. <bold>(D)</bold> RF. <bold>(E)</bold> XGBoost.</p>
</caption>
<graphic xlink:href="fvets-12-1660745-g007.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Five confusion matrices labeled A to E compare the performance of classifiers NB, KNN, SVM, RF, and XGBoost. Each matrix displays true labels versus predicted labels for six classes: PED, TGE, PPE, YSNP, WSP, and CEP. Darker colors indicate higher values.</alt-text>
</graphic>
</fig>
<p>To further assess the contribution of each feature to the RF model, <xref ref-type="fig" rid="fig8">Figure 8</xref> presents the importance scores of the text features and 8-class image features used in model construction. In the figure, blue bars denote text features and red bars denote image features. The results indicate that text features contribute more substantially to the model than image features. Further examination of the textual descriptions of PED and TGE revealed that some samples share highly similar keywords, which aligns with the partial overlap observed in <xref ref-type="fig" rid="fig6">Figure 6A</xref>. Moreover, as shown in <xref ref-type="fig" rid="fig6">Figure 6B</xref>, both diseases exhibit TSIW-type and HTSIW-type lesions in the anatomical images of the small intestine. These similarities in textual descriptions and lesion types collectively contribute to the difficulty in distinguishing certain PED and TGE cases.</p>
<fig position="float" id="fig8">
<label>Figure 8</label>
<caption>
<p>Importance scores for all features used in the RF model.</p>
</caption>
<graphic xlink:href="fvets-12-1660745-g008.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Bar chart showing the importance of different features. Text features 1&#x2013;10 are shown in blue, and image features 1&#x2013;4 are shown in red. Most text features are more important than the image features.</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec16">
<label>3.4</label>
<title>Comparison with YOLO</title>
<p>YOLO uses attention mechanisms and dynamic convolution to especially improve small object detection, achieving better detection accuracy and computational efficiency (<xref ref-type="bibr" rid="ref38">38</xref>). To further validate the advantage of the proposed Mask R-CNN in improving the recognition accuracy of CNN networks, YOLO was used to detect swine anatomical images, with results shown in <xref ref-type="table" rid="tab8">Table 8</xref>. It can be observed that under three different classification scenarios, the recognition performance of YOLOv8 is similar to that of the proposed Mask R-CNN model but slightly inferior. YOLOv8 (<xref ref-type="bibr" rid="ref39">39</xref>) demonstrates slightly better recognition performance than YOLOv12 (<xref ref-type="bibr" rid="ref40">40</xref>) and YOLOv13 (<xref ref-type="bibr" rid="ref41">41</xref>), making it the most effective YOLO model under the data conditions of this study.</p>
<table-wrap position="float" id="tab8">
<label>Table 8</label>
<caption>
<p>Detection results of the YOLO under different classification methods.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th>Models</th>
<th align="center" valign="top">Category number</th>
<th align="center" valign="top">SISMC /NDSIWL</th>
<th align="center" valign="top">SIMHh</th>
<th align="center" valign="top">SIMHp</th>
<th align="center" valign="top">TSIW</th>
<th align="center" valign="top">SIWH</th>
<th align="center" valign="top">SIWB</th>
<th align="center" valign="top">HTSIW</th>
<th align="center" valign="top">HHSIM</th>
<th align="center" valign="top">OA</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" rowspan="3">YOLOv8</td>
<td align="center" valign="middle">8</td>
<td align="center" valign="middle">0.2728</td>
<td align="center" valign="middle">0.1851</td>
<td align="center" valign="middle">0.2213</td>
<td align="center" valign="middle">0.4872</td>
<td align="center" valign="middle">0.2965</td>
<td align="center" valign="middle">0.3932</td>
<td align="center" valign="middle">0.5651</td>
<td align="center" valign="middle">0.1619</td>
<td align="center" valign="middle">0.3478</td>
</tr>
<tr>
<td align="center" valign="middle">6</td>
<td align="center" valign="middle">0.7384</td>
<td align="center" valign="middle">0.1794</td>
<td align="center" valign="middle">0.4205</td>
<td align="center" valign="middle">0.7151</td>
<td align="center" valign="middle">0.6944</td>
<td align="center" valign="middle">0.8207</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">0.6542</td>
</tr>
<tr>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0.9777</td>
<td align="center" valign="middle">0.894</td>
<td align="center" valign="middle">0.8213</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">0.9354</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="3">YOLOv12</td>
<td align="center" valign="middle">8</td>
<td align="center" valign="middle">0.2688</td>
<td align="center" valign="middle">0.1853</td>
<td align="center" valign="middle">0.2253</td>
<td align="center" valign="middle">0.4700</td>
<td align="center" valign="middle">0.2896</td>
<td align="center" valign="middle">0.3795</td>
<td align="center" valign="middle">0.5538</td>
<td align="center" valign="middle">0.1587</td>
<td align="center" valign="middle">0.3341</td>
</tr>
<tr>
<td align="center" valign="middle">6</td>
<td align="center" valign="middle">0.7136</td>
<td align="center" valign="middle">0.1853</td>
<td align="center" valign="middle">0.4110</td>
<td align="center" valign="middle">0.7133</td>
<td align="center" valign="middle">0.6355</td>
<td align="center" valign="middle">0.8113</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">0.6340</td>
</tr>
<tr>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0.9654</td>
<td align="center" valign="middle">0.8600</td>
<td align="center" valign="middle">0.7986</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">0.9140</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="3">YOLOv13</td>
<td align="center" valign="middle">8</td>
<td align="center" valign="middle">0.2758</td>
<td align="center" valign="middle">0.1875</td>
<td align="center" valign="middle">0.2253</td>
<td align="center" valign="middle">0.4864</td>
<td align="center" valign="middle">0.2997</td>
<td align="center" valign="middle">0.3988</td>
<td align="center" valign="middle">0.5567</td>
<td align="center" valign="middle">0.1651</td>
<td align="center" valign="middle">0.3431</td>
</tr>
<tr>
<td align="center" valign="middle">6</td>
<td align="center" valign="middle">0.7349</td>
<td align="center" valign="middle">0.1875</td>
<td align="center" valign="middle">0.4110</td>
<td align="center" valign="middle">0.7213</td>
<td align="center" valign="middle">0.6300</td>
<td align="center" valign="middle">0.8170</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">0.6480</td>
</tr>
<tr>
<td align="center" valign="middle">3</td>
<td align="center" valign="middle">0.9724</td>
<td align="center" valign="middle">0.8876</td>
<td align="center" valign="middle">0.8156</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">\</td>
<td align="center" valign="middle">0.9314</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The segmented images from YOLO v8 were input into the CNN networks for classification, with experimental results shown in <xref ref-type="table" rid="tab9">Table 9</xref>. Compared to the results in <xref ref-type="table" rid="tab4">Table 4</xref>, the classification accuracy of all CNN models in <xref ref-type="table" rid="tab9">Table 9</xref> is significantly lower. Among them, DenseNet achieved the highest classification performance, with accuracies of 94.34 and 94.81% for the 8-class and 6-class scenarios, respectively. These results indicate that the classification performance of YOLOv8-segmented images is generally lower than that of Mask R-CNN, demonstrating the superiority of Mask R-CNN&#x2019;s pixel-level image segmentation technology in classification and recognition tasks.</p>
<table-wrap position="float" id="tab9">
<label>Table 9</label>
<caption>
<p>Classification results of YoloV8-segmented images by different CNN models.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top" rowspan="2">Models</th>
<th align="center" valign="top" colspan="4">8 Categories</th>
<th align="center" valign="top" colspan="4">6 Categories</th>
</tr>
<tr>
<th align="center" valign="top">Acc</th>
<th align="center" valign="top">
<italic>P</italic>
</th>
<th align="center" valign="top">
<italic>R</italic>
</th>
<th align="center" valign="top">F1</th>
<th align="center" valign="top">Acc</th>
<th align="center" valign="top">
<italic>P</italic>
</th>
<th align="center" valign="top">
<italic>R</italic>
</th>
<th align="center" valign="top">F1</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">AlexNet</td>
<td align="center" valign="top">0.7453</td>
<td align="center" valign="top">0.7475</td>
<td align="center" valign="top">0.7362</td>
<td align="center" valign="top">0.7325</td>
<td align="center" valign="top">0.8443</td>
<td align="center" valign="top">0.8627</td>
<td align="center" valign="top">0.7783</td>
<td align="center" valign="top">0.8050</td>
</tr>
<tr>
<td align="left" valign="middle">DenseNet</td>
<td align="center" valign="top">0.9434</td>
<td align="center" valign="top">0.9350</td>
<td align="center" valign="top">0.9625</td>
<td align="center" valign="top">0.9438</td>
<td align="center" valign="top">0.9481</td>
<td align="center" valign="top">0.9517</td>
<td align="center" valign="top">0.9450</td>
<td align="center" valign="top">0.9450</td>
</tr>
<tr>
<td align="left" valign="middle">GoogleNet</td>
<td align="center" valign="top">0.9151</td>
<td align="center" valign="top">0.9225</td>
<td align="center" valign="top">0.9150</td>
<td align="center" valign="top">0.9113</td>
<td align="center" valign="top">0.9434</td>
<td align="center" valign="top">0.9533</td>
<td align="center" valign="top">0.9333</td>
<td align="center" valign="top">0.9417</td>
</tr>
<tr>
<td align="left" valign="middle">ResNet</td>
<td align="center" valign="top">0.7358</td>
<td align="center" valign="top">0.7100</td>
<td align="center" valign="top">0.7175</td>
<td align="center" valign="top">0.7075</td>
<td align="center" valign="top">0.8302</td>
<td align="center" valign="top">0.8550</td>
<td align="center" valign="top">0.8033</td>
<td align="center" valign="top">0.8217</td>
</tr>
<tr>
<td align="left" valign="middle">VGGNet</td>
<td align="center" valign="top">0.7972</td>
<td align="center" valign="top">0.8113</td>
<td align="center" valign="top">0.7475</td>
<td align="center" valign="top">0.7688</td>
<td align="center" valign="top">0.8066</td>
<td align="center" valign="top">0.7783</td>
<td align="center" valign="top">0.7733</td>
<td align="center" valign="top">0.7717</td>
</tr>
<tr>
<td align="left" valign="middle">EfficientNet</td>
<td align="center" valign="top">0.8661</td>
<td align="center" valign="top">0.8355</td>
<td align="center" valign="top">0.8461</td>
<td align="center" valign="top">0.8408</td>
<td align="center" valign="top">0.9023</td>
<td align="center" valign="top">0.8967</td>
<td align="center" valign="top">0.8764</td>
<td align="center" valign="top">0.8864</td>
</tr>
<tr>
<td align="left" valign="middle">Vision transformer</td>
<td align="center" valign="top">0.9017</td>
<td align="center" valign="top">0.8981</td>
<td align="center" valign="top">0.8871</td>
<td align="center" valign="top">0.8926</td>
<td align="center" valign="top">0.9324</td>
<td align="center" valign="top">0.9311</td>
<td align="center" valign="top">0.9246</td>
<td align="center" valign="top">0.9292</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="sec17">
<label>4</label>
<title>Discussion</title>
<sec id="sec18">
<label>4.1</label>
<title>Advantages of multimodal information fusion</title>
<p>In the diagnosis of PGID, traditional single-modal diagnostic methods, whether relying on empirical observation of clinical symptoms or single laboratory testing techniques, have significant limitations (<xref ref-type="fig" rid="fig6">Figure 6</xref>). In contrast, this study integrates anatomical images of swine small intestines with disease case information to construct a multimodal diagnostic model, effectively addressing these shortcomings.</p>
<p>Multimodal information fusion integrates information from multiple data sources to achieve complementarity and enhancement, thereby obtaining richer, more comprehensive, and more accurate information, which in turn improves detection performance (<xref ref-type="bibr" rid="ref42">42</xref>, <xref ref-type="bibr" rid="ref43">43</xref>). In this study, image data intuitively present visual features such as the morphology and location of small intestine lesions, while case information includes contextual and symptomatic details such as age at onset, season, and disease progression. These two modalities complement each other, providing a more comprehensive and multidimensional basis for disease diagnosis. At the model level, multimodal feature fusion enables the model to learn associations and complementary information between different modalities, enhancing its ability to represent complex disease characteristics. Experimental results demonstrate that the multimodal recognition model significantly outperforms single-modal models in disease classification accuracy. The RF algorithm, after fusing multimodal features, achieved a diagnostic accuracy of 87.58%, fully highlighting the substantial potential of multimodal information fusion in improving diagnostic accuracy and reliability.</p>
</sec>
<sec id="sec19">
<label>4.2</label>
<title>High-quality text generation by LLM</title>
<p>The collection of porcine digestive tract infectious disease cases is challenging, and the limited sample size results in insufficient training data, which is a key factor constraining the performance of diagnostic models. ChatGPT, as an advanced LLM, leverages its powerful semantic understanding and text generation capabilities to produce a large number of semantically consistent but diversely expressed text samples through simple prompt inputs (<xref ref-type="bibr" rid="ref18">18</xref>). This approach expanded the original text dataset from 106 to 636 samples, significantly enriching the training data.</p>
<p>These high-quality generated texts not only increase data diversity but also enable the model to learn a broader range of linguistic expressions, improving its ability to recognize and understand case information described in different styles. During actual training, models augmented with LLM-generated data exhibited better generalization when processing new, unseen text descriptions, effectively mitigating overfitting issues caused by data scarcity. Additionally, the process of text generation using LLMs requires no complex model fine-tuning, offering an efficient and convenient solution for addressing the issue of insufficient text data in swine disease diagnosis.</p>
</sec>
<sec id="sec20">
<label>4.3</label>
<title>Advantages of combining Mask R-CNN with CNN models</title>
<p>In swine anatomical images, small intestine lesion regions often occupy a small portion and are surrounded by complex backgrounds. Directly applying CNN for whole-image classification is prone to interference from irrelevant information, leading to reduced recognition accuracy. Mask R-CNN is a two-stage framework-based model capable of simultaneously performing object detection and instance segmentation (<xref ref-type="bibr" rid="ref44">44</xref>). Compared to single-stage detection algorithms like YOLO, Mask R-CNN demonstrates superior performance in small object detection and high-precision tasks. Previously, Li et al. employed the Mask R-CNN model to segment disease spots and insect spots on tea leaves, followed by classification using F-RNet, achieving precise segmentation and identification of the diseases and insect spots in tea leaves (<xref ref-type="bibr" rid="ref45">45</xref>). In this study, the improved Mask R-CNN, optimized by incorporating HRNet and CBAM, can accurately locate and segment lesion regions, effectively removing background noise and preserving detailed lesion information, thus providing high-quality image data for subsequent classification.</p>
<p>Building on this, CNN classification models leverage their robust feature extraction and classification capabilities to perform in-depth analysis of segmented lesion regions. Different CNN models utilize their unique network architectures to extract lesion features from various perspectives, capturing rich semantic information from low to high levels and enabling fine-grained classification of complex lesion characteristics. Experimental results show that, with preprocessing by the improved Mask R-CNN, CNN models significantly improved classification accuracy for lesion characteristics, achieving efficient and accurate identification and classification of small intestine lesions in swine.</p>
</sec>
<sec id="sec21">
<label>4.4</label>
<title>Shortcomings and future work</title>
<p>In livestock farming, early disease diagnosis faces significant challenges due to the heavy reliance on subjective and inefficient manual inspections, while laboratory methods such as PCR, though highly accurate, are time-consuming, costly, require complex procedures, and depend on specialized equipment and trained personnel. These issues are particularly pronounced in large-scale farms, often leading to the spread of epidemics and increased economic losses. Therefore, the use of AI to achieve swine disease detection can effectively break through the limitations of traditional approaches, and while ensuring accuracy, improve detection efficiency and reduce costs.</p>
<p>Although the multimodal diagnostic framework proposed in this study demonstrates good performance in identifying PGID, it still has some limitations. First, the current dataset is relatively small and suffers from severe class imbalance, which may limit its generalization ability in diagnosing swine diseases. To address this issue, a novel data augmentation technique was proposed, which improved model accuracy but had limited impact on class imbalance. Furthermore, due to the extremely small number of samples in certain categories (e.g., CEP), synthetic data-based augmentation and boosting algorithms also proved insufficient in resolving the class imbalance problem in this dataset. Second, the text modality enhancement relies on semantic descriptions generated by ChatGPT, which, while helpful in enhancing sample diversity, may also introduce semantic biases that could affect the stability of the text-image fusion model. These biases may manifest as inconsistencies between the generated textual features and the actual clinical presentation of the diseases, potentially leading to misalignment during multimodal feature fusion and reducing overall diagnostic accuracy.</p>
<p>In future work, we plan to collect a larger and more balanced dataset encompassing diverse regions and pig breeds to enhance the generalization ability of the model. To further improve diagnostic performance, more advanced deep learning architectures will be explored and compared. In particular, generative models will be considered to produce realistic synthetic samples for minority classes, thereby alleviating class imbalance and improving the robustness of the diagnostic framework. Furthermore, optimizing text enhancement is expected to improve semantic quality and credibility, with focusing on expert-annotated descriptions and domain-specific language model tuning to reduce bias and enhance multimodal robustness.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="sec22">
<label>5</label>
<title>Conclusion</title>
<p>This study presented an intelligent diagnostic method, multimodal AI and LLMs-based classification framework, for identifying 6 types of PGID. The framework integrates a MS-TextCNN model, an improved Mask R-CNN model, CNN image classification models, and machine learning algorithms, exhibiting improved classification performance and robustness. Our results indicate that multimodal diagnostic model can significantly enhance the accuracy and efficiency in complex disease diagnosis. This study provides an efficient and accurate intelligent solution for diagnosing PGID, offering valuable reference for disease prevention and control in the livestock industry.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec23">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="ethics-statement" id="sec24">
<title>Ethics statement</title>
<p>The animal study was approved by Heze University (China) Animal Welfare and Ethical Review Board. The study was conducted in accordance with the local legislation and institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="sec25">
<title>Author contributions</title>
<p>HW: Writing &#x2013; review &#x0026; editing, Data curation, Formal analysis, Methodology, Investigation, Conceptualization, Writing &#x2013; original draft. HS: Formal analysis, Methodology, Writing &#x2013; review &#x0026; editing, Writing &#x2013; original draft, Data curation, Conceptualization. JY: Formal analysis, Data curation, Writing &#x2013; original draft. ZF: Investigation, Resources, Writing &#x2013; original draft. HD: Writing &#x2013; original draft, Investigation, Data curation. LJ: Data curation, Investigation, Writing &#x2013; original draft. QS: Conceptualization, Supervision, Project administration, Writing &#x2013; review &#x0026; editing.</p>
</sec>
<sec sec-type="funding-information" id="sec26">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was jointly funded by the Key Research and Development Program of Shandong Province (Agricultural Good Variety Project, 2022LZGC003) and the new generation of artificial intelligence National Science and Technology Major Project (2021ZD0113802).</p>
</sec>
<ack>
<p>Heartfelt thanks to all the farm owners, administrators, and frontline veterinarians who have generously contributed to the vital task of sample collection.</p>
</ack>
<sec sec-type="COI-statement" id="sec27">
<title>Conflict of interest</title>
<p>HD was employed by Rizhao Jiacheng Animal Health Products Co., Ltd.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec28">
<title>Generative AI statement</title>
<p>The authors declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="sec29">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><label>1.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Fenghua</surname><given-names>W.</given-names></name><collab id="coll1">National Bureau of Statistics</collab></person-group>. (<year>2025</year>). <source>The agricultural economic situation was stable and showed a positive trend in 2024</source>. Available onine at: <ext-link xlink:href="https://www.stats.gov.cn/sj/sjjd/202501/t20250117_1958344.html" ext-link-type="uri">https://www.stats.gov.cn/sj/sjjd/202501/t20250117_1958344.html</ext-link>. (Accessed March 7, 2025).</citation></ref>
<ref id="ref2"><label>2.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>Q</given-names></name> <name><surname>Wang</surname><given-names>H</given-names></name></person-group>. <article-title>Porcine enteric coronaviruses: an updated overview of the pathogenesis, prevalence, and diagnosis</article-title>. <source>Vet Res Commun</source>. (<year>2021</year>) <volume>45</volume>:<fpage>75</fpage>&#x2013;<lpage>86</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11259-021-09808-0</pub-id>, PMID: <pub-id pub-id-type="pmid">34251560</pub-id></citation></ref>
<ref id="ref3"><label>3.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jang</surname><given-names>G</given-names></name> <name><surname>Lee</surname><given-names>D</given-names></name> <name><surname>Shin</surname><given-names>S</given-names></name> <name><surname>Lim</surname><given-names>J</given-names></name> <name><surname>Won</surname><given-names>H</given-names></name> <name><surname>Eo</surname><given-names>Y</given-names></name> <etal/></person-group>. <article-title>Porcine epidemic diarrhea virus: an update overview of virus epidemiology, vaccines, and control strategies in South Korea</article-title>. <source>J Vet Sci</source>. (<year>2023</year>) <volume>24</volume>:<fpage>e58</fpage>. doi: <pub-id pub-id-type="doi">10.4142/jvs.23090</pub-id>, PMID: <pub-id pub-id-type="pmid">37532301</pub-id></citation></ref>
<ref id="ref4"><label>4.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Uzal</surname><given-names>FA</given-names></name> <name><surname>Navarro</surname><given-names>MA</given-names></name> <name><surname>Asin</surname><given-names>J</given-names></name> <name><surname>Boix</surname><given-names>O</given-names></name> <name><surname>Ballar&#x00E0;-Rodriguez</surname><given-names>I</given-names></name> <name><surname>Gibert</surname><given-names>X</given-names></name></person-group>. <article-title>Clostridial diarrheas in piglets: a review</article-title>. <source>Vet Microbiol</source>. (<year>2023</year>) <volume>280</volume>:<fpage>109691</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.vetmic.2023.109691</pub-id>, PMID: <pub-id pub-id-type="pmid">36870204</pub-id></citation></ref>
<ref id="ref5"><label>5.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>J</given-names></name> <name><surname>Liu</surname><given-names>R</given-names></name> <name><surname>Liu</surname><given-names>H</given-names></name> <name><surname>Chen</surname><given-names>J</given-names></name> <name><surname>Li</surname><given-names>X</given-names></name> <name><surname>Zhang</surname><given-names>J</given-names></name> <etal/></person-group>. <article-title>Development of a multiplex quantitative PCR for detecting porcine epidemic diarrhea virus, transmissible gastroenteritis virus, and porcine Deltacoronavirus simultaneously in China</article-title>. <source>Vet Sci</source>. (<year>2023</year>) <volume>10</volume>:<fpage>402</fpage>. doi: <pub-id pub-id-type="doi">10.3390/vetsci10060402</pub-id>, PMID: <pub-id pub-id-type="pmid">37368788</pub-id></citation></ref>
<ref id="ref6"><label>6.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname><given-names>X</given-names></name> <name><surname>Li</surname><given-names>L</given-names></name> <name><surname>Su</surname><given-names>X</given-names></name> <name><surname>Li</surname><given-names>J</given-names></name> <name><surname>Liao</surname><given-names>J</given-names></name> <name><surname>Yang</surname><given-names>J</given-names></name> <etal/></person-group>. <article-title>Development of an indirect enzyme-linked immunosorbent assay based on the yeast-expressed CO-26K-equivalent epitope-containing antigen for detection of serum antibodies against porcine epidemic diarrhea virus</article-title>. <source>Viruses</source>. (<year>2023</year>) <volume>15</volume>:<fpage>882</fpage>. doi: <pub-id pub-id-type="doi">10.3390/v15040882</pub-id>, PMID: <pub-id pub-id-type="pmid">37112862</pub-id></citation></ref>
<ref id="ref7"><label>7.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pan</surname><given-names>X</given-names></name> <name><surname>Zhu</surname><given-names>J</given-names></name> <name><surname>Tai</surname><given-names>W</given-names></name> <name><surname>Fu</surname><given-names>Y</given-names></name></person-group>. <article-title>An automated method to quantify the composition of live pigs based on computed tomography segmentation using deep neural networks</article-title>. <source>Comput Electron Agric</source>. (<year>2021</year>) <volume>183</volume>:<fpage>105987</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2021.105987</pub-id>, PMID: <pub-id pub-id-type="pmid">40842713</pub-id></citation></ref>
<ref id="ref8"><label>8.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname><given-names>D</given-names></name> <name><surname>Wang</surname><given-names>Y</given-names></name> <name><surname>Han</surname><given-names>M</given-names></name> <name><surname>Song</surname><given-names>L</given-names></name> <name><surname>Shang</surname><given-names>Y</given-names></name> <name><surname>Zhang</surname><given-names>X</given-names></name> <etal/></person-group>. <article-title>Using a CNN-LSTM for basic behaviors detection of a single dairy cow in a complex environment</article-title>. <source>Comput Electron Agric</source>. (<year>2021</year>) <volume>182</volume>:<fpage>106016</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2021.106016</pub-id>, PMID: <pub-id pub-id-type="pmid">40842713</pub-id></citation></ref>
<ref id="ref9"><label>9.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chae</surname><given-names>JW</given-names></name> <name><surname>Choi</surname><given-names>YH</given-names></name> <name><surname>Lee</surname><given-names>JN</given-names></name> <name><surname>Park</surname><given-names>HJ</given-names></name> <name><surname>Jeong</surname><given-names>YD</given-names></name> <name><surname>Cho</surname><given-names>ES</given-names></name> <etal/></person-group>. <article-title>An intelligent method for pregnancy diagnosis in breeding sows according to ultrasonography algorithms</article-title>. <source>J Anim Sci Technol</source>. (<year>2023</year>) <volume>65</volume>:<fpage>365</fpage>&#x2013;<lpage>76</lpage>. doi: <pub-id pub-id-type="doi">10.5187/jast.2022.e107</pub-id>, PMID: <pub-id pub-id-type="pmid">37093914</pub-id></citation></ref>
<ref id="ref10"><label>10.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kittichai</surname><given-names>V</given-names></name> <name><surname>Kaewthamasorn</surname><given-names>M</given-names></name> <name><surname>Arnuphaprasert</surname><given-names>A</given-names></name> <name><surname>Jomtarak</surname><given-names>R</given-names></name> <name><surname>Naing</surname><given-names>KM</given-names></name> <name><surname>Tongloy</surname><given-names>T</given-names></name> <etal/></person-group>. <article-title>A deep contrastive learning-based image retrieval system for automatic detection of infectious cattle diseases</article-title>. <source>J Big Data</source>. (<year>2025</year>) <volume>12</volume>:<fpage>2</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s40537-024-01057-7</pub-id></citation></ref>
<ref id="ref11"><label>11.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Muhammad Saqib</surname><given-names>S</given-names></name> <name><surname>Iqbal</surname><given-names>M</given-names></name> <name><surname>Othman</surname><given-names>MTB</given-names></name> <name><surname>Shahazad</surname><given-names>T</given-names></name> <name><surname>Ghadi</surname><given-names>YY</given-names></name> <name><surname>Al-Amro</surname><given-names>S</given-names></name> <etal/></person-group>. <article-title>Lumpy skin disease diagnosis in cattle: a deep learning approach optimized with RMSProp and MobileNetV2</article-title>. <source>PLoS One</source>. (<year>2024</year>) <volume>19</volume>:<fpage>e0302862</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0302862</pub-id>, PMID: <pub-id pub-id-type="pmid">39102387</pub-id></citation></ref>
<ref id="ref12"><label>12.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname><given-names>H</given-names></name> <name><surname>Lee</surname><given-names>I-G</given-names></name> <name><surname>Oh</surname><given-names>J-Y</given-names></name> <name><surname>Kim</surname><given-names>J</given-names></name> <name><surname>Jeong</surname><given-names>J-H</given-names></name> <name><surname>Eom</surname><given-names>K</given-names></name></person-group>. <article-title>Deep learning-based ultrasonographic classification of canine chronic kidney disease</article-title>. <source>Front Vet Sci</source>. (<year>2024</year>) <volume>11</volume>:<fpage>1443234</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fvets.2024.1443234</pub-id>, PMID: <pub-id pub-id-type="pmid">39296582</pub-id></citation></ref>
<ref id="ref13"><label>13.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Buric</surname><given-names>M</given-names></name> <name><surname>Grozdanic</surname><given-names>S</given-names></name> <name><surname>Ivasic-Kos</surname><given-names>M</given-names></name></person-group>. <article-title>Diagnosis of ophthalmologic diseases in canines based on images using neural networks for image segmentation</article-title>. <source>Heliyon</source>. (<year>2024</year>) <volume>10</volume>:<fpage>e38287</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.heliyon.2024.e38287</pub-id>, PMID: <pub-id pub-id-type="pmid">39397908</pub-id></citation></ref>
<ref id="ref14"><label>14.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Keicher</surname><given-names>M</given-names></name> <name><surname>Burwinkel</surname><given-names>H</given-names></name> <name><surname>Bani-Harouni</surname><given-names>D</given-names></name> <name><surname>Paschali</surname><given-names>M</given-names></name> <name><surname>Czempiel</surname><given-names>T</given-names></name> <name><surname>Burian</surname><given-names>E</given-names></name> <etal/></person-group>. <article-title>Multimodal graph attention network for COVID-19 outcome prediction</article-title>. <source>Sci Rep</source>. (<year>2023</year>) <volume>13</volume>:<fpage>19539</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-023-46625-8</pub-id>, PMID: <pub-id pub-id-type="pmid">37945590</pub-id></citation></ref>
<ref id="ref15"><label>15.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname><given-names>H</given-names></name> <name><surname>Yu</surname><given-names>Y</given-names></name> <name><surname>Wang</surname><given-names>C</given-names></name> <name><surname>Zhang</surname><given-names>S</given-names></name> <name><surname>Gao</surname><given-names>Y</given-names></name> <name><surname>Pan</surname><given-names>J</given-names></name> <etal/></person-group>. <article-title>A transformer-based representation-learning model with unified processing of multimodal input for clinical diagnostics</article-title>. <source>Nat Biomed Eng</source>. (<year>2023</year>) <volume>7</volume>:<fpage>743</fpage>&#x2013;<lpage>55</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41551-023-01045-x</pub-id>, PMID: <pub-id pub-id-type="pmid">37308585</pub-id></citation></ref>
<ref id="ref16"><label>16.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nguyen</surname><given-names>HH</given-names></name> <name><surname>Blaschko</surname><given-names>MB</given-names></name> <name><surname>Saarakkala</surname><given-names>S</given-names></name> <name><surname>Tiulpin</surname><given-names>A</given-names></name></person-group>. <article-title>Clinically-inspired multi-agent transformers for disease trajectory forecasting from multimodal data</article-title>. <source>IEEE Trans Med Imaging</source>. (<year>2023</year>) <volume>43</volume>:<fpage>529</fpage>&#x2013;<lpage>41</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TMI.2023.3312524</pub-id>, PMID: <pub-id pub-id-type="pmid">37672368</pub-id></citation></ref>
<ref id="ref17"><label>17.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chung</surname><given-names>J</given-names></name> <name><surname>Pierce</surname><given-names>J</given-names></name> <name><surname>Franklin</surname><given-names>C</given-names></name> <name><surname>Olson</surname><given-names>RM</given-names></name> <name><surname>Morrison</surname><given-names>AR</given-names></name> <name><surname>Amos-Landgraf</surname><given-names>J</given-names></name></person-group>. <article-title>Translating animal models of SARS-CoV-2 infection to vascular, neurological and gastrointestinal manifestations of COVID-19</article-title>. <source>Dis Model Mech</source>. (<year>2025</year>) <volume>18</volume>:<fpage>dmm052086</fpage>. doi: <pub-id pub-id-type="doi">10.1242/dmm.052086</pub-id>, PMID: <pub-id pub-id-type="pmid">40195851</pub-id></citation></ref>
<ref id="ref18"><label>18.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dai</surname><given-names>H</given-names></name> <name><surname>Liu</surname><given-names>Z</given-names></name> <name><surname>Liao</surname><given-names>W</given-names></name> <name><surname>Huang</surname><given-names>X</given-names></name> <name><surname>Cao</surname><given-names>Y</given-names></name> <name><surname>Wu</surname><given-names>Z</given-names></name> <etal/></person-group>. <article-title>AugGPT: leveraging ChatGPT for text data augmentation</article-title>. <source>arXiv</source>. (<year>2025</year>). doi: <pub-id pub-id-type="doi">10.1109/TBDATA.2025.3536934</pub-id></citation></ref>
<ref id="ref19"><label>19.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fang</surname><given-names>Y</given-names></name> <name><surname>Li</surname><given-names>X</given-names></name> <name><surname>Thomas</surname><given-names>SW</given-names></name> <name><surname>Zhu</surname><given-names>X-D</given-names></name></person-group>. <article-title>ChatGPT as data augmentation for compositional generalization: a case study in open intent detection</article-title>. <source>ArXiv</source>. (<year>2023</year>). abs/2308.13517). doi: <pub-id pub-id-type="doi">10.48550/arXiv.2308.13517</pub-id></citation></ref>
<ref id="ref20"><label>20.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han</surname><given-names>P</given-names></name> <name><surname>Kocielnik</surname><given-names>R</given-names></name> <name><surname>Saravanan</surname><given-names>A</given-names></name> <name><surname>Jiang</surname><given-names>R</given-names></name> <name><surname>Sharir</surname><given-names>O</given-names></name> <name><surname>Anandkumar</surname><given-names>A</given-names></name></person-group>. <article-title>ChatGPT based data augmentation for improved parameter-efficient Debiasing of LLMs</article-title>. <source>ArXiv</source>. (<year>2024</year>). abs/2402.11764). doi: <pub-id pub-id-type="doi">10.48550/arXiv.2402.11764</pub-id></citation></ref>
<ref id="ref21"><label>21.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shi</surname><given-names>D</given-names></name> <name><surname>Li</surname><given-names>Z</given-names></name> <name><surname>Zurada</surname><given-names>J</given-names></name> <name><surname>Manikas</surname><given-names>A</given-names></name> <name><surname>Guan</surname><given-names>J</given-names></name> <name><surname>Weichbroth</surname><given-names>P</given-names></name></person-group>. <article-title>Ontology-based text convolution neural network (TextCNN) for prediction of construction accidents</article-title>. <source>Knowl Inf Syst</source>. (<year>2024</year>) <volume>66</volume>:<fpage>2651</fpage>&#x2013;<lpage>81</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10115-023-02036-9</pub-id></citation></ref>
<ref id="ref22"><label>22.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>H</given-names></name> <name><surname>Zhang</surname><given-names>Z</given-names></name> <name><surname>Huang</surname><given-names>S</given-names></name> <name><surname>Hu</surname><given-names>J</given-names></name> <name><surname>Ni</surname><given-names>W</given-names></name> <name><surname>Liu</surname><given-names>J</given-names></name></person-group>. <article-title>TextCNN-based ensemble learning model for Japanese text multi-classification</article-title>. <source>Comput Electr Eng</source>. (<year>2023</year>) <volume>109</volume>:<fpage>108751</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compeleceng.2023.108751</pub-id></citation></ref>
<ref id="ref23"><label>23.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bhatt</surname><given-names>D</given-names></name> <name><surname>Patel</surname><given-names>C</given-names></name> <name><surname>Talsania</surname><given-names>H</given-names></name> <name><surname>Patel</surname><given-names>J</given-names></name> <name><surname>Vaghela</surname><given-names>R</given-names></name> <name><surname>Pandya</surname><given-names>S</given-names></name> <etal/></person-group>. <article-title>CNN variants for computer vision: history, architecture, application, challenges and future scope</article-title>. <source>Electronics</source>. (<year>2021</year>) <volume>10</volume>:<fpage>2470</fpage>. doi: <pub-id pub-id-type="doi">10.3390/electronics10202470</pub-id></citation></ref>
<ref id="ref24"><label>24.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Simonyan</surname><given-names>K</given-names></name> <name><surname>Zisserman</surname><given-names>A.</given-names></name></person-group>, <article-title>Very deep convolutional networks for large-scale image recognition</article-title>. arXiv preprint arXiv:1409.1556, (<year>2014</year>).</citation></ref>
<ref id="ref25"><label>25.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pan</surname><given-names>X</given-names></name> <name><surname>Luo</surname><given-names>Z</given-names></name> <name><surname>Zhou</surname><given-names>L</given-names></name></person-group>. <article-title>Comprehensive survey of state-of-the-art convolutional neural network architectures and their applications in image classification. Innovations in applied</article-title>. <source>Eng Technol</source>. (<year>2022</year>) <volume>1</volume>:<fpage>1</fpage>&#x2013;<lpage>16</lpage>. doi: <pub-id pub-id-type="doi">10.62836/iaet.v1i1.1006</pub-id></citation></ref>
<ref id="ref26"><label>26.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname><given-names>K</given-names></name> <name><surname>Zhang</surname><given-names>X</given-names></name> <name><surname>Ren</surname><given-names>S</given-names></name> <name><surname>Sun</surname><given-names>J</given-names></name></person-group>. <article-title>Deep residual learning for image recognition</article-title>. <source>Proc IEEE Conf Comput Vis Pattern Recognit</source>. (<year>2016</year>) <fpage>770</fpage>&#x2013;<lpage>778</lpage>. doi: <pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id></citation></ref>
<ref id="ref27"><label>27.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Huang</surname><given-names>G.</given-names></name> <name><surname>Liu</surname><given-names>Z.</given-names></name> <name><surname>Van Der Maaten</surname><given-names>L.</given-names></name> <name><surname>Weinberger</surname><given-names>K.Q.</given-names></name></person-group> <article-title>Densely connected convolutional networks</article-title>. in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>. <publisher-loc>Los Alamitos, CA, USA</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>. (<year>2017</year>).</citation></ref>
<ref id="ref28"><label>28.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Tan</surname><given-names>M.</given-names></name> <name><surname>Le</surname><given-names>Q.</given-names></name></person-group> <article-title>EfficientNet: rethinking model scaling for convolutional neural networks</article-title>, in <conf-name>Proceedings of the 36th International Conference on Machine Learning</conf-name>, <person-group person-group-type="editor"><name><surname>Kamalika</surname><given-names>C.</given-names></name> <name><surname>Ruslan</surname><given-names>S.</given-names></name></person-group>, editors. (<year>2019</year>), PMLR: Proceedings of Machine Learning Research. p. <fpage>6105</fpage>&#x2013;<lpage>6114</lpage>.</citation></ref>
<ref id="ref29"><label>29.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Dosovitskiy</surname><given-names>A.</given-names></name> <name><surname>Beyer</surname><given-names>L.</given-names></name> <name><surname>Kolesnikov</surname><given-names>A.</given-names></name> <name><surname>Weissenborn</surname><given-names>D.</given-names></name> <name><surname>Zhai</surname><given-names>X.</given-names></name> <name><surname>Unterthiner</surname><given-names>T.</given-names></name> <etal/></person-group>., <article-title>An image is worth 16x16 words: transformers for image recognition at scale</article-title>. arXiv preprint arXiv:2010.11929 (<year>2020</year>).</citation></ref>
<ref id="ref30"><label>30.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Reddy</surname><given-names>EMK</given-names></name> <name><surname>Gurrala</surname><given-names>A</given-names></name> <name><surname>Hasitha</surname><given-names>VB</given-names></name> <name><surname>Kumar</surname><given-names>KVR</given-names></name></person-group>. <article-title>Introduction to naive Bayes and a review on its subtypes with applications</article-title>. <source>Bayesian reasoning and gaussian processes for machine learning applications</source>. <publisher-name>Chapman and Hall/CRC</publisher-name>. (<year>2022</year>) <fpage>1</fpage>&#x2013;<lpage>14</lpage>. doi: <pub-id pub-id-type="doi">10.1201/9781003164265-1</pub-id></citation></ref>
<ref id="ref31"><label>31.</label><citation citation-type="book"><person-group person-group-type="author"><name><surname>Villarreal-Hern&#x00E1;ndez</surname><given-names>J&#x00C1;</given-names></name> <name><surname>Morales-Rodr&#x00ED;guez</surname><given-names>ML</given-names></name> <name><surname>Rangel-Valdez</surname><given-names>N</given-names></name> <name><surname>G&#x00F3;mez-Santill&#x00E1;n</surname><given-names>C</given-names></name></person-group>. <article-title>Reusability analysis of K-nearest neighbors variants for classification models</article-title> In: <source>Innovations in machine and deep learning: case studies and applications</source>. <publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2023</year>) <fpage>63</fpage>&#x2013;<lpage>81</lpage>.</citation></ref>
<ref id="ref32"><label>32.</label><citation citation-type="book"><person-group person-group-type="author"><name><surname>Pisner</surname><given-names>DA</given-names></name> <name><surname>Schnyer</surname><given-names>DM</given-names></name></person-group>. <article-title>Support vector machine</article-title>. In: <source>Machine learning</source>. <publisher-loc>Amsterdam, Netherlands</publisher-loc>: <publisher-name>Elsevier</publisher-name> (<year>2020</year>) <fpage>101</fpage>&#x2013;<lpage>21</lpage>.</citation></ref>
<ref id="ref33"><label>33.</label><citation citation-type="book"><person-group person-group-type="author"><name><surname>Rincy</surname><given-names>TN</given-names></name> <name><surname>Gupta</surname><given-names>R</given-names></name></person-group>. <article-title>Ensemble learning techniques and its efficiency in machine learning: A survey</article-title> In: <source>2nd international conference on data, engineering and applications (IDEA)</source>: <publisher-name>IEEE</publisher-name> (<year>2020</year>)</citation></ref>
<ref id="ref34"><label>34.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>M&#x2019;hamdi</surname><given-names>O</given-names></name> <name><surname>Tak&#x00E1;cs</surname><given-names>S</given-names></name> <name><surname>Palot&#x00E1;s</surname><given-names>G</given-names></name> <name><surname>Ilahy</surname><given-names>R</given-names></name> <name><surname>Helyes</surname><given-names>L</given-names></name> <name><surname>P&#x00E9;k</surname><given-names>Z</given-names></name></person-group>. <article-title>A comparative analysis of XGBoost and neural network models for predicting some tomato fruit quality traits from environmental and meteorological data</article-title>. <source>Plants</source>. (<year>2024</year>) <volume>13</volume>:<fpage>746</fpage>. doi: <pub-id pub-id-type="doi">10.3390/plants13050746</pub-id>, PMID: <pub-id pub-id-type="pmid">38475592</pub-id></citation></ref>
<ref id="ref35"><label>35.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Pradipta</surname><given-names>G.A.</given-names></name> <name><surname>Wardoyo</surname><given-names>R.</given-names></name> <name><surname>Musdholifah</surname><given-names>A.</given-names></name> <name><surname>Sanjaya</surname><given-names>I.N.H.</given-names></name> <name><surname>Ismail</surname><given-names>M.</given-names></name></person-group> <article-title>SMOTE for handling imbalanced data problem: a review</article-title>. in <conf-name>2021 Sixth International Conference on Informatics and Computing (ICIC)</conf-name>. (<year>2021</year>). <publisher-name>IEEE</publisher-name>.</citation></ref>
<ref id="ref36"><label>36.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>He</surname><given-names>H.</given-names></name> <name><surname>Bai</surname><given-names>Y.</given-names></name> <name><surname>Garcia</surname><given-names>E. A.</given-names></name> <name><surname>Li</surname><given-names>S.</given-names></name></person-group> <article-title>ADASYN: adaptive synthetic sampling approach for imbalanced learning</article-title>. in <conf-name>2008 IEEE International Joint Conference on Neural Networks (IEEE World Congress on Computational Intelligence)</conf-name>. (<year>2008</year>). <publisher-name>IEEE</publisher-name>.</citation></ref>
<ref id="ref37"><label>37.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Chengsheng</surname><given-names>T.</given-names></name> <name><surname>Huacheng</surname><given-names>L.</given-names></name> <name><surname>Bing</surname><given-names>X.</given-names></name></person-group><article-title>AdaBoost typical algorithm and its application research</article-title>. In <conf-name>MATEC Web of Conferences</conf-name>. <publisher-loc>Les Ulis, France</publisher-loc>: <publisher-name>EDP Sciences</publisher-name>. (<year>2017</year>).</citation></ref>
<ref id="ref38"><label>38.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname><given-names>P</given-names></name> <name><surname>Ergu</surname><given-names>D</given-names></name> <name><surname>Liu</surname><given-names>F</given-names></name> <name><surname>Cai</surname><given-names>Y</given-names></name> <name><surname>Ma</surname><given-names>B</given-names></name></person-group>. <article-title>A review of Yolo algorithm developments</article-title>. <source>Proc Comput Sci</source>. (<year>2022</year>) <volume>199</volume>:<fpage>1066</fpage>&#x2013;<lpage>73</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.procs.2022.01.135</pub-id></citation></ref>
<ref id="ref39"><label>39.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Sohan</surname><given-names>M.</given-names></name> <name><surname>Sai Ram</surname><given-names>T.</given-names></name> <name><surname>Rami Reddy</surname><given-names>C.V.</given-names></name></person-group>. <article-title>A review on YOLOv8 and its advancements</article-title>. In <conf-name>Data intelligence and cognitive informatics</conf-name>. <publisher-name>Springer Nature Singapore</publisher-name>. <publisher-loc>Singapore</publisher-loc>. (<year>2024</year>).</citation></ref>
<ref id="ref40"><label>40.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Khanam</surname><given-names>R</given-names></name> <name><surname>Hussain</surname><given-names>M</given-names></name></person-group>. <article-title>A review of YOLOv12: attention-based enhancements vs. previous versions</article-title>. arXiv preprint arXiv:2504.11995 (<year>2025</year>). doi: <pub-id pub-id-type="doi">10.48550/arXiv.2504.11995</pub-id>,</citation></ref>
<ref id="ref41"><label>41.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Lei</surname><given-names>M</given-names></name> <name><surname>Li</surname><given-names>S</given-names></name> <name><surname>Wu</surname><given-names>Y</given-names></name> <name><surname>Hu</surname><given-names>H</given-names></name> <name><surname>Zhou</surname><given-names>Y</given-names></name> <name><surname>Zheng</surname><given-names>X</given-names></name> <etal/></person-group>. <article-title>YOLOv13: real-time object detection with hypergraph-enhanced adaptive visual perception</article-title>. arXiv 2025. arXiv preprint arXiv:2506.17733 (<year>2025</year>). doi: <pub-id pub-id-type="doi">10.48550/arXiv.2506.17733</pub-id>,</citation></ref>
<ref id="ref42"><label>42.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Du</surname><given-names>Z</given-names></name> <name><surname>Cui</surname><given-names>M</given-names></name> <name><surname>Xu</surname><given-names>X</given-names></name> <name><surname>Bai</surname><given-names>Z</given-names></name> <name><surname>Han</surname><given-names>J</given-names></name> <name><surname>Li</surname><given-names>W</given-names></name> <etal/></person-group>. <article-title>Harnessing multimodal data fusion to advance accurate identification of fish feeding intensity</article-title>. <source>Biosyst Eng</source>. (<year>2024</year>) <volume>246</volume>:<fpage>135</fpage>&#x2013;<lpage>49</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2024.08.001</pub-id></citation></ref>
<ref id="ref43"><label>43.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname><given-names>F</given-names></name> <name><surname>Zhang</surname><given-names>C</given-names></name> <name><surname>Geng</surname><given-names>B</given-names></name></person-group>. <article-title>Deep multimodal data fusion</article-title>. <source>ACM Comput Surv</source>. (<year>2024</year>) <volume>56</volume>:<fpage>1</fpage>&#x2013;<lpage>36</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3649447</pub-id>, PMID: <pub-id pub-id-type="pmid">40727313</pub-id></citation></ref>
<ref id="ref44"><label>44.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>He</surname><given-names>K.</given-names></name> <name><surname>Gkioxari</surname><given-names>G.</given-names></name> <name><surname>Doll&#x00E1;r</surname><given-names>P.</given-names></name> <name><surname>Girshick</surname><given-names>R.</given-names></name></person-group> <article-title>Mask R-CNN</article-title>. in <conf-name>Proceedings of the IEEE International Conference on Computer Vision</conf-name>. <publisher-loc>Los Alamitos, CA, USA</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name> (<year>2017</year>).</citation></ref>
<ref id="ref45"><label>45.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>H</given-names></name> <name><surname>Shi</surname><given-names>H</given-names></name> <name><surname>Du</surname><given-names>A</given-names></name> <name><surname>Mao</surname><given-names>Y</given-names></name> <name><surname>Fan</surname><given-names>K</given-names></name> <name><surname>Wang</surname><given-names>Y</given-names></name> <etal/></person-group>. <article-title>Symptom recognition of disease and insect damage based on mask R-CNN, wavelet transform, and F-RNet</article-title>. <source>Front Plant Sci</source>. (<year>2022</year>) <volume>13</volume>:<fpage>922797</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpls.2022.922797</pub-id>, PMID: <pub-id pub-id-type="pmid">35937317</pub-id></citation></ref>
<ref id="ref46"><label>46.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Su</surname><given-names>M</given-names></name> <name><surname>Li</surname><given-names>C</given-names></name> <name><surname>Qi</surname><given-names>S</given-names></name> <name><surname>Yang</surname><given-names>D</given-names></name> <name><surname>Jiang</surname><given-names>N</given-names></name> <name><surname>Yin</surname><given-names>B</given-names></name> <etal/></person-group>. <article-title>A molecular epidemiological investigation of PEDV in China: characterization of co-infection and genetic diversity of S1-based genes</article-title>. <source>Transbound Emerg Dis</source>. (<year>2020</year>) <volume>67</volume>:<fpage>1129</fpage>&#x2013;<lpage>40</lpage>. doi: <pub-id pub-id-type="doi">10.1111/tbed.13439</pub-id>, PMID: <pub-id pub-id-type="pmid">31785090</pub-id></citation></ref>
<ref id="ref47"><label>47.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>Y</given-names></name> <name><surname>Zhang</surname><given-names>Y</given-names></name> <name><surname>Wang</surname><given-names>X</given-names></name> <name><surname>Zhou</surname><given-names>J</given-names></name> <name><surname>Ma</surname><given-names>L</given-names></name> <name><surname>Li</surname><given-names>J</given-names></name> <etal/></person-group>. <article-title>Transmissible gastroenteritis virus: an update review and perspective</article-title>. <source>Viruses</source>. (<year>2023</year>) <volume>15</volume>:<fpage>359</fpage>. doi: <pub-id pub-id-type="doi">10.3390/v15020359</pub-id>, PMID: <pub-id pub-id-type="pmid">36851573</pub-id></citation></ref>
<ref id="ref48"><label>48.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Obradovic</surname><given-names>MR</given-names></name> <name><surname>Wilson</surname><given-names>HL</given-names></name></person-group>. <article-title>Immune response and protection against <italic>Lawsonia intracellularis</italic> infections in pigs</article-title>. <source>Vet Immunol Immunopathol</source>. (<year>2020</year>) <volume>219</volume>:<fpage>109959</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.vetimm.2019.109959</pub-id>, PMID: <pub-id pub-id-type="pmid">31710909</pub-id></citation></ref>
<ref id="ref49"><label>49.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luppi</surname><given-names>A</given-names></name></person-group>. <article-title>Swine enteric colibacillosis: diagnosis, therapy and antimicrobial resistance</article-title>. <source>Porcine Health Manag</source>. (<year>2017</year>) <volume>3</volume>:<fpage>16</fpage>&#x2013;<lpage>8</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s40813-017-0063-4</pub-id>, PMID: <pub-id pub-id-type="pmid">28794894</pub-id></citation></ref>
</ref-list>
</back>
</article>