<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bioeng. Biotechnol.</journal-id>
<journal-title>Frontiers in Bioengineering and Biotechnology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bioeng. Biotechnol.</abbrev-journal-title>
<issn pub-type="epub">2296-4185</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1247112</article-id>
<article-id pub-id-type="doi">10.3389/fbioe.2023.1247112</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Bioengineering and Biotechnology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>An approach to the diagnosis of lumbar disc herniation using deep learning models</article-title>
<alt-title alt-title-type="left-running-head">Prisilla et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fbioe.2023.1247112">10.3389/fbioe.2023.1247112</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Prisilla</surname>
<given-names>Ardha Ardea</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2109763/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Guo</surname>
<given-names>Yue Leon</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jan</surname>
<given-names>Yih-Kuen</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/47001/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lin</surname>
<given-names>Chih-Yang</given-names>
</name>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1438519/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lin</surname>
<given-names>Fu-Yu</given-names>
</name>
<xref ref-type="aff" rid="aff8">
<sup>8</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liau</surname>
<given-names>Ben-Yi</given-names>
</name>
<xref ref-type="aff" rid="aff9">
<sup>9</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1170619/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Tsai</surname>
<given-names>Jen-Yung</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1335531/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ardhianto</surname>
<given-names>Peter</given-names>
</name>
<xref ref-type="aff" rid="aff10">
<sup>10</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1674679/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Pusparani</surname>
<given-names>Yori</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff11">
<sup>11</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2177386/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Lung</surname>
<given-names>Chi-Wen</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<xref ref-type="aff" rid="aff12">
<sup>12</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/230630/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Fashion Design</institution>, <institution>LaSalle College Jakarta</institution>, <addr-line>Jakarta</addr-line>, <country>Indonesia</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Digital Media Design</institution>, <institution>Asia University</institution>, <addr-line>Taichung</addr-line>, <country>Taiwan</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Environmental and Occupational Medicine</institution>, <institution>College of Medicine</institution>, <institution>National Taiwan University (NTU) and NTU Hospital</institution>, <addr-line>Taipei</addr-line>, <country>Taiwan</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Graduate Institute of Environmental and Occupational Health Sciences</institution>, <institution>College of Public Health</institution>, <institution>National Taiwan University</institution>, <addr-line>Taipei</addr-line>, <country>Taiwan</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>National Institute of Environmental Health Sciences</institution>, <institution>National Health Research Institutes</institution>, <addr-line>Miaoli</addr-line>, <country>Taiwan</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Rehabilitation Engineering Lab</institution>, <institution>Department of Kinesiology and Community Health</institution>, <institution>University of Illinois at Urbana-Champaign</institution>, <addr-line>Urbana</addr-line>, <addr-line>IL</addr-line>, <country>United States</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>Department of Mechanical Engineering</institution>, <institution>National Central University</institution>, <addr-line>Taoyuan</addr-line>, <country>Taiwan</country>
</aff>
<aff id="aff8">
<sup>8</sup>
<institution>Department of Neurology</institution>, <institution>China Medical University Hospital</institution>, <addr-line>Taichung</addr-line>, <country>Taiwan</country>
</aff>
<aff id="aff9">
<sup>9</sup>
<institution>Department of Automatic Control Engineering</institution>, <institution>Feng Chia University</institution>, <addr-line>Taichung</addr-line>, <country>Taiwan</country>
</aff>
<aff id="aff10">
<sup>10</sup>
<institution>Department of Visual Communication Design</institution>, <institution>Soegijapranata Catholic University</institution>, <addr-line>Semarang</addr-line>, <country>Indonesia</country>
</aff>
<aff id="aff11">
<sup>11</sup>
<institution>Department of Visual Communication Design</institution>, <institution>Budi Luhur University</institution>, <addr-line>Jakarta</addr-line>, <country>Indonesia</country>
</aff>
<aff id="aff12">
<sup>12</sup>
<institution>Department of Creative Product Design</institution>, <institution>Asia University</institution>, <addr-line>Taichung</addr-line>, <country>Taiwan</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/149739/overview">Bernardo Innocenti</ext-link>, Universit&#xe9; libre de Bruxelles, Belgium</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2365404/overview">Luca Ciriello</ext-link>, Polytechnic University of Milan, Italy</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1806665/overview">Jan Kubicek</ext-link>, VSB-Technical University of Ostrava, Czechia</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Chi-Wen Lung, <email>cwlung@asia.edu.tw</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>09</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>11</volume>
<elocation-id>1247112</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>06</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>09</day>
<month>08</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Prisilla, Guo, Jan, Lin, Lin, Liau, Tsai, Ardhianto, Pusparani and Lung.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Prisilla, Guo, Jan, Lin, Lin, Liau, Tsai, Ardhianto, Pusparani and Lung</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<bold>Background:</bold> In magnetic resonance imaging (MRI), lumbar disc herniation (LDH) detection is challenging due to the various shapes, sizes, angles, and regions associated with bulges, protrusions, extrusions, and sequestrations. Lumbar abnormalities in MRI can be detected automatically by using deep learning methods. As deep learning models gain recognition, they may assist in diagnosing LDH with MRI images and provide initial interpretation in clinical settings. YOU ONLY LOOK ONCE (YOLO) model series are often used to train deep learning algorithms for real-time biomedical image detection and prediction. This study aims to confirm which YOLO models (YOLOv5, YOLOv6, and YOLOv7) perform well in detecting LDH in different regions of the lumbar intervertebral disc.</p>
<p>
<bold>Materials and methods:</bold> The methodology involves several steps, including converting DICOM images to JPEG, reviewing and selecting MRI slices for labeling and augmentation using ROBOFLOW, and constructing YOLOv5x, YOLOv6, and YOLOv7 models based on the dataset. The training dataset was combined with the radiologist&#x2019;s labeling and annotation, and then the deep learning models were trained using the training/validation dataset.</p>
<p>
<bold>Results:</bold> Our result showed that the 550-dataset with augmentation (AUG) or without augmentation (non-AUG) in YOLOv5x generates satisfactory training performance in LDH detection. The AUG dataset overall performance provides slightly higher accuracy than the non-AUG. YOLOv5x showed the highest performance with 89.30% mAP compared to YOLOv6, and YOLOv7. Also, YOLOv5x in non-AUG dataset showed the balance LDH region detections in L2-L3, L3-L4, L4-L5, and L5-S1 with above 90%. And this illustrates the competitiveness of using non-AUG dataset to detect LDH.</p>
<p>
<bold>Conclusion:</bold> Using YOLOv5x and the 550 augmented dataset, LDH can be detected with promising both in non-AUG and AUG dataset. By utilizing the most appropriate YOLO model, clinicians have a greater chance of diagnosing LDH early and preventing adverse effects for their patients.</p>
</abstract>
<kwd-group>
<kwd>augmentation</kwd>
<kwd>automatic detection</kwd>
<kwd>low back pain</kwd>
<kwd>MRI</kwd>
<kwd>YOLO models</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Biomechanics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Highlight</title>
<p>
<list list-type="simple">
<list-item>
<p>1. The YOLOv5x in the dataset with augmentation (AUG) successfully performed the highest mAP compared to all YOLO models tested, which indicated the model with the highest performance model to detect lumbar disc herniation (LDH).</p>
</list-item>
<list-item>
<p>2. YOLOv5x without augmentation (non-AUG) dataset performed well in detecting LDH in L2-L3, L3-L4, L4-L5, and L5-S1 regions with values above 90%, which showed effectiveness in using a non-AUG dataset for training.</p>
</list-item>
<list-item>
<p>3. YOLOv5x showed the shortest training duration and the lightest weight compared to YOLOv6 and YOLOv7, indicating the most efficient LDH detection model.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s2">
<title>1 Introduction</title>
<p>Lumbar disc herniation (LDH) is caused by the bulging or rupture of a spinal disc segment, misaligning its position, and irritating the nerve roots, which causes sciatica (<xref ref-type="bibr" rid="B40">Vialle et al., 2010</xref>; <xref ref-type="bibr" rid="B4">Amin et al., 2017</xref>). In particular, the lumbar vertebrae are more susceptible to misalignment since they support the body&#x2019;s weight (<xref ref-type="bibr" rid="B30">Mushtaq et al., 2022</xref>). LDH occurs in 2%&#x2013;3% of the worldwide population, interfering with everyday activities and productivity (<xref ref-type="bibr" rid="B40">Vialle et al., 2010</xref>). The cost of treating lumbar disc herniation in the United States with medications and surgery amounted to $4.0 billion in 2015 (<xref ref-type="bibr" rid="B29">Martin et al., 2019</xref>).</p>
<p>Four most commonly occurring forms of LDH include bulging, protrusion, extrusion, and sequestration (<xref ref-type="bibr" rid="B16">Gopalakrishnan et al., 2015</xref>). These forms are caused by a rupture of the fibrous layers of the annulus (the bony outer shell), which can cause a leak of the nucleus pulposus (soft inner core) and irritate adjacent nerve roots (<xref ref-type="bibr" rid="B16">Gopalakrishnan et al., 2015</xref>). An early diagnosis of LDH can assist in curing the disease in its earliest stages and protect the patient from harmful repercussions. One of the most used medical imaging techniques for diagnosing LDH is magnetic resonance imaging (MRI) (<xref ref-type="bibr" rid="B9">Choi et al., 2017</xref>). The use of MRI in diagnosing LDH has been widely adopted because of its ability to reveal the shape of intervertebral discs. Intervertebral discs with elliptical shapes are robust, whereas discs with abnormal shapes are deformed and flattened (<xref ref-type="bibr" rid="B3">Alomari et al., 2014</xref>; <xref ref-type="bibr" rid="B8">Chen et al., 2021</xref>).</p>
<p>However, clinicians must undergo extensive training to interpret and analyze MRI of LDH. Although well-trained medical professionals can analyze the MRI, the diagnosis results might be inconsistent (<xref ref-type="bibr" rid="B3">Alomari et al., 2014</xref>). Radiologists are reported to be biased in their interpretation of MRIs, with disagreement on a variation of bulging discs, which indicates the need for standardized mechanisms in MRI interpretations (<xref ref-type="bibr" rid="B3">Alomari et al., 2014</xref>). Researchers have demonstrated that MRI LDH can be detected automatically by using deep learning methods, which can increase radiology practice efficiency (<xref ref-type="bibr" rid="B5">Azimi et al., 2020</xref>). By using deep learning to interpret MRI images, the existing bias variability can be reduced, and diagnostic decisions can be standardized (<xref ref-type="bibr" rid="B3">Alomari et al., 2014</xref>). It has been shown that deep learning architectures can successfully address image recognition and classification accurately from MRI using automated learning features (<xref ref-type="bibr" rid="B39">Tsai et al., 2021</xref>). Consequently, researchers strive to improve performance results while designing different deep learning architectures (<xref ref-type="bibr" rid="B5">Azimi et al., 2020</xref>).</p>
<p>In the past, two-stage detectors deep learning models were used to classify and detect LDH in MRI. <xref ref-type="bibr" rid="B3">Alomari et al. (2014)</xref> proposed utilizing a coordinated active shape and a gradient vector flow active contour models to extract shape features for detecting LDH. Wang et al., proposed a two-stage detector fusion model that utilizes DenseNet and Inception-Resnet-V2 models to increase the number of image features and improve image recognition accuracy (<xref ref-type="bibr" rid="B42">Wang et al., 2020</xref>). Su et al., developed another two-stage detector model using ResNet-50 that consists of three fully connected networks based on a backbone network for feature extraction and performs classification tasks on lumbar MRI of LDH (<xref ref-type="bibr" rid="B35">Su et al., 2022</xref>).</p>
<p>Several studies adopted the single-stage detector YOU ONLY LOOK ONCE (YOLO) model to train deep learning algorithms for real-time biomedical image detection and prediction established on anchor base and intersection over union techniques (<xref ref-type="bibr" rid="B39">Tsai et al., 2021</xref>; <xref ref-type="bibr" rid="B17">Guinebert et al., 2022</xref>; <xref ref-type="bibr" rid="B30">Mushtaq et al., 2022</xref>). Our previous study tested only YOLOv3 in Darknet to detect LDH in MRI (<xref ref-type="bibr" rid="B39">Tsai et al., 2021</xref>). However, a variety of more recent YOLO models have been developed, including YOLOv5, YOLOv6, and YOLOv7 (<xref ref-type="bibr" rid="B21">Jocher, 2020</xref>; <xref ref-type="bibr" rid="B26">Li et al., 2022</xref>; <xref ref-type="bibr" rid="B41">Wang et al., 2022</xref>). Contrary to previous releases based on Darknet, these models are based on PyTorch, which is more typically used for computer vision and natural language processing (<xref ref-type="bibr" rid="B21">Jocher, 2020</xref>; <xref ref-type="bibr" rid="B26">Li et al., 2022</xref>; <xref ref-type="bibr" rid="B41">Wang et al., 2022</xref>).</p>
<p>YOLOv5 has been used in biomedical image detection with promising results (<xref ref-type="bibr" rid="B30">Mushtaq et al., 2022</xref>). According to Mushtaq et al., YOLOv5 performance has been reported to be higher in accuracy than YOLOv3 at identifying lumbar lordotic angles (LLA) and lumbosacral angles (LSA) (<xref ref-type="bibr" rid="B30">Mushtaq et al., 2022</xref>). However, YOLOv6 and YOLOv7 were published in June and July 2022 and remain relatively novel (<xref ref-type="bibr" rid="B26">Li et al., 2022</xref>; <xref ref-type="bibr" rid="B41">Wang et al., 2022</xref>). An evaluation of YOLOv5, YOLOv6, and YOLOv7 has been conducted to determine which model detects safety helmets the most effectively. YOLOv7 outperformed YOLOv5 and YOLOv6 in detecting safety helmets (<xref ref-type="bibr" rid="B44">Yung et al., 2022</xref>). However, further research is required to confirm YOLOv6 and YOLOv7 effectiveness compared to YOLOv5 in detecting biomedical images.</p>
<p>To the best of our knowledge, YOLOv5, YOLOv6, and YOLOv7 have not yet been evaluated for their ability to detect LDH in MRI, although the three models provide promising object detection performance. In YOLOv5, several features performed well on the validation set and were more efficient during interpretation (<xref ref-type="bibr" rid="B21">Jocher, 2020</xref>). YOLOv6 also outperforms YOLOv5 in detection accuracy and is more confident about its label in industrial applications (<xref ref-type="bibr" rid="B26">Li et al., 2022</xref>). YOLO7 is also reported to be more accurate in detecting Microsoft common objects in context (MS COCO) than previous YOLO models (<xref ref-type="bibr" rid="B41">Wang et al., 2022</xref>).</p>
<p>Our previous study has shown that YOLOv3 could detect LDH in MRI best with image augmentation (<xref ref-type="bibr" rid="B39">Tsai et al., 2021</xref>). However, since YOLOv5, YOLOv6, and YOLOv7 have improved image detection features, we hypothesized our study aims as follows:<list list-type="simple">
<list-item>
<p>&#x2022; &#x2003;Based on YOLO metrics performance, we would like to compare YOLOv5, YOLOv6, and YOLOv7 models for LDH detection to determine which model would provide the highest level of accuracy.</p>
</list-item>
<list-item>
<p>&#x2022; &#x2003;To assess further how well YOLO models perform, we would like to compare them with and without the augmentation dataset.</p>
</list-item>
<list-item>
<p>&#x2022; &#x2003;The three YOLO models will be used to determine the optimal training duration for clinical use.</p>
</list-item>
</list>
</p>
<p>An accurate model can assist clinicians in determining the LDH earlier in MRI images. Following this, we will discuss the materials and methods used for the proposed deep learning models, which include datasets and detailed methodologies. In addition, we will discuss the performance results of models which achieved the LDH detection results and explain them in greater detail.</p>
</sec>
<sec sec-type="materials|methods" id="s3">
<title>2 Materials and methods</title>
<sec id="s3-1">
<title>2.1 Image dataset</title>
<p>MRI images were derived from a publicly available dataset of lumbar spines by <xref ref-type="bibr" rid="B36">Sudirman et al. (2019a)</xref> (<ext-link ext-link-type="uri" xlink:href="https://data.mendeley.com/datasets/k57fr854j2/2">https://data.mendeley.com/datasets/k57fr854j2/2</ext-link>) based on an anonymized clinical study. The dataset was collected from patients at the Irbid Specialty Hospital in Jordan who reported symptomatic back pain between September 2015 and July 2016 (<xref ref-type="bibr" rid="B2">Al-Kafri et al., 2019</xref>; <xref ref-type="bibr" rid="B36">Sudirman et al., 2019a</xref>). The lumbar MRI images were stored in Digital Imaging and Communications in Medicine (DICOM) files. In order to ensure similar physiology for the lumbar spine, the MRI images were taken from patients at least 17&#xa0;years of age (<xref ref-type="bibr" rid="B2">Al-Kafri et al., 2019</xref>). MRI images include T1 and T2 weighted images with sagittal and axial views. Most images have a resolution of 320x320 pixels with a precision of 12-bit per pixel (<xref ref-type="bibr" rid="B36">Sudirman et al., 2019a</xref>). We extracted the DICOM data and converted the images to JPEG by using the provided MATLAB code from the dataset source (<xref ref-type="bibr" rid="B37">Sudirman et al., 2019b</xref>).</p>
<p>Before deep learning model training, all images must be examined, standardized, and transformed into organized data (<xref ref-type="bibr" rid="B39">Tsai et al., 2021</xref>) (<xref ref-type="fig" rid="F1">Figure 1</xref>). From the DICOM raw data, we used T2-weighted images to obtain better brightness and darkness features. A clinical trial has been conducted to demonstrate the acceptable equivalent of using T2-WI in sagittal to assess LDH (<xref ref-type="bibr" rid="B3">Alomari et al., 2014</xref>). Therefore, we used the T2-WI in the sagittal view in this study. In this study, a total of 550 images were used from 110 different subjects. The ground truth and MATLAB code for data extraction were also provided from the source (<xref ref-type="bibr" rid="B37">Sudirman et al., 2019b</xref>). We supervised the selection of subjects for this study based on the ground truth, which included only individuals with disc herniations between L1 and S1. We used five midsagittal slices for each subject, including the middle slice and two symmetrical slices on either side of the vertebral body, representing approximately the full transverse diameter of the vertebral body (<xref ref-type="bibr" rid="B14">Friska and Sudirman, 2021</xref>). We selected five slices that provided clear views of the lumbar region. The 550 images were divided into 80% training images (440 images from 88 subjects) and 20% validation images (110 images from 22 subjects). The sagittal view of upper and lower lumbar MRI has been used in a clinical trial for the detection and segmentation of lumbar MRI (<xref ref-type="bibr" rid="B15">Ghosh and Chaudhary, 2014</xref>). Hence in this study, we use the sagittal view as the initial identification of LDH. Further analysis was performed by combining the training and validation datasets with radiologists&#x2019; diagnosis records.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Process flow of data processing for object detection in deep learning.</p>
</caption>
<graphic xlink:href="fbioe-11-1247112-g001.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>2.2 Object detection architecture</title>
<p>Object detection techniques have been widely utilized in many medical diagnostic applications, such as YOLO algorithms for detecting medical images in MRI (<xref ref-type="bibr" rid="B39">Tsai et al., 2021</xref>; <xref ref-type="bibr" rid="B30">Mushtaq et al., 2022</xref>). The YOLO framework uses a single-stage detection approach for real-time object recognition. Single-stage detectors are designed to detect objects using relatively simple architecture by focusing on all the spatial regions. It improves detection accuracy and reduces the time required for inferences (<xref ref-type="bibr" rid="B21">Jocher, 2020</xref>; <xref ref-type="bibr" rid="B26">Li et al., 2022</xref>; <xref ref-type="bibr" rid="B41">Wang et al., 2022</xref>). Advances in YOLO models over the years have resulted in different performance levels among the models. This study assesses the MRI lumbar image and determines the bounding box based on the performance of the YOLOv5, YOLOv6, and YOLOv7 models.</p>
<p>YOLOv5 uses an adaptive anchor strategy known as the auto anchor, in which the backbone comprises a focused structure and a CSP backbone. As a pre-training tool, auto anchor checks and adjusts anchor boxes if their fit is not optimal for the dataset and training settings. YOLOv5 network also uses a PANet neck to improve localization within layers (<xref ref-type="bibr" rid="B21">Jocher, 2020</xref>) (<xref ref-type="fig" rid="F2">Figure 2A</xref>). Our study uses the YOLOv5x model, which is ideal for datasets containing smaller objects and is designed to provide high performance. According to a test using the MS COCO dataset test-dev 2017, YOLOv5x achieved an average percentage of 50.7% with an image size of 640 pixels and 200 frames per second (FPS) speed using an NVIDIA V100 (<xref ref-type="bibr" rid="B38">Terven and Cordova-Esparza, 2023</xref>). In YOLOv6, there are two scaled re-parameterizable backbones and necks to accommodate models of different sizes and a decoupled head efficiently implemented with a hybrid channel method. The hybrid channel has both single and multiple channels with enhanced quantization techniques that employ post-training quantization and channel-wise distillation. This has resulted in faster and more accurate detectors than previous versions of YOLOv5 (<xref ref-type="bibr" rid="B26">Li et al., 2022</xref>; <xref ref-type="bibr" rid="B38">Terven and Cordova-Esparza, 2023</xref>) (<xref ref-type="fig" rid="F2">Figure 2B</xref>). According to a test using the MS COCO dataset test-dev 2017, the largest model of YOLOv6 achieved an average percentage of 57.2% with a speed of 29 FPS using an NVIDIA Tesla T4. Yolov7 proposed several architecture changes and a number of &#x201c;bag-of-freebies,&#x201d; which significantly increased the model&#x2019;s accuracy without affecting its inference speed (<xref ref-type="bibr" rid="B38">Terven and Cordova-Esparza, 2023</xref>). YOLOv7 uses the extended efficient layer aggregation network (E-ELAN) backbone, model scaling, and model re-parameterization. The E-ELAN combines the characteristics of different groups by shuffling and merging cardinality in order to enhance the network&#x2019;s learning capability without destroying the gradient path. The target detector in YOLOv7 is also implemented with extend and compound scaling, resulting in a substantial acceleration in detection (<xref ref-type="bibr" rid="B41">Wang et al., 2022</xref>) (<xref ref-type="fig" rid="F2">Figure 2C</xref>). According to a test using the MS COCO dataset test-dev 2017, YOLOv7-E6 achieved an average percentage of 55.9% and AP<sub>50</sub> of 73.5% with an image size of 1,280 pixels and 50 FPS on an NVIDIA V100 (<xref ref-type="bibr" rid="B38">Terven and Cordova-Esparza, 2023</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>YOLO series Network Architecture; <bold>(A)</bold> YOLOv5 Network Architecture; <bold>(B)</bold> YOLOv6 Network Architecture; <bold>(C)</bold> YOLOv7 Network Architecture. YOLO, You Only Look Once. CSP, Cross Stage Partial. SPP, Spatial Pyramid Pooling. Rep, reparameterized. CBS, Convolutional, Batch normalization, SiLu activation blocks; E-ELAN, Extended efficient layer aggregation network; MP1/MP2, Max Pool-1/Max Pool-2; Sppcspc, Spatial Pyramid Pooling and Convolutional Spatial Pyramid Pool structure.</p>
</caption>
<graphic xlink:href="fbioe-11-1247112-g002.tif"/>
</fig>
<p>We used boundary box labeling in Roboflow to detect LDH of the lumbar intervertebral disc in 5 regions, the first and second lumbar vertebrae (L1-L2), the second and third lumbar vertebrae (L2-L3), third and fourth lumbar vertebrae (L3-L4), fourth and fifth lumbar vertebrae (L4-L5), the fifth lumbar vertebrae and the first sacral vertebrae (L5-S1). We put the LDH location labels on both the training and validation dataset. The lumbar vertebrae and disc sections are identified on the MRI scans, namely, L1-L2, L2-L3, L3-L4, L4-L5, and L5-S1 (<xref ref-type="fig" rid="F3">Figure 3</xref>). YOLOv5x, YOLOv6, and YOLOv7 were trained to locate the LDH region on MRI images. This study was conducted using Windows 10 running Python 3.7.6 on a machine with the following specifications: Core (TM) i7-11700 CPU, 32&#xa0;GB RAM, and an NVIDIA GeForce RTX 3090 GPU with 24&#xa0;GB of GDDR6X memory. In this study, we trained the annotated dataset of YOLOv5, YOLOv6, and YOLOv7 using 16 batch sizes and 100 epochs.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>
<bold>(A)</bold> The lumbar vertebrae and disc sections are from L1-L2, L2-L3, L3-L4, L4-L5, and L5-S1; <bold>(B)</bold> The same lumbar MRI displays LDH detection at L4-L5 and L5&#x2013;S1. L1, first lumbar vertebra; L2, second lumbar vertebra; L3, third lumbar vertebra; L4, fourth lumbar vertebra; L5, fifth lumbar vertebra; S1, first sacral vertebra; LDH, lumbar disc herniation.</p>
</caption>
<graphic xlink:href="fbioe-11-1247112-g003.tif"/>
</fig>
</sec>
<sec id="s3-3">
<title>2.3 Images augmentation</title>
<p>The dataset&#x2019;s insufficient quantity of images can cause overfitting or underfitting and is one of many factors that affect deep learning performance. Earlier studies have used data augmentation to prevent and mitigate overfitting for deep learning (<xref ref-type="bibr" rid="B10">Ciregan et al., 2012</xref>; <xref ref-type="bibr" rid="B1">Abdelhafiz et al., 2019</xref>). During training, the augmentation addressed the issue of an excessively homogeneous dataset and improved the performance of deep learning models by simulating real-world situations.</p>
<p>In image processing, the augmentation type and range settings are used to increase the volume and features of the image (<xref ref-type="bibr" rid="B11">Dao, 2019</xref>; <xref ref-type="bibr" rid="B34">S&#xe1;nchez-Peralta et al., 2020</xref>). Our study selected the augmentation types in brightness, hue, exposure, and rotation to achieve the best results (<xref ref-type="bibr" rid="B19">Hussain, 2017</xref>; <xref ref-type="bibr" rid="B34">S&#xe1;nchez-Peralta et al., 2020</xref>). We augmented the images with Roboflow after we had finished labeling them. The brightness, hue, and exposure adjustments represent various magnetic resonance machines and room lighting. The image rotation feature simulates various patient positions during MRI capture. Configuration images for YOLOv5, YOLOv6, and YOLOv7 were set to the brightness 10% and &#x2212;10%, hue 10&#xb0; and &#x2212;10&#xb0;, exposure 10% and &#x2212;10%, and rotation 5&#xb0; and &#x2212;5&#xb0; (<xref ref-type="fig" rid="F4">Figure 4</xref>).</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Lumbar images augmentation: <bold>(A)</bold> Original image, <bold>(B)</bold> Brightness 10%, <bold>(C)</bold> Brightness &#x2212;10%, <bold>(D)</bold> Hue 10&#xb0;, <bold>(E)</bold> Hue &#x2212;10&#xb0;, <bold>(F)</bold> Exposure 10%, <bold>(G)</bold> Exposure &#x2212;10%, <bold>(H)</bold> rotation 5&#xb0;, and <bold>(I)</bold> rotation &#x2212;5&#xb0;.</p>
</caption>
<graphic xlink:href="fbioe-11-1247112-g004.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>2.4 Deep learning performance</title>
<p>In most cases, The YOLO algorithm predicts the bounding box of the training and validation of the YOLO results using the average precision (AP) and the mean average precision (mAP) parameters on the LDH regions from L1-L2, L2-L3, L3-L4, L4-L5, and L5-S1 (<xref ref-type="bibr" rid="B39">Tsai et al., 2021</xref>). AP can be used as a comprehensive evaluation index to balance the effects of Precision and Recall. And we use a simple average F1 score and the AP value as an additional measure to demonstrate how well the methods perform on a complete dataset (<xref ref-type="bibr" rid="B18">Haque and Neubert, 2020</xref>). We selected the following metrics to assess the algorithm&#x2019;s performance: Precision, Recall, AP, mAP, and F1 score. We then compared the Precision, Recall, AP, mAP, and F1 score for all LDH regions of YOLOv5x, YOLOv6, and YOLOv7 from L1-L2, L2-L3, L3-L4, L4-L5, and L5-S1, to determine which model was most suitable. A larger performance value indicates a more accurate model (<xref ref-type="bibr" rid="B28">Loram et al., 2020</xref>). It is also essential to use the mAP index to determine the network model&#x2019;s overall performance as well as to prevent extreme and weak functioning during the evaluation process. This study further calculates and validates the performance of YOLOv5x, YOLOv6, and YOLOv7 to detect LDH. The three models were evaluated using the Accuracy metric.</p>
<p>The formula of the Precision (Eq. <xref ref-type="disp-formula" rid="e1">1</xref>), Recall (Eq. <xref ref-type="disp-formula" rid="e2">2</xref>), AP (Eq. <xref ref-type="disp-formula" rid="e3">3</xref>), mAP (Eq. <xref ref-type="disp-formula" rid="e4">4</xref>), F1 score (Eq. <xref ref-type="disp-formula" rid="e5">5</xref>), Accuracy (Eq. <xref ref-type="disp-formula" rid="e6">6</xref>) are as follows:<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>y</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>N</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>Precision, as defined above, measures the percentage of correctly predicted disease pixels corresponding to the ground truth. Precision is an important performance measure since it is sensitive to over-segmentation, leading to low precision scores [29]. TP, True Positive; FP, False Positive.<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>y</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>N</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>G</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>Recall, as defined above, is an indicator of the proportion of correctly predicted disease pixels corresponding to the ground truth. It is susceptible to under-segmentation, resulting in low recall scores (<xref ref-type="bibr" rid="B18">Haque and Neubert, 2020</xref>). TP, True Positive; FN, False Negative.<disp-formula id="e3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>AP is the area under the precision-recall curve. It can be used as a comprehensive evaluation index to balance the effects of Precision and Recall. AP is calculated for each class separately (<xref ref-type="bibr" rid="B39">Tsai et al., 2021</xref>). AP; average precision; R, Recall; P, Precision; n, threshold number.<disp-formula id="e4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>The mAP is the average of AP over all detected classes and is used to evaluate the training, validate the results, and determine the overall model performance (<xref ref-type="bibr" rid="B39">Tsai et al., 2021</xref>). N, total number of the class; i, a score function to show an object similarity.<disp-formula id="e5">
<mml:math id="m5">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mfrac>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>F1 score, is a weighted average of Precision versus Recall. One point is added to Precision if the result is relevant, and one point is added to Recall if at least one result is relevant. This way, a model&#x2019;s performance can be measured effectively (<xref ref-type="bibr" rid="B39">Tsai et al., 2021</xref>).<disp-formula id="e6">
<mml:math id="m6">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>Accuracy, as defined above, is calculated by taking the number of correctly predicted samples out of all possible samples. TP, True Positive; TN, True Negative; FP, False Positive; FN, False Negative.</p>
</sec>
</sec>
<sec sec-type="results" id="s4">
<title>3 Results</title>
<p>This study examined the performance of YOLOv5x, YOLOv6, and YOLOv7 in non-augmented (non-AUG) and augmented (AUG) dataset, compared the overall performance and the detection of regions on L1-L2, L2-L3, L3-L4, L4-L5, and L5-S1, and observed the training durations of YOLOv5x, YOLOv6, and YOLOv7.</p>
<sec id="s4-1">
<title>3.1 Performance of the non-AUG and AUG dataset</title>
<p>The dataset includes 550 trained images without augmentation (550-non-AUG) and 550 trained images with 3 times augmentation (550-AUG). We compared the increased performance rates of the 550-non-AUG dataset to the 550-AUG dataset of all YOLO models by evaluating the increment rate of the mAP, which we calculated manually. The mAP is calculated based on Precision and Recall and thus can be used to determine the better model between the non-AUG and AUG YOLO models. There was a 0.34% increment rate between YOLOv5x and YOLOv5x-AUG, a 12.35% increment rate between YOLOv6 and YOLOv6-AUG, and a 1.51% increment rate between YOLOv7 and YOLOv7-AUG. While YOLOv6-AUG has the highest increment, its mAP performance remains slightly lower than YOLOv5x-AUG, with 2% less performance. Based on the metrics performance evaluation, the 550-AUG dataset outperforms the 550-non-AUG dataset on all YOLO models (<xref ref-type="table" rid="T1">Table 1</xref>).</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>YOLOv5x, YOLOv6, YOLOv7, metrics performance comparison in the 550-trained dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="3" align="left">Metrics performance comparison</th>
<th colspan="6" align="center">Model performance evaluation in the 550-trained dataset</th>
</tr>
<tr>
<th colspan="2" align="center">YOLOv5x</th>
<th colspan="2" align="center">YOLOv6</th>
<th colspan="2" align="center">YOLOv7</th>
</tr>
<tr>
<th align="right">Non-AUG (%)</th>
<th align="left">AUG (%)</th>
<th align="right">Non-AUG (%)</th>
<th align="left">AUG (%)</th>
<th align="right">Non-AUG (%)</th>
<th align="left">AUG (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Precision</td>
<td align="right">82.20</td>
<td align="left">75.00</td>
<td align="right">62.60</td>
<td align="left">90.70</td>
<td align="right">44.10</td>
<td align="left">46.30</td>
</tr>
<tr>
<td align="left">Recall</td>
<td align="right">84.40</td>
<td align="left">90.00</td>
<td align="right">86.00</td>
<td align="left">72.00</td>
<td align="right">75.80</td>
<td align="left">68.40</td>
</tr>
<tr>
<td align="left">F1 score</td>
<td align="right">83.29</td>
<td align="left">81.82</td>
<td align="right">72.50</td>
<td align="left">80.30</td>
<td align="right">55.76</td>
<td align="left">55.22</td>
</tr>
<tr>
<td align="left">mAP</td>
<td align="right">89.00</td>
<td align="left">89.30</td>
<td align="right">77.70</td>
<td align="left">87.30</td>
<td align="right">59.70</td>
<td align="left">60.60</td>
</tr>
<tr>
<td align="left">AP (L1-L2)</td>
<td align="right">69.30</td>
<td align="left">82.80</td>
<td align="right">65.10</td>
<td align="left">79.30</td>
<td align="right">37.40</td>
<td align="left">24.50</td>
</tr>
<tr>
<td align="left">AP (L2-L3)</td>
<td align="right">90.20</td>
<td align="left">84.10</td>
<td align="right">58.70</td>
<td align="left">73.60</td>
<td align="right">37.90</td>
<td align="left">50.00</td>
</tr>
<tr>
<td align="left">AP (L3-L4)</td>
<td align="right">93.60</td>
<td align="left">94.70</td>
<td align="right">84.70</td>
<td align="left">95.80</td>
<td align="right">49.20</td>
<td align="left">50.20</td>
</tr>
<tr>
<td align="left">AP (L4-L5)</td>
<td align="right">96.80</td>
<td align="left">97.30</td>
<td align="right">97.30</td>
<td align="left">97.80</td>
<td align="right">93.50</td>
<td align="left">93.00</td>
</tr>
<tr>
<td align="left">AP (L5-S1)</td>
<td align="right">94.90</td>
<td align="left">87.60</td>
<td align="right">82.80</td>
<td align="left">90.20</td>
<td align="right">80.30</td>
<td align="left">85.20</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Note: YOLO, you only look once; AUG, with augmentation; non-AUG, without augmentation; mAP, mean average precision; AP, average precision; L1, first lumbar vertebra; L2, second lumbar vertebra; L3, third lumbar vertebra; L4, fourth lumbar vertebra; L5, fifth lumbar vertebra; S1, first sacral vertebra.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4-2">
<title>3.2 Performance of deep learning model</title>
<p>When comparing the model performance in <xref ref-type="table" rid="T1">Table 1</xref>, YOLOv5x in 550-AUG dataset has the highest performance value amongst all the YOLO models tested. Hence, we further compared the performance of YOLO models to other deep learning models in 550-AUG. The results revealed that YOLOv5x showed the highest Recall, F1-score, and mAP compared to all YOLO models, with 90.00%, 81.82%, and 89.30%, respectively. YOLOv5x only fell short in the Precision compared to YOLOv6, with 75% for YOLOv5x and 90.70% for YOLOv6. We then examined the mAP difference to evaluate the overall model performance on all models. YOLOv5x showed the highest mAP at 89.30%, YOLOv6 showed a lower mAP at 87.30%, and YOLOv7 showed the lowest mAP at 60.60% (<xref ref-type="fig" rid="F5">Figure 5</xref>).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>The 550-non-AUG and 550-AUG metric performance comparison of Precision, Recall, F1 score, mAP using YOLOv5x, YOLOv6, and YOLOv7. YOLO, You Only Look Once; AUG, with augmentation; non-AUG, without augmentation; mAP, mean average precision.</p>
</caption>
<graphic xlink:href="fbioe-11-1247112-g005.tif"/>
</fig>
</sec>
<sec id="s4-3">
<title>3.3 Performance of region detection</title>
<p>We observed the different AP outcomes in each lumbar region on all the models in the non-AUG and AUG dataset. YOLOv5x non-AUG showed AP values above 90% in 4 regions, L2-L3, L3-L4, L4-L5, and L5-S1 at 90.20%, 93.60%, 96.80%, and 94.90%, respectively, except in L1-L2, which is 69.30%. YOLOv6 showed the highest AP amongst all YOLO models tested in L3-L4, L4-L5, and L5-S1 at 95.80%, 97.80%, and 90.20%, respectively. Amongst all YOLO models during training, YOLOv7 showed the lowest performance on all L1-S1 regions (<xref ref-type="fig" rid="F6">Figure 6</xref>).</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>The 550-non-AUG and 550-AUG metrics performance comparison of L1-L2, L2-L3, L3-L4, L4-L5, and L5-S1 using YOLOv5x, YOLOv6, and YOLOv7. YOLO, You Only Look Once; AUG, with augmentation; non-AUG, without augmentation; AP, average precision; L1, first lumbar vertebra; L2, second lumbar vertebra; L3, third lumbar vertebra; L4, fourth lumbar vertebra; L5, fifth lumbar vertebra; S1, first sacral vertebra.</p>
</caption>
<graphic xlink:href="fbioe-11-1247112-g006.tif"/>
</fig>
<p>Furthermore, we also calculated and validated the accuracy performance of YOLOv5x, YOLOv6, and YOLOv7 models. The YOLOv5x and YOLOv7 in non-AUG and AUG dataset showed accuracy above 70% in all L1-S1 lumbar regions. In comparison, YOLOv6 in the AUG dataset showed lower accuracy with a value of 69.77% in L3-L4, and in the non-AUG dataset showed the lowest accuracy with values of 69.77%, 67.24%, 63.60%, and 67.75% in L1-L2, L2-L3, L3-L4, and L4-L5 respectively (<xref ref-type="table" rid="T2">Table 2</xref>).</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Comparison of lumbar disc region detection accuracy.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Models</th>
<th rowspan="2" align="left">Dataset</th>
<th colspan="5" align="center">Lumbar disc region detection accuracy</th>
</tr>
<tr>
<th align="center">L1-L2 (%)</th>
<th align="center">L2-L3 (%)</th>
<th align="center">L3-L4 (%)</th>
<th align="center">L4-L5 (%)</th>
<th align="center">L5-S1 (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">YOLOv5x</td>
<td align="left">non-AUG</td>
<td align="center">77.00</td>
<td align="center">72.00</td>
<td align="center">70.00</td>
<td align="center">73.00</td>
<td align="center">72.00</td>
</tr>
<tr>
<td align="center">YOLOv5x</td>
<td align="left">AUG</td>
<td align="center">75.75</td>
<td align="center">70.69</td>
<td align="center">72.00</td>
<td align="center">72.81</td>
<td align="center">70.48</td>
</tr>
<tr>
<td align="center">YOLOv6</td>
<td align="left">non-AUG</td>
<td align="center">69.77</td>
<td align="center">67.24</td>
<td align="center">63.60</td>
<td align="center">67.75</td>
<td align="center">74.04</td>
</tr>
<tr>
<td align="center">YOLOv6</td>
<td align="left">AUG</td>
<td align="center">74.93</td>
<td align="center">70.77</td>
<td align="center">69.77</td>
<td align="center">72.63</td>
<td align="center">72.18</td>
</tr>
<tr>
<td align="center">YOLOv7</td>
<td align="left">non-AUG</td>
<td align="center">76.21</td>
<td align="center">75.92</td>
<td align="center">72.52</td>
<td align="center">71.61</td>
<td align="center">70.14</td>
</tr>
<tr>
<td align="center">YOLOv7</td>
<td align="left">AUG</td>
<td align="center">73.95</td>
<td align="center">71.05</td>
<td align="center">71.33</td>
<td align="center">70.88</td>
<td align="center">72.34</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Note: YOLO, you only look once; AUG, with augmentation; non-AUG, without augmentation; L1, first lumbar vertebra; L2, second lumbar vertebra; L3, third lumbar vertebra; L4, fourth lumbar vertebra; L5, fifth lumbar vertebra; S1, first sacral vertebra.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4-4">
<title>3.4 Performance in training duration and weight</title>
<p>We also observed the performance in training duration and the weight difference on all YOLO models according to non-AUG and AUG dataset. YOLOv5x non-AUG showed the shortest duration and the lightest folder weight after training, with 0.097&#xa0;h and 14.8&#xa0;MB, respectively. YOLOv5x AUG showed a slightly longer training duration and weight than YOLOv5x non-AUG by 3 times value, but the weight after training was similar. YOLOv6 non-AUG showed a slightly longer duration and heavier weight than YOLOv5x non-AUG with 0.140&#xa0;h and 264&#xa0;MB, respectively. However, YOLOv6 AUG showed a shorter duration than YOLOv5x AUG, although the weight is still far heavier than YOLOv5x AUG by 16 times value. And, YOLOv7 non-AUG and YOLOv7 AUG showed the longest duration and heaviest folder weight after training compared to all YOLOv5x and YOLOv6 models (<xref ref-type="table" rid="T3">Table 3</xref>).</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>The 550-non-AUG and 550-AUG training duration and weight using YOLOv5x, YOLOv6, and YOLOv7.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Models</th>
<th align="left">Dataset</th>
<th align="center">Epoch</th>
<th align="center">Batch size</th>
<th align="left">Total image</th>
<th align="center">Training duration (hours)</th>
<th align="right">Weight (MB)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">YOLOv5x</td>
<td align="left">non-AUG</td>
<td align="center">100</td>
<td align="center">16</td>
<td align="center">550</td>
<td align="center">0.097</td>
<td align="right">14.8</td>
</tr>
<tr>
<td align="left">YOLOv5x</td>
<td align="left">AUG</td>
<td align="center">100</td>
<td align="center">16</td>
<td align="center">1,650</td>
<td align="center">0.274</td>
<td align="right">14.8</td>
</tr>
<tr>
<td align="left">YOLOv6</td>
<td align="left">non-AUG</td>
<td align="center">100</td>
<td align="center">16</td>
<td align="center">550</td>
<td align="center">0.140</td>
<td align="right">264.0</td>
</tr>
<tr>
<td align="left">YOLOv6</td>
<td align="left">AUG</td>
<td align="center">100</td>
<td align="center">16</td>
<td align="center">1,650</td>
<td align="center">0.258</td>
<td align="right">239.0</td>
</tr>
<tr>
<td align="left">YOLOv7</td>
<td align="left">non-AUG</td>
<td align="center">100</td>
<td align="center">16</td>
<td align="center">550</td>
<td align="center">0.361</td>
<td align="right">5,290.0</td>
</tr>
<tr>
<td align="left">YOLOv7</td>
<td align="left">AUG</td>
<td align="center">100</td>
<td align="center">16</td>
<td align="center">1,650</td>
<td align="center">0.959</td>
<td align="right">5,290.0</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Note: YOLO, you only look once; AUG, with augmentation; non-AUG, without augmentation.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="s5">
<title>4 Discussion</title>
<p>We demonstrated the possibility of using the YOLOv5x, YOLOv6, and YOLOv7 to determine the LDH in MRI images based on bulging, protrusion, extrusion, and sequestration. In this study, YOLOv5x AUG was considered the most suitable model among the other YOLO models tested, showing the highest mAP performance. Moreover, our study showed that YOLOv5x non-AUG was an efficient LDH region predictor, as it had the highest AP scores in 4 regions, L2-L3, L3-L4, L4-L5, and L5-S1. YOLOv5x also showed the shortest training duration and the lightest weight compared to all YOLO models tested.</p>
<p>Based on the results from the current study, the YOLOv5x AUG had the best mAP performance among all YOLO models trained. YOLOv5x network architecture with PANet neck combines the 550-AUG images before they are sent for prediction, increasing the accuracy (<xref ref-type="bibr" rid="B21">Jocher, 2020</xref>). During training, YOLO models provide several parameters such as Precision, Recall, F1-Score, mAP, AP per lumbar disc region (L1-S1). The Precision and Recall are parameters that indicated the data sensitivity and predicted pixels accordingly, which are not enough to determine the overall performance of a deep learning model. The efficiency balance performance measured by F1-score and overall performance measured by mAP is calculated based on Precision and Recall to determine the best model among YOLOv5x, YOLOv6, and YOLOv7. From <xref ref-type="table" rid="T1">Table 1</xref>, YOLOv6-AUG has the highest detection region in L2-L3, L4-L5, and L5-S1. However, the overall performance of YOLOv6-AUG measured by mAP is still slightly lower than YOLOv5x-AUG. Thus based on the results of this study, YOLOv5x demonstrated a promising performance as a deep learning model for identifying LDH in MRI compared to all models trained. The greater mAP value observed with YOLOv5x proved the model is more accurate in predicting the presence of LDH in the MRI, indicating its potential for clinical use. Although YOLOv6 and YOLOv7 are the latest versions of YOLO, their effectiveness in biomedical applications is not substantially better. Our results are consistent with a study that found YOLOv6 and YOLOv7 to be less effective at detecting biomedical images than YOLOv5, possibly as a result of preliminary experiments, tune-ups, and revisions since these versions were published recently (<xref ref-type="bibr" rid="B7">Chen et al., 2022</xref>).</p>
<p>Our study found that the 550-AUG dataset had an improved performance compared to the 550 non-AUG dataset. However, other parameters in AUG dataset results, such as Precision, and F1 score of YOLOv5x, and Recall and F1 score of YOLOv7, were less accurate than the non-AUG dataset, as shown in <xref ref-type="table" rid="T1">Table 1</xref>. In this study, augmentation is performed using Roboflow, which randomly places the augmentation following the parameters we specify. However, a problem with random augmentation is that each image will have different combinations assigned, and not every augmentation will be applied to every image (<xref ref-type="bibr" rid="B20">Iwana and Uchida, 2021</xref>). It is possible for some images to be assigned a brightness and rotation augmentation, while others might be assigned a hue and exposure augmentation or a combination of these four augmentations. In general, the results of the augmentation were mixed. Several data augmentation methods have improved the accuracy, but some combinations with rotation methods might be detrimental (<xref ref-type="bibr" rid="B20">Iwana and Uchida, 2021</xref>). Based on our result in <xref ref-type="table" rid="T1">Table 1</xref>, we demonstrated the efficiency of training YOLOv5x model using a non-AUG dataset.</p>
<p>In the lumbar regions, our training result showed that YOLOv5x non-AUG had a more stable performance compared to all models, as it had detection performance above 90% in 4 regions, such as L2-L3, L3-L4, L4-L5, and L5-S1. In these regions, the disc herniation might cause radicular pain, which may compress the nerve root, resulting in pain and dysfunction symptoms (<xref ref-type="bibr" rid="B33">Reihani-Kermani, 2003</xref>; <xref ref-type="bibr" rid="B12">Fang et al., 2016</xref>). Deep learning&#x2019;s ability to recognize small objects, such as bulging, protrusion, extrusion, and sequestration in MRI lumbar, helps identify the LDH on different sizes and scales (<xref ref-type="bibr" rid="B27">Liu et al., 2020</xref>). During training, YOLOv5x non-AUG only demonstrated slightly low detection in L1-L2, which is not in the lower region where disc disease is commonly found (<xref ref-type="bibr" rid="B23">Katz et al., 2022</xref>). However, we validate the accuracy performance of YOLOv5x non-AUG and further confirm the potential to use the non-AUG dataset for LDH detection with all regions detection above 70%. Therefore, deep learning utilization may contribute to the early identification of spine abnormalities with greater accuracy in the four lower regions: L2-L3, L3-L4, L4-L5, and L5-S1, thereby helping clinicians determine the appropriate therapy in the earliest possible time and protecting patients from harmful consequences.</p>
<p>Deep learning models can be challenging and expensive to train, taking hours or weeks to complete (<xref ref-type="bibr" rid="B25">Lee et al., 2017</xref>). Our results showed that YOLOv5x had the shortest training duration and the lightest weight, showing that this model is most efficient in detecting LDH. To the best of our knowledge, this is the first study to compare the training duration of LDH detection across YOLOv5x, YOLOv6, and YOLOv7 models. A model&#x2019;s performance during training will determine how well it is able to perform when a user eventually uses it. Clinical settings could certainly benefit from using this as a computational reference in the future.</p>
<p>This study compared related studies to evaluate the LDH detection performance and total dataset used in MRI based on other models&#x2019; performances (<xref ref-type="table" rid="T4">Table 4</xref>). <xref ref-type="bibr" rid="B3">Alomari et al. (2014)</xref> showed a high accuracy detection using four classification stages and detection in a two-stage detector model to identify LDH). Wang et al., and Su et al., also showed a high accuracy detection using a two-stage detector model to identify LDH (<xref ref-type="bibr" rid="B42">Wang et al., 2020</xref>; <xref ref-type="bibr" rid="B35">Su et al., 2022</xref>). Compared with other studies, (<xref ref-type="bibr" rid="B45">Zhou et al., 2019</xref>) used 2,739 images; (<xref ref-type="bibr" rid="B35">Su et al., 2022</xref>) used 15,254 and 1,273 images; and (<xref ref-type="bibr" rid="B17">Guinebert et al., 2022</xref>) used 40 volumes of MRI (3,660 images) to train for deep learning. Their dataset included more than a thousand images. This study used 550 (non-AUG) and 1,650 (AUG) lumbar MRI images to train YOLOv5x, YOLOv6, and YOLOv7. Our previous study showed a higher accuracy with YOLOv3 than our current study, but it used twice as many slices to test the deep learning (<xref ref-type="bibr" rid="B39">Tsai et al., 2021</xref>). As our current dataset contains more patients, we will see a greater variation of LDH in MRI during the training process, which produces a reasonable result with a similar small-scale dataset. Compared to other studies, these results demonstrate the competitiveness of non-AUG dataset. In addition, we also used a single-stage detector, whereas some of the others used a two-stage detector model.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>The comparison of LDH detection performance with the related literature study using lumbar MRI with different deep learning algorithms.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">References</th>
<th align="center">Method</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1-score</th>
<th align="center">mAP</th>
<th align="center">Accuracy</th>
<th align="center">Total images</th>
<th align="center">No. of patients</th>
<th align="center">No. of slices</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B3">Alomari et al. (2014)</xref>
</td>
<td align="left">ASM &#x2b; GVF-Snake</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">93.90%</td>
<td align="right">390</td>
<td align="right">65</td>
<td align="right">6</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B45">Zhou et al. (2019)</xref>
</td>
<td align="left">Siamese Network</td>
<td align="center">98.90%</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">98.60%</td>
<td align="right">2,739</td>
<td align="right">2,739</td>
<td align="right">1</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B42">Wang et al. (2020)</xref>
</td>
<td align="left">DenseNet &#x2b; Inception-Resnet-V2</td>
<td align="center">96.65%</td>
<td align="center">95.13%</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">96.22%</td>
<td align="right">790</td>
<td align="right">395</td>
<td align="right">2</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B39">Tsai et al. (2021)</xref>
</td>
<td align="left">YOLOv3 (AUG)</td>
<td align="center">87.20%</td>
<td align="center">91.70%</td>
<td align="center">89.40%</td>
<td align="center">92.40%</td>
<td align="center">81.10%</td>
<td align="right">714</td>
<td align="right">65</td>
<td align="right">11</td>
</tr>
<tr>
<td rowspan="2" align="left">
<xref ref-type="bibr" rid="B35">Su et al. (2022)</xref>
</td>
<td align="left">ResNet-50 &#x2b; 3 FC networks (1)</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">84.17%</td>
<td align="right">15,254</td>
<td align="right">1,115</td>
<td align="right">&#x2014;</td>
</tr>
<tr>
<td align="left">ResNet-50 &#x2b; 3 FC networks (2)</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">74.20%</td>
<td align="right">1,273</td>
<td align="right">100</td>
<td align="right">&#x2014;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B17">Guinebert et al. (2022)</xref>
</td>
<td align="left">YOLOv5x</td>
<td align="center">75.00%</td>
<td align="center">76.50%</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="right">3,660</td>
<td align="right">244</td>
<td align="right">15</td>
</tr>
<tr>
<td rowspan="6" align="left">Our study</td>
<td align="left">YOLOv5x</td>
<td align="center">82.20%</td>
<td align="center">84.40%</td>
<td align="center">83.29%</td>
<td align="center">89.00%</td>
<td align="center">72.80%</td>
<td align="right">550</td>
<td align="right">110</td>
<td align="right">5</td>
</tr>
<tr>
<td align="left">YOLOv6</td>
<td align="center">62.60%</td>
<td align="center">86.00%</td>
<td align="center">72.50%</td>
<td align="center">77.70%</td>
<td align="center">68.48%</td>
<td align="right">550</td>
<td align="right">110</td>
<td align="right">5</td>
</tr>
<tr>
<td align="left">YOLOv7</td>
<td align="center">44.10%</td>
<td align="center">75.80%</td>
<td align="center">55.76%</td>
<td align="center">59.70%</td>
<td align="center">73.28%</td>
<td align="right">550</td>
<td align="right">110</td>
<td align="right">5</td>
</tr>
<tr>
<td align="left">YOLOv5x (AUG)</td>
<td align="center">75.00%</td>
<td align="center">90.00%</td>
<td align="center">81.82%</td>
<td align="center">89.30%</td>
<td align="center">72.35%</td>
<td align="right">1,650</td>
<td align="right">110</td>
<td align="right">5</td>
</tr>
<tr>
<td align="left">YOLOv6 (AUG)</td>
<td align="center">90.70%</td>
<td align="center">72.00%</td>
<td align="center">80.30%</td>
<td align="center">87.30%</td>
<td align="center">72.06%</td>
<td align="right">1,650</td>
<td align="right">110</td>
<td align="right">5</td>
</tr>
<tr>
<td align="left">YOLOv7 (AUG)</td>
<td align="center">46.30%</td>
<td align="center">68.40%</td>
<td align="center">55.22%</td>
<td align="center">60.60%</td>
<td align="center">71.91%</td>
<td align="right">1,650</td>
<td align="right">110</td>
<td align="right">5</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In recent years, clinical practice has significantly changed due to the use of deep learning models in diagnosis assistance, including the automatic detection of LDH (<xref ref-type="bibr" rid="B24">Lee and Yoon, 2021</xref>). In medical imaging, deep learning can provide efficient and accurate results and is regarded as one of the most promising methods for future application in the healthcare sector (<xref ref-type="bibr" rid="B32">Razzak et al., 2018</xref>). To prevent misdiagnosis, patients with borderline LDH visibility may require multiple MRI, which can be very costly (<xref ref-type="bibr" rid="B6">Bruno et al., 2015</xref>). However, even in the presence of blurriness or noise, deep learning models can be trained to be more robust at identifying LDH (<xref ref-type="bibr" rid="B17">Guinebert et al., 2022</xref>). Consequently, these deep learning models could provide clinicians with a helpful LDH prediction for more accurate diagnosis, thus reducing hospital visits and optimizing healthcare costs.</p>
<p>Our first limitation is the lack of AP value of the L1-L2 region during training using our most efficient model YOLOv5x in the non-AUG dataset. Such an issue might be because the LDH development in MRI may not be visible in these regions, or the protrusion may be too small to be noticed. In addition, the number of subjects with LDH in L1-L2 regions from the dataset was also limited. There were 226 LDH cases separated from L1 to S1 at 9, 25, 54, 88, and 50 (<xref ref-type="fig" rid="F7">Figure 7A</xref>). Based on the number of cases, our study was also consistent with the findings of other studies about LDH (<xref ref-type="bibr" rid="B13">Faur et al., 2019</xref>; <xref ref-type="bibr" rid="B39">Tsai et al., 2021</xref>). L1-L2 had the lowest frequency of LDH, while L4-L5 had the highest frequency of symptoms. As a whole, the AP value for L4-L5 was more distinguishable than the number of LDH for each of the lumbar vertebrae regions. The training results indicate that the number of symptoms affects deep learning in each LDH region (<xref ref-type="fig" rid="F7">Figure 7B</xref>). Insufficient image numbers in the small dataset result in limited detection due to deep learning underfitting and overfitting.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>
<bold>(A)</bold> LDH cases in different lumbar vertebrae regions. <bold>(B)</bold> LDH cases and average precision (AP) related trend in different lumbar vertebrae regions. L1, first lumbar vertebra; L2, second lumbar vertebra; L3, third lumbar vertebra; L4, fourth lumbar vertebra; L5, fifth lumbar vertebra; S1, first sacral vertebra.</p>
</caption>
<graphic xlink:href="fbioe-11-1247112-g007.tif"/>
</fig>
<p>Our second limitation is the LDH multi-labeling format on all YOLO models. In this study, we put the label only on the discs with bulges, protrusions, extrusions, and sequestrations as indicators of LDH in the MRI. Detailed labeling on each disc from L1-S1 to confirm the appearance of LDH might provide more specificity on the clinical diagnosis. However, putting confirmation labels on each disc in MRI might be challenging, as different regions will be affected by LDH. A multi-label classification typically requires additional effort in extracting and describing the associated label information to achieve satisfactory training results (<xref ref-type="bibr" rid="B31">Rastogi and Kumar, 2023</xref>). Furthermore, more labels require a larger dataset to avoid missing labels that could hinder the deep learning process. If incomplete labeled data is used for training, it may result in noisy classifiers with inadequate prediction capabilities (<xref ref-type="bibr" rid="B43">Wu et al., 2014</xref>).</p>
<p>As of today, YOLO is one of the fastest-growing and best algorithms available, with the current YOLOv8 algorithm being released in 2023. Our initial speculation by utilizing detection feature improvements in YOLOv8 may increase the accuracy of the LDH detection. Due to the speed, accuracy, and ease of use of YOLOv8, it is an excellent choice for several object detection, instance segmentation, and image classification applications (<xref ref-type="bibr" rid="B22">Ju and Cai, 2023</xref>). Despite this, <xref ref-type="bibr" rid="B22">Ju and Cai (2023)</xref> found that when they trained YOLOv8 on biomedical images, the accuracy range of the system was still 60%, which is still considered to be low in terms of detection accuracy. Because of these reasons, it is still uncertain whether YOLOv8 is capable of detecting LDH. It would be worth investigating the accuracy of YOLOv8 to detect the LDH accuracy in the L1-L2 region and to also implement a more detailed multi-labeling image of the LDH region in the future.</p>
<p>Our third limitation is using the MRI sagittal view, which lacks LDH visualization of the specific annular tear angle. A deep learning algorithm can identify the annular tears in which the nucleus pulposus protrudes and compresses the lumbar discs (<xref ref-type="bibr" rid="B35">Su et al., 2022</xref>). From our sagittal view results, we were able to detect the LDH development based on the protrusion on the MRI image. A sagittal view is also used by neuroradiologists for the initial examination of the three lowest intervertebral discs&#x2014;L3/L4, L4/L5, and L5/S1 (<xref ref-type="bibr" rid="B23">Katz et al., 2022</xref>). However, in most cases, neuroradiologists diagnose neural foraminal stenosis based on an axial view of the spine. Neural foraminal stenosis is important to indicate where an annular tear begins, which compresses the disc at a specific angle, either symmetrical or asymmetrical (<xref ref-type="bibr" rid="B4">Amin et al., 2017</xref>). Future studies might confirm the LDH development based on the annular tear on axial views. As such, we believe that deep learning detection does not replace medical personnel but provides fast access to additional information that accelerates initial diagnosis and helps focus on specific areas of concern. In this manner, the diagnosis may be made more quickly and with greater certainty before the team of multidisciplinary professionals decides on future surgery or treatment.</p>
</sec>
<sec sec-type="conclusion" id="s6">
<title>5 Conclusion</title>
<p>This study contributes to the automatic detection of LDH using deep learning and further identifies the best model for the YOLO series. Our study showed that YOLOv5x, YOLOv6, and YOLOv7 are promising deep learning methods for detecting LDH from MRI. In the present study, we observed that YOLOv5x AUG showed the highest overall performance based on mAP value. In addition, the YOLOv5x non-AUG showed stable LDH detection levels in L2-L3, L3-L4, L4-L5, and L5-S1 based on AP values which show competitiveness in using the non-AUG dataset for training. Further, YOLOv5x showed the most efficient training duration, which may prove useful in clinical settings where a computational application is required. Finally, this study demonstrated that YOLOv5x can detect LDH, and its application in biomedical imaging may be beneficial.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary Material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s8">
<title>Ethics statement</title>
<p>Ethical approval for using the dataset has been granted to the original team of researchers who own the dataset (<xref ref-type="bibr" rid="B2">Al-Kafri et al., 2019</xref>). All procedures have been conducted following the ethical standards of the United Kingdom and the Kingdom of Jordan and the Helsinki Declaration of 1964 and its amendments. Approval was granted by the Medical Ethical Committee of Irbid Speciality Hospital in Jordan.</p>
</sec>
<sec id="s9">
<title>Author contributions</title>
<p>Conceptualization: AP and C-WL; methodology: AP, PA, and J-YT; supervision: C-YL, F-YL, YG, and YP; investigation: C-YL, Y-KJ, YG, B-YL, F-YL, and C-WL; writing&#x2013;original draft: AP; writing&#x2013;review and editing: Y-KJ and C-WL; all authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s10">
<title>Funding</title>
<p>This study was supported by the Ministry of Science and Technology of the Republic of China (MOST111-2221-E-468-002 and MOST111-2923-E-155-004-MY3).</p>
</sec>
<ack>
<p>The authors wish to express gratitude to Mr. Elvin Nur Furqon, Mr. Fahni Haris, and Ms. Maftuhah Rahimah Rum for their assistance.</p>
</ack>
<sec sec-type="COI-statement" id="s11">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abdelhafiz</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ammar</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Nabavi</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Deep convolutional neural networks for mammography: advances, challenges and applications</article-title>. <source>BMC Bioinforma.</source> <volume>20</volume> (<issue>11</issue>), <fpage>281</fpage>&#x2013;<lpage>320</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-019-2823-4</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Kafri</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Sudirman</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hussain</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Al-Jumeily</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Natalia</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Meidia</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Boundary delineation of MRI images for lumbar spinal stenosis detection through semantic segmentation using deep neural networks</article-title>. <source>IEEE Access</source> <volume>7</volume>, <fpage>43487</fpage>&#x2013;<lpage>43501</lpage>. <pub-id pub-id-type="doi">10.1109/access.2019.2908002</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Alomari</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Corso</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Chaudhary</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Dhillon</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Lumbar spine disc herniation diagnosis with a joint shape model</article-title>,&#x201d; in <source>Computational methods and clinical applications for spine imaging</source> (<publisher-name>Springer</publisher-name>), <fpage>87</fpage>&#x2013;<lpage>98</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Amin</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Andrade</surname>
<given-names>N. S.</given-names>
</name>
<name>
<surname>Neuman</surname>
<given-names>B. J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Lumbar disc herniation</article-title>. <source>Curr. Rev. Musculoskelet. Med.</source> <volume>10</volume> (<issue>4</issue>), <fpage>507</fpage>&#x2013;<lpage>516</lpage>. <pub-id pub-id-type="doi">10.1007/s12178-017-9441-4</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Azimi</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Yazdanian</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Benzel</surname>
<given-names>E. C.</given-names>
</name>
<name>
<surname>Aghaei</surname>
<given-names>H. N.</given-names>
</name>
<name>
<surname>Azhari</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sadeghi</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>A review on the use of artificial intelligence in spinal diseases</article-title>. <source>Asian Spine J.</source> <volume>14</volume> (<issue>4</issue>), <fpage>543</fpage>&#x2013;<lpage>571</lpage>. <pub-id pub-id-type="doi">10.31616/asj.2020.0147</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bruno</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Walker</surname>
<given-names>E. A.</given-names>
</name>
<name>
<surname>Abujudeh</surname>
<given-names>H. H.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Understanding and confronting our mistakes: the epidemiology of error in radiology and strategies for error reduction</article-title>. <source>Radiographics</source> <volume>35</volume> (<issue>6</issue>), <fpage>1668</fpage>&#x2013;<lpage>1676</lpage>. <pub-id pub-id-type="doi">10.1148/rg.2015150023</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Shalaginov</surname>
<given-names>M. Y.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>T. H.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Real-time detection of acute lymphoblastic leukemia cells using deep learning</article-title>,&#x201d; in <conf-name>2022 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</conf-name>, <conf-loc>Las Vegas, NV, USA</conf-loc>, <conf-date>6-8 Dec. 2022</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>3788</fpage>&#x2013;<lpage>3790</lpage>. <pub-id pub-id-type="doi">10.1109/BIBM55620.2022.9995131</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>K.-T.</given-names>
</name>
<name>
<surname>Tseng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>L. W.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C. M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Technical considerations of interlaminar approach for lumbar disc herniation</article-title>. <source>World Neurosurg.</source> <volume>145</volume>, <fpage>612</fpage>&#x2013;<lpage>620</lpage>. <pub-id pub-id-type="doi">10.1016/j.wneu.2020.06.211</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Choi</surname>
<given-names>K.-C.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>C. K.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Percutaneous endoscopic lumbar discectomy: minimally invasive technique for multiple episodes of lumbar disc herniation</article-title>. <source>BMC Musculoskelet. Disord.</source> <volume>18</volume> (<issue>1</issue>), <fpage>329</fpage>&#x2013;<lpage>336</lpage>. <pub-id pub-id-type="doi">10.1186/s12891-017-1697-8</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ciregan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Meier</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Schmidhuber</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>Multi-column deep neural networks for image classification</article-title>,&#x201d; in <conf-name>2012 IEEE conference on computer vision and pattern recognition</conf-name>, <conf-loc>Hadhramout, Yemen</conf-loc>, <conf-date>15-16 Dec. 2019</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1109/ICOICE48418.2019.9035162</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Dao</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>A kernel theory of modern data augmentation</article-title>,&#x201d; in <source>International conference on machine learning</source> (<publisher-name>PMLR</publisher-name>).</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Which level is responsible for gluteal pain in lumbar disc hernia?</article-title> <source>BMC Musculoskelet. Disord.</source> <volume>17</volume> (<issue>1</issue>), <fpage>356</fpage>&#x2013;<lpage>364</lpage>. <pub-id pub-id-type="doi">10.1186/s12891-016-1204-7</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Faur</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Patrascu</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Haragus</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Anglitoiu</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Correlation between multifidus fatty atrophy and lumbar disc degeneration in low back pain</article-title>. <source>BMC Musculoskelet. Disord.</source> <volume>20</volume> (<issue>1</issue>), <fpage>414</fpage>&#x2013;<lpage>416</lpage>. <pub-id pub-id-type="doi">10.1186/s12891-019-2786-7</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Friska</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Sudirman</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <source>Classification of sagittal lumbar spine MRI for lumbar spinal stenosis detection using transfer learning of a deep convolutional neural network</source>. <publisher-name>IEEE Explore</publisher-name>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ghosh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chaudhary</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Supervised methods for detection and segmentation of tissues in clinical lumbar MRI</article-title>. <source>Comput. Med. Imaging Graph.</source> <volume>38</volume> (<issue>7</issue>), <fpage>639</fpage>&#x2013;<lpage>649</lpage>. <pub-id pub-id-type="doi">10.1016/j.compmedimag.2014.03.005</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gopalakrishnan</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Nadhamuni</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Karthikeyan</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Categorization of pathology causing low back pain using magnetic resonance imaging (MRI)</article-title>. <source>J. Clin. Diagnostic Res. JCDR</source> <volume>9</volume> (<issue>1</issue>), <fpage>TC17</fpage>&#x2013;<lpage>TC20</lpage>. <pub-id pub-id-type="doi">10.7860/JCDR/2015/10951.5470</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guinebert</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Petit</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Bousson</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Bodard</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Amoretti</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kastler</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Automatic semantic segmentation and detection of vertebras and intervertebral discs by neural networks</article-title>. <source>Comput. Methods Programs Biomed. Update</source> <volume>2</volume>, <fpage>100055</fpage>. <pub-id pub-id-type="doi">10.1016/j.cmpbup.2022.100055</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haque</surname>
<given-names>I. R. I.</given-names>
</name>
<name>
<surname>Neubert</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deep learning approaches to biomedical image segmentation</article-title>. <source>Inf. Med. Unlocked</source> <volume>18</volume>, <fpage>100297</fpage>. <pub-id pub-id-type="doi">10.1016/j.imu.2020.100297</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hussain</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Differential data augmentation techniques for medical imaging classification tasks</article-title>,&#x201d; in <source>AMIA annual symposium proceedings</source> (<publisher-name>American Medical Informatics Association</publisher-name>).</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Iwana</surname>
<given-names>B. K.</given-names>
</name>
<name>
<surname>Uchida</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>An empirical survey of data augmentation for time series classification with neural networks</article-title>. <source>Plos one</source> <volume>16</volume> (<issue>7</issue>), <fpage>e0254841</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0254841</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Jocher</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). <source>yolov5</source>. <publisher-name>Code repository</publisher-name>.</citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ju</surname>
<given-names>R.-Y.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Fracture detection in pediatric wrist trauma X-ray images using YOLOv8 algorithm</source>. <comment>arXiv preprint arXiv:2304.05071</comment>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Katz</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Zimmerman</surname>
<given-names>Z. E.</given-names>
</name>
<name>
<surname>Mass</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Makhni</surname>
<given-names>M. C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Diagnosis and management of lumbar spinal stenosis: A review</article-title>. <source>Jama</source> <volume>327</volume> (<issue>17</issue>), <fpage>1688</fpage>&#x2013;<lpage>1699</lpage>. <pub-id pub-id-type="doi">10.1001/jama.2022.5921</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yoon</surname>
<given-names>S. N.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Application of artificial intelligence-based technologies in the healthcare industry: opportunities and challenges</article-title>. <source>Int. J. Environ. Res. Public Health</source> <volume>18</volume> (<issue>1</issue>), <fpage>271</fpage>. <pub-id pub-id-type="doi">10.3390/ijerph18010271</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>J.-G.</given-names>
</name>
<name>
<surname>Jun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cho</surname>
<given-names>Y. W.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>G. B.</given-names>
</name>
<name>
<surname>Seo</surname>
<given-names>J. B.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Deep learning in medical imaging: general overview</article-title>. <source>kjr</source> <volume>18</volume> (<issue>4</issue>), <fpage>570</fpage>&#x2013;<lpage>584</lpage>. <pub-id pub-id-type="doi">10.3348/kjr.2017.18.4.570</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Weng</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Geng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <source>YOLOv6: A single-stage object detection framework for industrial applications</source>. <comment>arXiv preprint arXiv:2209.02976</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.2209.02976</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>T. W.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>C. Y.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>W. H.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>GODoc: high-throughput protein function prediction using novel k-nearest-neighbor and voting algorithms</article-title>. <source>World Sci. Res. J.</source> <volume>6</volume> (<issue>11</issue>), <fpage>276</fpage>&#x2013;<lpage>284</lpage>. <pub-id pub-id-type="doi">10.1186/s12859-020-03556-9</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Loram</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Siddique</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sanchez</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Harding</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Silverdale</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kobylecki</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Objective analysis of neck muscle boundaries for cervical dystonia using ultrasound imaging and deep learning</article-title>. <source>IEEE J. Biomed. health Inf.</source> <volume>24</volume> (<issue>4</issue>), <fpage>1016</fpage>&#x2013;<lpage>1027</lpage>. <pub-id pub-id-type="doi">10.1109/jbhi.2020.2964098</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Martin</surname>
<given-names>B. I.</given-names>
</name>
<name>
<surname>Mirza</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Spina</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Spiker</surname>
<given-names>W. R.</given-names>
</name>
<name>
<surname>Lawrence</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Brodke</surname>
<given-names>D. S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Trends in lumbar fusion procedure rates and associated hospital costs for degenerative spinal diseases in the United States, 2004 to 2015</article-title>. <source>Spine</source> <volume>44</volume> (<issue>5</issue>), <fpage>369</fpage>&#x2013;<lpage>376</lpage>. <pub-id pub-id-type="doi">10.1097/brs.0000000000002822</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mushtaq</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Akram</surname>
<given-names>M. U.</given-names>
</name>
<name>
<surname>Alghamdi</surname>
<given-names>N. S.</given-names>
</name>
<name>
<surname>Fatima</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Masood</surname>
<given-names>R. F.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Localization and edge-based segmentation of lumbar spine vertebrae to identify the deformities using deep learning models</article-title>. <source>Sensors</source> <volume>22</volume> (<issue>4</issue>), <fpage>1547</fpage>. <pub-id pub-id-type="doi">10.3390/s22041547</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rastogi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Discriminatory label-specific weights for multi-label learning with missing labels</article-title>. <source>Neural Process. Lett.</source> <volume>55</volume> (<issue>2</issue>), <fpage>1397</fpage>&#x2013;<lpage>1431</lpage>. <pub-id pub-id-type="doi">10.1007/s11063-022-10945-z</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Razzak</surname>
<given-names>M. I.</given-names>
</name>
<name>
<surname>Naz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zaib</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Deep learning for medical image processing: overview, challenges and the future</article-title>,&#x201d; in <source>Classification in BioApps. Lecture notes in computational vision and Biomechanics</source> (<publisher-loc>Springer, Cham</publisher-loc>, <volume>26</volume>, <fpage>323</fpage>&#x2013;<lpage>350</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-65981-7_12</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Reihani-Kermani</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2003</year>). <source>Level-diagnosis of lumbar disc herniation</source>.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>S&#xe1;nchez-Peralta</surname>
<given-names>L. F.</given-names>
</name>
<name>
<surname>Pic&#xf3;n</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>S&#xe1;nchez-Margallo</surname>
<given-names>F. M.</given-names>
</name>
<name>
<surname>Pagador</surname>
<given-names>J. B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Unravelling the effect of data augmentation transformations in polyp segmentation</article-title>. <source>Int. J. Comput. assisted radiology Surg.</source> <volume>15</volume> (<issue>12</issue>), <fpage>1975</fpage>&#x2013;<lpage>1988</lpage>. <pub-id pub-id-type="doi">10.1007/s11548-020-02262-4</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname>
<given-names>Z. H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z. Y.</given-names>
</name>
<name>
<surname>You</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Automatic grading of disc herniation, central canal stenosis and nerve roots compression in lumbar magnetic resonance image diagnosis</article-title>. <source>Front. Endocrinol. (Lausanne)</source> <volume>13</volume>, <fpage>890371</fpage>. <pub-id pub-id-type="doi">10.3389/fendo.2022.890371</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sudirman</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kafri</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Natalia</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Meidia</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Afriliana</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Al-Rashdan</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2019a</year>). <source>Lumbar spine MRI dataset</source>. <publisher-name>Mendeley Data</publisher-name>.</citation>
</ref>
<ref id="B37">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sudirman</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kafri</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Natalia</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Meidia</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Afriliana</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Al-Rashdan</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2019b</year>). <source>MATLAB source code for developing ground truth dataset, semantic segmentation, and evaluation for the lumbar spine MRI dataset</source>. <publisher-name>Mendeley Data</publisher-name>.</citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Terven</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cordova-Esparza</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). <source>A comprehensive review of YOLO: From YOLOv1 to YOLOv8 and beyond</source>. <comment>arXiv preprint arXiv:2304.00501</comment>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tsai</surname>
<given-names>J.-Y.</given-names>
</name>
<name>
<surname>Hung</surname>
<given-names>I. Y. J.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Y. L.</given-names>
</name>
<name>
<surname>Jan</surname>
<given-names>Y. K.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>C. Y.</given-names>
</name>
<name>
<surname>Shih</surname>
<given-names>T. T. F.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Lumbar disc herniation automatic detection in magnetic resonance imaging based on deep learning</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>9</volume>, <fpage>691</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2021.708137</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vialle</surname>
<given-names>L. R.</given-names>
</name>
<name>
<surname>Vialle</surname>
<given-names>E. N.</given-names>
</name>
<name>
<surname>Su&#xe1;rez Henao</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Giraldo</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Lumbar disc herniation</article-title>. <source>Rev. Bras. Ortop. (English Ed.</source> <volume>45</volume> (<issue>1</issue>), <fpage>17</fpage>&#x2013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1016/s2255-4971(15)30211-1</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>C.-Y.</given-names>
</name>
<name>
<surname>Bochkovskiy</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>H.-Y. M.</given-names>
</name>
</person-group> (<year>2022</year>). <source>YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors</source>. <comment>arXiv preprint arXiv:2207.02696</comment>.</citation>
</ref>
<ref id="B42">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Automatic diagnosis of disc herniation based on DenseNet fusion model</article-title>,&#x201d; in <source>2020 8th international conference on digital home (ICDH)</source> (<publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B43">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>B. G.</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Multi-label learning with missing labels</article-title>,&#x201d; in <conf-name>2014 22nd International Conference on Pattern Recognition</conf-name>, <conf-loc>Stockholm, Sweden</conf-loc>, <conf-date>24-28 Aug. 2014</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>1964</fpage>&#x2013;<lpage>1968</lpage>. <pub-id pub-id-type="doi">10.1109/ICPR.2014.343</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yung</surname>
<given-names>N. D. T.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>W. K.</given-names>
</name>
<name>
<surname>Juwono</surname>
<given-names>F. H.</given-names>
</name>
<name>
<surname>Sim</surname>
<given-names>Z. A.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Safety helmet detection using deep learning: implementation and comparative study using YOLOv5, YOLOv6, and YOLOv7</article-title>,&#x201d; in <conf-name>2022 International Conference on Green Energy, Computing and Sustainable Technology (GECOST)</conf-name>, <conf-loc>Miri Sarawak, Malaysia</conf-loc>, <conf-date>26-28 Oct. 2022</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>164</fpage>&#x2013;<lpage>170</lpage>. <pub-id pub-id-type="doi">10.1109/GECOST55694.2022.10010490</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Sui</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Automatic lumbar MRI detection and identification based on deep learning</article-title>. <source>J. digital imaging</source> <volume>32</volume>, <fpage>513</fpage>&#x2013;<lpage>520</lpage>. <pub-id pub-id-type="doi">10.1007/s10278-018-0130-7</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>