<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article article-type="research-article" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1655091</article-id><article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading"><subject>Original Research</subject></subj-group>
</article-categories>
<title-group>
<article-title>Crack detection in structural images using a hybrid Swin Transformer and enhanced features representation block</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes"><name><surname>Anusha</surname> <given-names>N.</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3114926"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
</contrib>
<contrib contrib-type="author"><name><surname>Anbarasi</surname> <given-names>L. Jani</given-names></name><xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2640915"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>Department of IoT, School of Computer Science and Engineering, Vellore Institute of Technology</institution>, <city>Vellore</city>, <country country="in">India</country></aff>
<aff id="aff2"><label>2</label><institution>School of Computer Science and Engineering, Vellore Institute of Technology</institution>, <city>Chennai</city>, <country country="in">India</country></aff>
<author-notes><corresp id="c001"><label>&#x002A;</label>Correspondence: N. Anusha, <email xlink:href="mailto:anusha.n@vit.ac.in">anusha.n@vit.ac.in</email></corresp></author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-12-01">
<day>01</day>
<month>12</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1655091</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>23</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Anusha and Anbarasi.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Anusha and Anbarasi</copyright-holder>
<license><ali:license_ref start_date="2025-12-01">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>This paper presents a crack detection framework employing a hybrid model that integrates the Swin Transformer with an Enhanced Features Representation Block (EFRB) to precisely detect cracks in images.</p>
</sec>
<sec>
<title>Methods</title>
<p>The Swin Transformer captures long-range dependencies and efficiently processes complex images, forming the backbone of the feature extraction process. The EFRB improved spatial granularity through depthwise convolutions, that focus on spatial features independently across each channel, and pointwise convolutions to improve channel representation. The proposed model used residual connections to enable deeper networks to overcome vanishing gradient problem.</p>
</sec>
<sec>
<title>Results and discussion</title>
<p>The training process is optimized using population-based feature selection, resulting in robust performance. The network is trained on a dataset split into 80% training and 20% testing, with a learning rate of 1e-3, batch size of 16, and 30 epochs. Evaluation results show that the model achieves an accuracy of 98%, with precision, recall, and F1-scores as 0.97, 0.99, and 0.98 for crack detection, respectively. These results show the effectiveness of the proposed architecture for real-world crack detection applications in structural monitoring.</p>
</sec>
</abstract>
<kwd-group>
<kwd>swin transformer</kwd>
<kwd>crack detection</kwd>
<kwd>convolutional neural network</kwd>
<kwd>residual network</kwd>
<kwd>population-based optimization</kwd>
</kwd-group><funding-group><funding-statement>The author(s) declare that no financial support was received for the research and/or publication of this article.</funding-statement></funding-group>
<counts>
<fig-count count="11"/>
<table-count count="4"/>
<equation-count count="20"/>
<ref-count count="45"/>
<page-count count="15"/>
<word-count count="9275"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Machine Learning and Artificial Intelligence</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>Machine vision technology has seen significant advancement in the field of road crack detection (<xref ref-type="bibr" rid="ref37">Yin et al., 2023</xref>). Image and video analysis demonstrate a remarkable capacity for detecting and identifying early signs of road cracks. They play a crucial role in monitoring and early warning, helping to prevent potential crack-related incidents and safeguard lives and property (<xref ref-type="bibr" rid="ref20">Ma and Mei, 2021</xref>). Traditional pavement crack detection methods include techniques like minimum-path algorithms, image thresholding, and wavelet transformations. To enhance accuracy, few methods integrate free-form anisotropy and morphological filters, to attain a clearer depiction of crack intensity and features. Also, collaborative crack detection techniques have employed Sobel edge detectors with two-dimensional empirical mode decomposition, improving the precision of surface crack differentiation. Convolutional Neural Networks (CNN) are used for analyzing pavement crack images and surface characteristics to effectively enhance crack identification accuracy.</p>
<p>Cracks in road surfaces are common during road construction, initially arising from material aging and degradation over time and also due to climatic factors like precipitation and snow. These elements result various types of surface cracks, which gradually expand and lead to both surface and structural deterioration, compromising road safety and durability. In recent decades, various researchers and experts have proposed multiple techniques for detecting cracks in road surfaces, including physical inspection, machine vision, and infrared imaging. Physical inspections are intensive, inefficient, and prone to human errors (<xref ref-type="bibr" rid="ref44">Zhao et al., 2025</xref>). Machine vision techniques allows automated detection but needs high-quality image data and algorithms. Infrared imaging analyse the temperature on road surfaces that may indicate cracks but involves substantial equipment costs (<xref ref-type="bibr" rid="ref26">Oloufa et al., 2004</xref>; <xref ref-type="bibr" rid="ref17">Liu et al., 2024</xref>).</p>
<p>Recently, deep learning has significant importance in image processing, leading to the extensive use for road surface crack detection due to their high efficacy. Identifying and detecting structural surface issues, particularly cracks, can offer consistent data for the maintenance of buildings. Traditional crack detection methods, often produce subjective results (<xref ref-type="bibr" rid="ref45">Zhao et al., 2022</xref>) and lack a standardized global framework, reducing accuracy. Advances in computer vision have resulted in crack detection algorithms that offer automation, efficiency, and non-contact capabilities, effectively addressing the limitations of manual methods. In particular, the rapid progress of deep learning technology in recent years has enabled CNN models to significantly enhance detection accuracy and efficiency. Currently, CNN based detection methods are applied to identify surface damage in buildings, bridges, and tunnels. The image classification schemes identify the category of the input image, the object detection model estimates object locations within the image, and the semantic segmentation model performs pixel-level analysis to pinpoint objects. While semantic segmentation provides the highest accuracy, it requires pixel based labelled data for training, which is difficult and limits the CNN models in the field of structural crack detection. The main contribution of the proposed work is as follows:</p>
<list list-type="bullet">
<list-item>
<p>A Hybrid framework integrating the Swin Transformer with an Enhanced Features Representation Block (EFRB) to capture both global dependencies and fine-grained spatial features for accurate crack detection.</p>
</list-item>
<list-item>
<p>Depthwise and pointwise convolutions in the EFRB enhanced spatial granularity and channel-wise representation while residual connections and normalization stabilized the training and prevent overfitting.</p>
</list-item>
<list-item>
<p>Population-based feature selection with adaptive optimization further refines discriminative features, reducing computational complexity thus improving model generalization.</p>
</list-item>
<list-item>
<p>This approach outperforms, achieving 98% accuracy with better precision, recall, and F1-scores.</p>
</list-item>
</list>
<p>The paper is organized as follows: Section 2 details the review of existing crack detection methodologies. Section 3 presents the proposed hybrid Swin Transformer with Enhanced Features Representation Block (EFRB) and population-based feature optimization approach. Section 4 describes the dataset, evaluation metrics, ablation studies, and performance analysis of the proposed model. Section 5 concludes the paper.</p>
</sec>
<sec id="sec2">
<label>2</label>
<title>Related work</title>
<p><xref ref-type="bibr" rid="ref23">Nguyen et al. (2021)</xref> proposed a two-stage CNN where the first stage reduces noise and isolates potential cracks, while the second stage focuses on learning contextual features of cracks within the identified areas. The DeepCrack dataset, used for detection and segmentation validation, includes 537 images of 544&#x202F;&#x00D7;&#x202F;384 pixels with pixel-level ground truth annotations (<xref ref-type="bibr" rid="ref19">Liu et al., 2019</xref>). Another dataset, CrackIT (<xref ref-type="bibr" rid="ref25">Oliveira and Correia, 2014</xref>), was compiled in Portugal and Canada for crack analysis. <xref ref-type="bibr" rid="ref6">Fan et al. (2020)</xref> introduced an automated crack detection system using a U-Hierarchical Dilated Network (U-HDN). This model uses hierarchical feature learning and dilated convolution for detailed crack detection on road pavements. By integrating multiple context sizes through a multi-dilation module, the U-HDN model improves its capability to capture complex crack patterns at various scales. Tests on public crack datasets show that U-HDN outperforms existing methods by effectively combining diverse context sizes and multi-scale feature maps, leading to an increased detection accuracy of 0.93. <xref ref-type="bibr" rid="ref15">Liu et al. (2022a)</xref> leveraged deep learning, specifically CNNs, and infrared thermography to categorize asphalt pavement crack into four categories: no crack, low, medium, and high severity. Results showed fusion images resulted the good accuracy for models built from scratch, while visible images performed best in transfer learning, with EfficientNet-B3 achieving the highest accuracy across all categories for both methods.</p>
<p><xref ref-type="bibr" rid="ref5">Elghaish et al. (2022)</xref> evaluated models like AlexNet, GoogleNet, and two others for highway crack identification and classification, and introduced a new CNN model optimized for accuracy across diverse learning rates. The novel CNN model achieved 97.62% accuracy using a dataset of 4,663 crack images grouped into three categories, outperforming GoogleNet&#x2019;s 89.08% and AlexNet&#x2019;s 87.82%, utilizing Adam optimization at a learning rate of 0.001 for efficient highway crack recognition. <xref ref-type="bibr" rid="ref1">Ahmadi et al. (2022)</xref> proposed a approach combining segmentation, noise reduction, heuristic-based feature extraction, and the Hough transform with crack classification using six classifiers. The hybrid model achieved the highest accuracy at 93.86%, surpassing individual classifiers. <xref ref-type="bibr" rid="ref16">Liu et al. (2022b)</xref> employed infrared thermography and CNNs to classify asphalt pavement fatigue crack severity into four levels, using three image types. CNN models, including EfficientNet-B4, were trained, with accuracy surpassing 0.95 across all image types, particularly on infrared images. Grad-CAM and Guided Grad-CAM analyses indicated fusion images are highly effective for reliable fatigue crack classification.</p>
<p><xref ref-type="bibr" rid="ref25">Oliveira and Correia (2014)</xref> developed a comprehensive MATLAB toolbox for crack detection and characterisation on road pavement surfaces, which includes algorithms for preprocessing, crack identification, and classification. The toolbox includes 84 pavement surface images obtained from standard road surveys, providing a valuable resource for evaluating crack detection algorithms. <xref ref-type="bibr" rid="ref36">Yamaguchi and Hashimoto (2010)</xref> presented a rapid crack identification method for concrete surfaces using percolation-based image processing, which reduces computational time by incorporating skip processes and assessing pixel circularity. Experimental results show reduced computation costs while maintaining high crack detection accuracy. <xref ref-type="bibr" rid="ref34">Vivekananthan et al. (2023)</xref> developed a grey intensity adjustment model for crack detection, employing grey level discrimination and the Otsu method to set threshold ranges and Sobel&#x2019;s filter for edge detection. This approach achieved a maximum detection accuracy of 95% while addressing constraints related to aspect ratio and margin parameters. <xref ref-type="bibr" rid="ref13">Liu et al. (2023)</xref> introduced a tunnel crack detection method using image processing with deep learning, comparing SVM and AlexNet based models. AlexNet achieved 96.7% test accuracy, indicating deep CNN models&#x2019; superior performance for identifying structural flaws in subway tunnels. <xref ref-type="bibr" rid="ref21">Malek et al. (2023)</xref> created a real-time augmented reality (AR) crack detection system, overcoming traditional AR limitations by adapting the Canny algorithm to the AR headset platform for autonomous processing. Experimental results confirm this AR method&#x2019;s efficiency and practicality for real-time crack detection in field inspections. <xref ref-type="bibr" rid="ref33">Tran et al. (2023)</xref> presented a process based deep learning approach for bridge deck crack detection and segmentation, testing five object detection networks including YOLOv7 and achieved a detection accuracy of 92.38%. The proposed U-Net also exhibited enhanced performance, successfully identifying and quantifying cracks on bridge decks.</p>
<p><xref ref-type="bibr" rid="ref29">Pham et al. (2023)</xref> employed U-Net, LinkNet, FPN, and Deeplabv3, achieving F1 scores between 0.877 and 0.896 at 7.48&#x2013;8.01 frames per second (FPS), notably outperforming traditional image processing methods in speed and accuracy. <xref ref-type="bibr" rid="ref24">Nyathi et al. (2023)</xref> developed an approach for measuring concrete crack, achieving high precision with absolute error between 0.02&#x202F;mm and 0.57&#x202F;mm, facilitating compliance with international standards. <xref ref-type="bibr" rid="ref42">Zhang et al. (2023b)</xref> presented a lightweight crack detection technique for bridges using YOLOv4, incorporating lighter networks to reduce computational demands for edge devices. The modified YOLO v4 achieved 93.96% precision, 90.12% recall, and F1 score of 92%, requiring only 23.4&#x202F;MB and running at 140.2 FPS. <xref ref-type="bibr" rid="ref35">Xu et al. (2023)</xref> introduced the YOLOv5-IDS model, integrating the YOLOv5 architecture with a bilateral segmentation network for concrete crack detection and measurement, achieving an mAP@0.5 of 84.33% and an mIoU of 94.78%, with rapid detection at 159 FPS.</p>
<p><xref ref-type="bibr" rid="ref10">Hu et al. (2024)</xref> proposed an advanced approach to road surface crack detection using an enhanced YOLOv5 model, addressing the complexities of information extraction from vehicle-mounted imagery. Key improvements include the Slim-Neck architecture for targeted crack focus, the C2f structure and Decoupled Head for optimized data utilization, and a split SPPCSPC structure for enhanced efficiency and precision. Experimental results demonstrate significant improvements across multiple evaluation metrics compared to five other sophisticated models, affirming the effectiveness of the proposed approach. <xref ref-type="bibr" rid="ref3">Dong et al. (2024)</xref> introduced YOLOv8-Crack Detection (YOLOv8-CD), a lightweight, optimized algorithm for concrete crack detection aimed at boosting infrastructure safety and maintenance efficiency. The model leverages visual attention networks and a Large Separable Kernel Attention module to enhance crack shape detection and feature extraction. Experimental findings reveal substantial gains in mAP scores and detection speed, achieving 88 FPS while reducing processing demands, thereby validating its advantage over other object detection techniques. <xref ref-type="bibr" rid="ref2">Chen et al. (2023)</xref> explored the role of deep learning, specifically transfer learning, in automating the detection of building cracks. Addressing the need for efficient large-scale inspections, transfer learning significantly improved CNN performance, boosting accuracy from 89 to 94%, demonstrating its efficacy in image classification with limited data, aligning with national smart nation goals for intelligent technology in construction.</p>
<p><xref ref-type="bibr" rid="ref40">Zhang et al. (2023a)</xref> present a lightweight learning model for concrete crack detection, named MobileNetV3-BLS, which overcomes the challenges of complex architectures and high computational requirements. This method improves feature extraction by integrating MobileNetV3&#x2019;s inverted residual structure as a convolutional module, employing random mapping and enhancement nodes to train the model. MobileNetV3-BLS exhibits enhanced accuracy and training speed, facilitating dynamic updates for incremental learning with new data and nodes. <xref ref-type="bibr" rid="ref39">Zadeh et al. (2024)</xref> evaluate multiple deep learning architectures: InceptionV3, VGG19, ResNet50, and EfficientNetV2 using fine-tuning for concrete crack detection. Results show EfficientNetV2 achieves 99.6% accuracy, 99.3% precision, and a recall of 1, leading to a balanced F1 score of 99.6%, effectively minimizing false positives and maximizing true crack identification.</p>
<p><xref ref-type="bibr" rid="ref9">Guo F. et al. (2024)</xref> analysed in two stages where Stage I detects images with pixel cracks using a CNN-based classifier, while Stage II uses a separation combination approach and CTv2 (Crack Transformer v2) for pixel level detection. Extensive testing confirms the framework&#x2019;s advantage and efficiency, facilitating scalable automated pavement crack detection. <xref ref-type="bibr" rid="ref12">Karimi et al. (2024)</xref> developed a robust deep learning model for detecting cracks across various Cultural Heritage (CH) materials using the YOLO object detection network. The study examines masonry types (stone, brick, cob, tile) and modern materials like concrete with a dataset of 1,213 images across categories. Results show mean average precision values of 94.4% for concrete, 93.9% for concrete and cob, 92.7% for cob, 87.2% for stone, 83.4% for stone and brick, 81.6% for brick, and 70.3% for tile, highlighting the model&#x2019;s potential for efficient CH crack detection, aiding specialists in damage assessment.</p>
<p><xref ref-type="bibr" rid="ref7">Guo C. et al. (2024)</xref> proposed SegCrackNet, an innovative neural network with multi-level output fusion, dropout layers, and T-bridge block configurations to reduce overfitting and enhance the utilization of contextual information. Experimental results reveal notable improvements over other models, with IoU score increases of 4.3%, 9.4%, and 3.7% for the Crack500, Crack200, and pavement images datasets, respectively. <xref ref-type="bibr" rid="ref38">Yu et al. (2024)</xref> presented an optimized lightweight segmentation model similar to BiSeNetv for automated pavement crack detection. Results show that this model outperforms prior methods with an F1 score improvement of 10.14%, underscoring its precision and robustness in segmenting pavement cracks.</p>
<p><xref ref-type="bibr" rid="ref18">Liu and Xu (2023)</xref> utilized a VGG16-based CNN for crack classification, incorporating an enhanced Class Activation Map (CAM) technique for precise localization and distribution of cracks. Integrating simple linear iterative clustering (SLIC) superpixel segmentation with CAM, the semantic segmentation accuracy is improved to a greater extend. Bayesian optimization identifies ideal parameters, and test results that indicate the algorithm&#x2019;s support on image-level labelling that significantly reduced labour and cost thus maintaining accuracy. <xref ref-type="table" rid="tab1">Table 1</xref> details an overview of the various crack detection methods.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>An overview of different DL methodologies for crack detection.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Ref</th>
<th align="left" valign="top">Methodology</th>
<th align="left" valign="top">Categories</th>
<th align="left" valign="top">Dataset</th>
<th align="left" valign="top">Metric</th>
<th align="left" valign="top">Inference</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref23">Nguyen et al. (2021)</xref>
</td>
<td align="left" valign="middle">CNN</td>
<td align="left" valign="top">crack detection and segmentation</td>
<td align="left" valign="middle">DeepCrack, CrackIT, 2StagesCrack dataset</td>
<td align="left" valign="middle">F1-measure - 0.91</td>
<td align="left" valign="middle">Achieves high performance on noisy, low-resolution, and imbalanced data.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref6">Fan et al. (2020)</xref>
</td>
<td align="left" valign="middle">U-Hierarchical Dilated Network</td>
<td align="left" valign="top">Pavement crack detection</td>
<td align="left" valign="middle">AigleRN</td>
<td align="left" valign="middle">Pr: 0.92, Re: 0.93, F1: 0.92 accuracy &#x2013; 93%</td>
<td align="left" valign="middle">Superior detection accuracy through multi-scale feature extraction.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref15">Liu et al. (2022a)</xref>
</td>
<td align="left" valign="middle">Transfer Learning Models</td>
<td align="left" valign="top">Pavement crack severity classification</td>
<td align="left" valign="middle">Asphalt</td>
<td align="left" valign="middle">Accuracy &#x2212;93%</td>
<td align="left" valign="middle">Fusion images yield better accuracy; EfficientNet-B3 performs best across all image types.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref5">Elghaish et al. (2022)</xref>
</td>
<td align="left" valign="middle">convolutional neural network</td>
<td align="left" valign="top">Highway cracks</td>
<td align="left" valign="middle">4,663 images of highway cracks</td>
<td align="left" valign="middle">97.62% accuracy</td>
<td align="left" valign="middle">New CNN model outperforms existing models like GoogleNet and AlexNet.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref1">Ahmadi et al. (2022)</xref>
</td>
<td align="left" valign="middle">neural network, decision tree, SVM, KNN, Bagged Trees,</td>
<td align="left" valign="top">Crack classification</td>
<td align="left" valign="middle">400 images</td>
<td align="left" valign="middle">93.86% accuracy</td>
<td align="left" valign="middle">Hybrid model surpasses individual classifiers for crack classification.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref16">Liu et al. (2022b)</xref>
</td>
<td align="left" valign="middle">EfficientNet-B4</td>
<td align="left" valign="top">Pavement fatigue crack severity classification</td>
<td align="left" valign="top">2,211 images, while their size is 640&#x202F;&#x00D7;&#x202F;480.<break/>asphalt</td>
<td align="left" valign="middle">Accuracy &#x2013; 95%</td>
<td align="left" valign="middle">Fusion images are effective for fatigue crack severity classification.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref25">Oliveira and Correia (2014)</xref>
</td>
<td align="left" valign="middle">CrackIT toolbox algorithms</td>
<td align="left" valign="top">Crack detection</td>
<td align="left" valign="middle">84 pavement surface</td>
<td align="left" valign="middle">re&#x202F;=&#x202F;98.4%<break/>pr&#x202F;=&#x202F;95.5%<break/>100% of recall</td>
<td align="left" valign="middle">Toolbox provides crack detection and characterization algorithms for research use.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref36">Yamaguchi and Hashimoto (2010)</xref>
</td>
<td align="left" valign="middle">Percolation-based image processing</td>
<td align="left" valign="top">Concrete crack detection</td>
<td align="left" valign="middle">60 images concrete surfaces images</td>
<td align="left" valign="middle">Pre-0.95</td>
<td align="left" valign="middle">Reduced computation costs.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref34">Vivekananthan et al. (2023)</xref>
</td>
<td align="left" valign="middle">Gray intensity adjustment model for crack detection</td>
<td/>
<td align="left" valign="middle">2068 crack images</td>
<td align="left" valign="middle">95% detection accuracy</td>
<td align="left" valign="middle">Otsu and Sobel methods improve crack detection accuracy.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref13">Liu et al. (2023)</xref>
</td>
<td align="left" valign="middle">Image processing and deep learning and SVM</td>
<td align="left" valign="top">Crack images in subway tunnels,</td>
<td align="left" valign="middle">3,000 data images</td>
<td align="left" valign="middle">SVM: 88%; AlexNet: 96.7%</td>
<td align="left" valign="middle">Performs effectively for crack detection in tunnels.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref33">Tran et al. (2023)</xref>
</td>
<td align="left" valign="middle">Process-based deep learning for bridge deck crack detection</td>
<td align="left" valign="top">Crack detection and segmentation</td>
<td align="left" valign="middle">Two bridge datasets l bridge decks in South Korea</td>
<td align="left" valign="middle">Precision &#x2212;0.83</td>
<td align="left" valign="middle">Outperforms other networks in speed and accuracy for crack detection.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref29">Pham et al. (2023)</xref>
</td>
<td align="left" valign="middle">U-Net, LinkNet, Feature Pyramid Network and Deeplabv3</td>
<td align="left" valign="top">Ground crack detection</td>
<td align="left" valign="middle">510 crack, 185 slope and 325 field images</td>
<td align="left" valign="middle">F1 score: 0.877&#x2013;0.896</td>
<td align="left" valign="middle">Outperform traditional methods for crack identification and measurement.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref42">Zhang et al. (2023b)</xref>
</td>
<td align="left" valign="middle">YOLO v4</td>
<td align="left" valign="top">Bridge crack detection</td>
<td align="left" valign="middle">About 800 photos of bridges around Guizhou University.</td>
<td align="left" valign="middle">Precision: 93.96%; Recall: 90.12%</td>
<td align="left" valign="middle">Method is effective for edge deployment with minimal computational requirements.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref35">Xu et al. (2023)</xref>
</td>
<td align="left" valign="middle">YOLOv5-IDS</td>
<td align="left" valign="top">Crack detection and segmentation</td>
<td align="left" valign="middle">302 crack images</td>
<td align="left" valign="middle">mAP@0.5: 84.33%; mIoU: 94.78%</td>
<td align="left" valign="middle">Achieves high accuracy and processing speed for crack detection.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref10">Hu et al. (2024)</xref>
</td>
<td align="left" valign="middle">Improved YOLOv5</td>
<td align="left" valign="top">Road surface</td>
<td align="left" valign="middle">13,508 images</td>
<td align="left" valign="middle">F1 score &#x2013; 0.5876</td>
<td align="left" valign="middle">Enhance information extraction from vehicle-mounted images for better crack detection.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref3">Dong et al. (2024)</xref>
</td>
<td align="left" valign="middle">YOLOv8-Crack Detection (YOLOv8-CD)</td>
<td align="left" valign="top">Crack detection</td>
<td align="left" valign="middle">RDD2022 and Wall Crack datasets</td>
<td align="left" valign="middle">precision of 91.5%</td>
<td align="left" valign="middle">Improves feature extraction and detection speed for concrete surface cracks.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref2">Chen et al. (2023)</xref>
</td>
<td align="left" valign="middle">Transfer learning for automated building facade crack inspection</td>
<td align="left" valign="top">Crack detection</td>
<td align="left" valign="middle">3,600 building crack images</td>
<td align="left" valign="middle">Traditional CNN: 89%; Transfer learning: 94%</td>
<td align="left" valign="middle">Significantly enhances crack classification performance.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref39">Zadeh et al. (2024)</xref>
</td>
<td align="left" valign="middle">Deep learning architectures</td>
<td align="left" valign="top">Surface crack detection and classification</td>
<td align="left" valign="middle">20,000 images from structures within the METU Campus</td>
<td align="left" valign="middle">InceptionV3&#x2013;94%</td>
<td align="left" valign="middle">Achieves high accuracy and minimizes false positives in crack detection.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref9">Guo F. et al. (2024)</xref>
</td>
<td align="left" valign="middle">Two-stage framework CNN with a transformer model</td>
<td align="left" valign="top">Pavement surface crack detection</td>
<td align="left" valign="middle">CrackSD dataset</td>
<td align="left" valign="middle">Deep LabV3&#x2013;97.21</td>
<td align="left" valign="middle">Identifies and detects pixel-level cracks for large-scale applications.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref12">Karimi et al. (2024)</xref>
</td>
<td align="left" valign="middle">YOLOv5</td>
<td align="left" valign="top">Crack damages</td>
<td align="left" valign="middle">1,213 bricks</td>
<td align="left" valign="middle">Mean AP: 94.4%</td>
<td align="left" valign="middle">Identifies cracks in various materials, supporting inspection professionals in damage assessments.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref7">Guo C. et al. (2024)</xref>
</td>
<td align="left" valign="middle">SegCrackNet for crack detection</td>
<td/>
<td align="left" valign="middle">Crack500, Crack200, and pavement images datasets</td>
<td align="left" valign="middle">79.85, 44.97 and 49.66%</td>
<td align="left" valign="middle">Effectively detects subtle variations and improves crack detection accuracy.</td>
</tr>
<tr>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref38">Yu et al. (2024)</xref>
</td>
<td align="left" valign="middle">BiSeNetv2</td>
<td align="left" valign="top">Pavement surface crack detection</td>
<td align="left" valign="middle">CFD dataset, Crack500, CrackSC</td>
<td align="left" valign="middle">Recall - 91.09</td>
<td align="left" valign="middle">Demonstrates effectiveness and robustness in segmenting pavement surface cracks.</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><xref ref-type="bibr" rid="ref31">Shaoze et al. (2025)</xref> introduced a variant in YOLO achieving 86.4% mAP@50, demonstrating strong detection performance in complex environments. <xref ref-type="bibr" rid="ref32">Tang et al. (2024)</xref> proposed BsS-YOLO for road crack detection, integrating improved PAN and BiFPN feature fusion structures along with attention mechanisms, leading to a 2.8% mAP gain over baseline YOLO models. For bridge crack detection, <xref ref-type="bibr" rid="ref4">Dong et al. (2025)</xref> proposed YOLO11n-BD that incorporates an Efficient Multi-Scale Cross Attention (EMSCA) module and a Lightweight Dynamic Head (LDH), achieving 94.3% mAP@50 and an F1-score of 89.2% while maintaining real-time performance at 555 FPS. <xref ref-type="bibr" rid="ref46">Zhu et al. (2025)</xref> proposed FD<sup>2</sup>-YOLO that enhanced YOLOv11n with a dual-stream architecture combining spatial and frequency-domain features, improving detection robustness on noisy surfaces with 88.3% mAP@50 and 88.4% precision. <xref ref-type="bibr" rid="ref44">Zhao et al. (2025)</xref> developed a YOLOv11 for intelligent tunnel lining crack detection, achieving 93.3% accuracy, 94.5% recall, and 96.9% average precision. Their approach effectively identifies cracks under complex lighting and structural conditions, ensuring robust performance for real-world tunnel inspections.</p>
</sec>
<sec id="sec3">
<label>3</label>
<title>Proposed methodology</title>
<p>The proposed study presents custom hybrid framework for detecting surface cracks in concrete floors as illustrated in <xref ref-type="fig" rid="fig1">Figure 1</xref>. The main objective is to create a computationally efficient and accurate model that integrates the strengths of the Swin Transformer, skip learning, Enhanced Features Representation Block, along with an attention mechanism to precisely identify surface cracks in concrete structures.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Architecture of the proposed system.</p>
</caption>
<graphic xlink:href="frai-08-1655091-g001.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Diagram showcasing a SWIN transformer model for feature selection. Input images of cracks lead to patch partitioning and four processing stages. EFRB and population-based optimization interact with stages, leading to an output class.</alt-text>
</graphic>
</fig>
<p>This model used the Swin Transformer (<inline-formula>
<mml:math id="M1">
<mml:msub>
<mml:mi>ST</mml:mi>
<mml:mi mathvariant="normal">B</mml:mi>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula>and skip connections, fine-tuned through an Enhanced Feature Representation Block (<inline-formula>
<mml:math id="M2">
<mml:msub>
<mml:mi>EFR</mml:mi>
<mml:mi mathvariant="normal">B</mml:mi>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</inline-formula>, with varying filter sizes. Feature selection and optimization are achieved using Population-Based Optimization (<inline-formula>
<mml:math id="M3">
<mml:msub>
<mml:mtext>POFS</mml:mtext>
<mml:mi mathvariant="normal">B</mml:mi>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</inline-formula> which conducts a randomized search to identify optimal solutions for the efficient crack detection in images.</p>
<sec id="sec4">
<label>3.1</label>
<title>Swin Transformer</title>
<p>Unlike CNNs, Vision Transformers (ViTs) utilize the attention mechanism of Transformers for image data. A key benefit of ViT is its ability to represent global features without depending on local receptive fields. Transformers self-attention necessitates calculating weights between all other tokens, leading to increased computational complexity. As a result, the computational cost associated with super-resolution images can be substantial. In contrast to ViT, the Swin Transformer (<xref ref-type="bibr" rid="ref14">Liu et al., 2021</xref>) incorporates a mechanism known as the shifted window, which segments into non-overlapping localized. Features are further processed among windows through this shifting process. Swin Transformer employed a hierarchical process composed of various stages, each containing several transformer blocks. <xref ref-type="fig" rid="fig2">Figure 2</xref> provides the summary of the Swin Transformer architecture.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Flow of layers in Swin Transformer.</p>
</caption>
<graphic xlink:href="frai-08-1655091-g002.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Diagram illustrating a two-part model architecture. Panel a shows a process involving four stages: each stage includes "Patch Partition," followed by "Linear Embedding" and "ST Block." Panel b displays a loop structure with "LN," "W-MSA," "SW-MSA," and "MLP" blocks connected through arrows, indicating iterative processing.</alt-text>
</graphic>
</fig>
<p>The input image, of size <inline-formula>
<mml:math id="M4">
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:math>
</inline-formula>, is splitted into non-overlapping patches of size <inline-formula>
<mml:math id="M5">
<mml:mfrac>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mn>4</mml:mn>
</mml:mfrac>
<mml:mo>&#x00D7;</mml:mo>
<mml:mfrac>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mn>4</mml:mn>
</mml:mfrac>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>48</mml:mn>
</mml:math>
</inline-formula>. The input data is processed at the final stage through a linear layer that transforms the feature into <inline-formula>
<mml:math id="M6">
<mml:mi mathvariant="normal">C</mml:mi>
</mml:math>
</inline-formula> which is enhanced through an attention model. The same operations are repeated in the subsequent three stages. The adjacent <inline-formula>
<mml:math id="M7">
<mml:mn>2</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>2</mml:mn>
</mml:math>
</inline-formula> patches are combined through a patch merging, which reduces size by half through a linear layer followed by multiple blocks to enhance the merged patches using attention blocks. Ultimately, the resulting data has dimensions of <inline-formula>
<mml:math id="M8">
<mml:mfrac>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mn>32</mml:mn>
</mml:mfrac>
<mml:mo>&#x00D7;</mml:mo>
<mml:mfrac>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mn>32</mml:mn>
</mml:mfrac>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>8</mml:mn>
<mml:mi mathvariant="normal">C</mml:mi>
</mml:math>
</inline-formula>. <xref ref-type="fig" rid="fig2">Figure 2</xref> depicts two consecutive Swin Transformer blocks, where the conventional multi-head self-attention mechanism (MSA) is substituted with window-based multi-head self-attention (W-MSA) and shifted window multi-head self-attention (SW-MSA). By leveraging the partitioning shifted window technique, the representation generated by successive Swin Transformer blocks can be expressed as <xref ref-type="disp-formula" rid="EQ1 EQ2 EQ3 EQ4">Equations 1&#x2013;4</xref>:</p>
<disp-formula id="EQ1">
<mml:math id="M9">
<mml:msup>
<mml:mover accent="true">
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>MSA</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:msub>
<mml:mo stretchy="true">(</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
<label>(1)</label></disp-formula>
<disp-formula id="EQ2">
<mml:math id="M10">
<mml:msup>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mi>MLP</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:msub>
<mml:mo stretchy="true">(</mml:mo>
<mml:msup>
<mml:mover accent="true">
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:msup>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mover accent="true">
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:msup>
</mml:math>
<label>(2)</label></disp-formula>
<disp-formula id="EQ3">
<mml:math id="M11">
<mml:msup>
<mml:mover accent="true">
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mi>SW</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>MSA</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:msub>
<mml:mo stretchy="true">(</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:msup>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:msup>
</mml:math>
<label>(3)</label></disp-formula>
<disp-formula id="EQ4">
<mml:math id="M12">
<mml:msup>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mi>MLP</mml:mi>
<mml:mspace width="0.25em"/>
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
</mml:msub>
<mml:mo stretchy="true">(</mml:mo>
<mml:msup>
<mml:mover accent="true">
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mover accent="true">
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
<label>(4)</label></disp-formula>
<p>Where <inline-formula>
<mml:math id="M13">
<mml:msup>
<mml:mover accent="true">
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
</mml:msup>
</mml:math>
</inline-formula> represent the output of the (S)W-MSA and the MLP module of block &#x1D459;, respectively; and SW-MSA and W-MSA refer to window-based multi-head self-attention mechanisms that utilize standard and shifted window partitioning processes, respectively. Consider each window includes <inline-formula>
<mml:math id="M14">
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi mathvariant="normal">M</mml:mi>
</mml:math>
</inline-formula> patches, the complexity of multi-head self-attention module and W-MSA for <inline-formula>
<mml:math id="M15">
<mml:mi mathvariant="normal">h</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi mathvariant="normal">w</mml:mi>
</mml:math>
</inline-formula> patches are as given in <xref ref-type="disp-formula" rid="EQ5 EQ6">Equations 5, 6</xref>:</p>
<disp-formula id="EQ5">
<mml:math id="M16">
<mml:mi>&#x03A9;</mml:mi>
<mml:mi>MSA</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>4</mml:mn>
<mml:msup>
<mml:mi>hwC</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mn>2</mml:mn>
<mml:mspace width="0.25em"/>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>hw</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mi mathvariant="normal">C</mml:mi>
</mml:math>
<label>(5)</label></disp-formula>
<disp-formula id="EQ6">
<mml:math id="M17">
<mml:mi mathvariant="italic">&#x03A9;W</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>MSA</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>4</mml:mn>
<mml:msup>
<mml:mi>hwC</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mn>2</mml:mn>
<mml:mspace width="0.25em"/>
<mml:msup>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mi>hwC</mml:mi>
</mml:math>
<label>(6)</label></disp-formula>
<p>The complexity of MSA is quadratically linked to patch count, meaning it increases significantly with a larger number of patches. In contrast, when the size <inline-formula>
<mml:math id="M18">
<mml:mi mathvariant="normal">M</mml:mi>
</mml:math>
</inline-formula> is constant, the complexity of W-MSA remains linear. As a result, the rise in complexity is quite modest even with a greater number of patches. This characteristic improves the scalability of W-MSA for processing large-scale images.</p>
</sec>
<sec id="sec5">
<label>3.2</label>
<title>Enhanced features representation block</title>
<p>A neural network block combining Depthwise and Pointwise Convolutions leverages Depthwise Convolutions (<xref ref-type="bibr" rid="ref8">Guo et al., 2019</xref>) to capture spatial features independently across each channel, enhancing spatial granularity. The Pointwise Convolution (<xref ref-type="bibr" rid="ref11">Hua et al., 2018</xref>) then integrates these spatially focused features, creating a rich, channel-combined representation that enhances the model&#x2019;s ability to capture essential and discriminative features efficiently. Depthwise Convolution where each filter is applied to only one input channel. In contrast to standard convolutions, depthwise convolutions reduce computational complexity as given in <xref ref-type="disp-formula" rid="EQ7">Equation 7</xref>.</p>
<disp-formula id="EQ7">
<mml:math id="M19">
<mml:msub>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mi mathvariant="normal">v</mml:mi>
</mml:msub>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mo movablelimits="false">&#x2211;</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">j</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi mathvariant="normal">Y</mml:mi>
<mml:mi mathvariant="normal">v</mml:mi>
</mml:msub>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">j</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>&#x22C5;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">v</mml:mi>
</mml:msub>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">j</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
<label>(7)</label></disp-formula>
<p>where <inline-formula>
<mml:math id="M20">
<mml:msub>
<mml:mi mathvariant="normal">Y</mml:mi>
<mml:mi mathvariant="normal">v</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> is the v<sup>th</sup> channel of the input, <inline-formula>
<mml:math id="M21">
<mml:msub>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">v</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> is the depthwise filter for that channel, and <inline-formula>
<mml:math id="M22">
<mml:msub>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mi mathvariant="normal">v</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> is the corresponding output. This is a standard 2D convolution applied after the depthwise convolution. It combines the output from the depthwise convolution across channels as shown in <xref ref-type="disp-formula" rid="EQ8">Equation 8</xref>.</p>
<disp-formula id="EQ8">
<mml:math id="M23">
<mml:mi mathvariant="normal">O</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mo movablelimits="false">&#x2211;</mml:mo>
<mml:mi mathvariant="normal">v</mml:mi>
</mml:munder>
<mml:msub>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mi mathvariant="normal">v</mml:mi>
</mml:msub>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>&#x22C5;</mml:mo>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
</mml:math>
<label>(8)</label></disp-formula>
<p>Where <inline-formula>
<mml:math id="M24">
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>v</mml:mi>
</mml:msub>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula>represents the filters in this conv2D layer applied across the depthwise operator. GELU is a smooth activation function, where the output is a stochastic binary decision with some non-linearity as shown in <xref ref-type="disp-formula" rid="EQ9 EQ10">Equations 9, 10</xref>.</p>
<disp-formula id="EQ9">
<mml:math id="M25">
<mml:mtext mathvariant="italic">Gelu</mml:mtext>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>&#x22C5;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>Y</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>&#x22C5;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mo stretchy="true">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:mi>e</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mfrac>
<mml:mi>y</mml:mi>
<mml:msqrt>
<mml:mn>2</mml:mn>
</mml:msqrt>
</mml:mfrac>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
<label>(9)</label></disp-formula>
<disp-formula id="EQ10">
<mml:math id="M26">
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B2;</mml:mi>
</mml:mrow>
<mml:mi>&#x03B3;</mml:mi>
</mml:mfrac>
</mml:math>
<label>(10)</label></disp-formula>
<p>This process normalizes the activations across the features within a layer to improve stability and training efficiency. <inline-formula>
<mml:math id="M27">
<mml:mi>&#x03B2;</mml:mi>
</mml:math>
</inline-formula> refers mean and <inline-formula>
<mml:math id="M28">
<mml:mi>&#x03B3;</mml:mi>
</mml:math>
</inline-formula> represents standard deviation within a layer. The output of the depthwise convolution branch is added back to the input via a residual connection, helping in training deeper networks by avoiding vanishing gradient issues. Pointwise Convolution is a 1&#x00D7;1 convolution applied across the channels, used to fuse information across channels without altering the spatial dimensions as given in <xref ref-type="disp-formula" rid="EQ11">Equation 11</xref>.</p>
<disp-formula id="EQ11">
<mml:math id="M29">
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mo movablelimits="false">&#x2211;</mml:mo>
<mml:mi>v</mml:mi>
</mml:munder>
<mml:mi>Y</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>&#x22C5;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
<label>(11)</label></disp-formula>
<p>where <inline-formula>
<mml:math id="M30">
<mml:mi>v</mml:mi>
</mml:math>
</inline-formula> refers the channels in the input and the pointwise convolution across all channels. Conv2D, GELU, Layer Normalization, Dropout follow the same principles as the depthwise convolution branch. The Conv2D is used to mix information across channels after the pointwise convolution. GELU activation, Layer Normalization, and Dropout work identically in both branches to introduce non-linearity, normalize activations, and prevent overfitting, respectively. Similar to the depthwise branch, the output from the pointwise convolution branch is added back to the input. The depthwise convolution enahances the spatial features whereas Pointwise convolutions combine features across channels efficiently. The residual connections helped to overcome the vanishing gradients, and the normalization layers stabilized the learning process effectively. Layer Normalization and Dropout layers help improve the model&#x2019;s generalization by stabilizing training and reducing overfitting, respectively. <xref ref-type="fig" rid="fig3">Figure 3</xref> details the layers included in the enhanced feature representation model.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Overview of the enhanced feature representation model.</p>
</caption>
<graphic xlink:href="frai-08-1655091-g003.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Flowchart depicting two parallel convolutional processes. The left side shows a sequence of Depthwise Conv, Conv2D, GeLU, Layer Normalization, and Drop, with a skip connection returning to the beginning. The right side mirrors this with Pointwise Conv in place of Depthwise Conv. Both processes converge at a sum node.</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec6">
<label>3.3</label>
<title>Optimised feature selection</title>
<p>Population-based optimization techniques use random searches to identify the optimal solutions. Also, an adaptive local search technique called Adaptive <italic>&#x03B2;</italic>-Hill Climbing (A&#x03B2;HC) is used to fine-tune the selected feature. This feature forms a mapping to the output classes, resulting in the Hierarchical Deep Learning Classifier (HDLC) to effectively distinguish between cracked and non-cracked surfaces based on the refined input features. The Sine-Cosine Algorithm (SCA) is a population-based metaheuristic used for feature selection and optimization (<xref ref-type="bibr" rid="ref22">Mirjalili, 2016</xref>). It uses sine and cosine functions in an iterative process with two phases: exploration, which introduces diverse solutions to search broadly, and exploitation, which fine-tunes solutions by reducing randomness. <xref ref-type="disp-formula" rid="EQ12">Equation 12</xref> defines how positions are updated using these functions.</p>
<disp-formula id="EQ12">
<mml:math id="M31">
<mml:msubsup>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mo stretchy="true">{</mml:mo>
<mml:mtable displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>&#x00D7;</mml:mo>
<mml:mo>sin</mml:mo>
<mml:mo stretchy="true">(</mml:mo>
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>&#x00D7;</mml:mo>
<mml:mo>&#x2223;</mml:mo>
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi>N</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>&#x2223;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>4</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>&#x003C;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msubsup>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>&#x00D7;</mml:mo>
<mml:mo>cos</mml:mo>
<mml:mo stretchy="true">(</mml:mo>
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>&#x00D7;</mml:mo>
<mml:mo>&#x2223;</mml:mo>
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mi>N</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>&#x2223;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>4</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo stretchy="true">}</mml:mo>
</mml:math>
<label>(12)</label></disp-formula>
<p>Here, <inline-formula>
<mml:math id="M32">
<mml:msubsup>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula>is the position in <inline-formula>
<mml:math id="M33">
<mml:msup>
<mml:mi>j</mml:mi>
<mml:mi mathvariant="italic">th</mml:mi>
</mml:msup>
</mml:math>
</inline-formula> dimension of <inline-formula>
<mml:math id="M34">
<mml:msup>
<mml:mi>i</mml:mi>
<mml:mi mathvariant="italic">th</mml:mi>
</mml:msup>
</mml:math>
</inline-formula> search element at <inline-formula>
<mml:math id="M35">
<mml:msup>
<mml:mi>m</mml:mi>
<mml:mi mathvariant="italic">th</mml:mi>
</mml:msup>
</mml:math>
</inline-formula> iteration. <inline-formula>
<mml:math id="M36">
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mspace width="0.25em"/>
<mml:mtext mathvariant="italic">and</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>4</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mspace width="0.25em"/>
<mml:mi mathvariant="italic">are</mml:mi>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula> random numbers, <inline-formula>
<mml:math id="M37">
<mml:msubsup>
<mml:mi>N</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>m</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> represents the position of <inline-formula>
<mml:math id="M38">
<mml:msup>
<mml:mi>j</mml:mi>
<mml:mi mathvariant="italic">th</mml:mi>
</mml:msup>
</mml:math>
</inline-formula> best solution at <inline-formula>
<mml:math id="M39">
<mml:msup>
<mml:mi>m</mml:mi>
<mml:mi mathvariant="italic">th</mml:mi>
</mml:msup>
</mml:math>
</inline-formula> iteration and || denotes the absolute value. A random value <inline-formula>
<mml:math id="M40">
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula>enables the transition from exploration to exploitation as shown in <xref ref-type="disp-formula" rid="EQ13">Equation 13</xref>.</p>
<disp-formula id="EQ13">
<mml:math id="M41">
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mi>&#x03B2;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mfrac>
<mml:mi>&#x03B2;</mml:mi>
<mml:mi>M</mml:mi>
</mml:mfrac>
</mml:math>
<label>(13)</label></disp-formula>
<p>Here, <inline-formula>
<mml:math id="M42">
<mml:mi>&#x03B2;</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M43">
<mml:mi>M</mml:mi>
</mml:math>
</inline-formula> characterize the constant value, and iterations, respectively. <inline-formula>
<mml:math id="M44">
<mml:mspace width="0.25em"/>
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula>decides whether the search region is for <inline-formula>
<mml:math id="M45">
<mml:mo stretchy="true">(</mml:mo>
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>&#x2208;</mml:mo>
<mml:mo stretchy="true">[</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="true">]</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</inline-formula> or exploration <inline-formula>
<mml:math id="M46">
<mml:mo stretchy="true">(</mml:mo>
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>&#x2208;</mml:mo>
<mml:mo stretchy="true">[</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo stretchy="true">]</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula>or <inline-formula>
<mml:math id="M47">
<mml:mo stretchy="true">(</mml:mo>
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mo>&#x2208;</mml:mo>
<mml:mo stretchy="true">[</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo stretchy="true">]</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
</mml:math>
</inline-formula>. The stochastic variable <inline-formula>
<mml:math id="M48">
<mml:msubsup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msubsup>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula>, ranging within <inline-formula>
<mml:math id="M49">
<mml:mo stretchy="true">[</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>&#x03C0;</mml:mi>
<mml:mo stretchy="true">]</mml:mo>
</mml:math>
</inline-formula>, controls the search agent&#x2019;s direction relative to the destination, aligning with the sine and cosine cycle. <inline-formula>
<mml:math id="M50">
<mml:msubsup>
<mml:mi mathvariant="normal">k</mml:mi>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">j</mml:mi>
</mml:mrow>
<mml:mi mathvariant="normal">m</mml:mi>
</mml:msubsup>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula>balances exploration and exploitation by assigning a random weight between 0 and 2, influencing step size&#x2014;greater than 1 emphasizes, while less than 1 de-emphasizes the destination&#x2019;s impact. <inline-formula>
<mml:math id="M51">
<mml:msubsup>
<mml:mi mathvariant="normal">k</mml:mi>
<mml:mrow>
<mml:mn>4</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">j</mml:mi>
</mml:mrow>
<mml:mi mathvariant="normal">m</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> manages the switch between sine and cosine functions, as outlined in <xref ref-type="disp-formula" rid="EQ14">Equation 14</xref>. The SCA feature optimization process is summarized in the flowchart shown in <xref ref-type="fig" rid="fig4">Figure 4</xref>. In order to enhance the exploitation ability Adaptive &#x1D6FD;-Hill Climbing (A&#x1D6FD;HC) is integrated that utilizes local search-based techniques using two control parameters and <inline-formula>
<mml:math id="M52">
<mml:mspace width="0.25em"/>
<mml:msub>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi>hc</mml:mi>
</mml:msub>
</mml:math></inline-formula> and <inline-formula><mml:math id="M152">
<mml:msub>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi>hc</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>. The parameter <inline-formula>
<mml:math id="M53">
<mml:mspace width="0.25em"/>
<mml:msub>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi>hc</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> is assigned close to 1 value which gradually decreases as the search process progresses. This permits the process to dynamically adjust <inline-formula>
<mml:math id="M54">
<mml:mspace width="0.25em"/>
<mml:msub>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi>hc</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> to advance the search performance, as given in <xref ref-type="disp-formula" rid="EQ14">Equation 14</xref>.</p>
<disp-formula id="EQ14">
<mml:math id="M55">
<mml:msubsup>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi>hc</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:msup>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi mathvariant="normal">p</mml:mi>
</mml:mfrac>
</mml:msup>
<mml:msubsup>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi>max</mml:mi>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi mathvariant="normal">p</mml:mi>
</mml:mfrac>
</mml:msubsup>
</mml:mfrac>
</mml:math>
<label>(14)</label></disp-formula>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>The flowchart for the population-based optimization algorithm.</p>
</caption>
<graphic xlink:href="frai-08-1655091-g004.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Flowchart illustrating a process that starts with initializing a population and assigning fitness values. Updates occur based on the variable \( k_{m, j} \). A decision diamond checks if \( k_{m, j} &#x003E; 0.5 \), leading to different update formulas for each outcome. After the update, another decision checks if the process should terminate. If yes, the best value is selected. If no, the process continues. It concludes with a stop.</alt-text>
</graphic>
</fig>
<p>Here, <inline-formula>
<mml:math id="M56">
<mml:msubsup>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi>hc</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> denotes <inline-formula>
<mml:math id="M57">
<mml:mspace width="0.25em"/>
<mml:msub>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi>hc</mml:mi>
</mml:msub>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula>at time m, P linearly decrease the <inline-formula>
<mml:math id="M58">
<mml:mspace width="0.25em"/>
<mml:msub>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi>hc</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> to a value close to 0 and <inline-formula>
<mml:math id="M59">
<mml:msub>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi>max</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> represents the upper limit B parameter adapts a range <inline-formula>
<mml:math id="M60">
<mml:mo>&#x2208;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi>hc</mml:mi>
<mml:mi>min</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi>hc</mml:mi>
<mml:mi>max</mml:mi>
</mml:msubsup>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula>, mathematically expressed in <xref ref-type="disp-formula" rid="EQ15">Equation 15</xref>.</p>
<disp-formula id="EQ15">
<mml:math id="M61">
<mml:msubsup>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi>hc</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi>hc</mml:mi>
<mml:mi>min</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi>hc</mml:mi>
<mml:mi>max</mml:mi>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi>hc</mml:mi>
<mml:mi>min</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi>max</mml:mi>
</mml:msub>
</mml:mfrac>
</mml:math>
<label>(15)</label></disp-formula>
<p>Here, <inline-formula>
<mml:math id="M62">
<mml:msubsup>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi>hc</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
</mml:msubsup>
</mml:math>
</inline-formula> represents the rate of <inline-formula>
<mml:math id="M63">
<mml:mspace width="0.25em"/>
<mml:msub>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi>hc</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> at iteration <inline-formula>
<mml:math id="M64">
<mml:mi mathvariant="normal">m</mml:mi>
</mml:math>
</inline-formula>, with <inline-formula>
<mml:math id="M65">
<mml:msubsup>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi>hc</mml:mi>
<mml:mi>min</mml:mi>
</mml:msubsup>
<mml:mspace width="0.25em"/>
<mml:mtext>and</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:msubsup>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi>hc</mml:mi>
<mml:mi>max</mml:mi>
</mml:msubsup>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula>indicating the minimum and maximum value of <inline-formula>
<mml:math id="M66">
<mml:mspace width="0.25em"/>
<mml:msub>
<mml:mi mathvariant="normal">B</mml:mi>
<mml:mi>hc</mml:mi>
</mml:msub>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula>respectively, <inline-formula>
<mml:math id="M67">
<mml:msub>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi>max</mml:mi>
</mml:msub>
<mml:mspace width="0.25em"/>
</mml:math>
</inline-formula>denotes the total number of iterations, and <inline-formula>
<mml:math id="M68">
<mml:mi mathvariant="normal">m</mml:mi>
</mml:math>
</inline-formula> refers to the current iteration. <xref ref-type="fig" rid="fig4">Figure 4</xref> details the flowchart for the Population-based optimization algorithm (<xref ref-type="bibr" rid="ref30">Raghaw et al., 2024</xref>) detailing the selection process of the optimized features.</p>
</sec>
</sec>
<sec id="sec7">
<label>4</label>
<title>Experimental results and discussion</title>
<p>Overview of the dataset, data augmentation methods, experimental setup, model training, and validation. It also details the performance metrics used to analyse the proposed model is detailed in this section.</p>
<sec id="sec8">
<label>4.1</label>
<title>Description of the dataset</title>
<p>Concrete surface cracks is a defect commonly identified in civil infrastructure. Building inspection is crucial for assessing the structural integrity and tensile strength of these constructions. Crack detection (<xref ref-type="bibr" rid="ref27">&#x00D6;zgenel, 2019</xref>; <xref ref-type="bibr" rid="ref28">&#x00D6;zgenel and G&#x00F6;nen&#x00E7; Sorgu&#x00E7;, 2018</xref>) plays a vital role in this process by identifying structural flaws and evaluating the overall condition of the building. These images are organized into two class: negative (no cracks) and positive (with cracks), suitable for image classification tasks. Each category includes 20,000 images, resulting in a total of 40,000 RGB images, each with a resolution of 227 &#x00D7; 227 pixels. The dataset was developed from 458 high-resolution images (4,032 &#x00D7; 3,024 pixels) following the method introduced by <xref ref-type="bibr" rid="ref43">Zhang et al. (2016)</xref>. These high-resolution images display considerable variation in surface texture and lighting conditions. <xref ref-type="fig" rid="fig5">Figure 5</xref> shows few sample images for each class.</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>Sample images for both crack and non-crack images.</p>
</caption>
<graphic xlink:href="frai-08-1655091-g005.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Two rows of four close-up images each, showing concrete surfaces. The top row consists of four images without visible cracks, highlighting textured surfaces with varying shades of gray. The bottom row displays surfaces with distinct cracks, showing differences in size and orientation. Each image includes x and y axes labeled from zero to two hundred.</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec9">
<label>4.2</label>
<title>Environmental setup</title>
<p>The proposed model was implemented using PyTorch, an open-source deep learning framework. To optimize the model and minimize loss, the Adam optimizer was incorporated with a learning rate set to 0.0001. The training was performed on an Azure virtual machine powered by an NVIDIA Tesla P40 GPU.</p>
</sec>
<sec id="sec10">
<label>4.3</label>
<title>Evaluation metrics</title>
<p>The metrics that are used to evaluate the model are as given in <xref ref-type="disp-formula" rid="EQ16 EQ17 EQ18 EQ19 EQ20">Equations 16&#x2013;20</xref>. Accuracy measures how the predicted values are similar with the actual values. Precision identified true positive values. Specificity indicates the model&#x2019;s ability to correctly identify true negatives, computed as the ratio of true negatives to the total number of negative cases. Recall represents the proportion of correctly predicted positive cases out of all actual positive data in the dataset. The F1 score, which is the harmonic mean of precision and recall, which reflects the model&#x2019;s effectiveness in detecting positive samples. These evaluation metrics are calculated based on True Positives (TP), False Positives (FP), True Negatives (TN), and False Negatives (FN), as given in <xref ref-type="disp-formula" rid="EQ8 EQ9 EQ10 EQ11 EQ12">Equations 8&#x2013;12</xref>.</p>
<disp-formula id="EQ16">
<mml:math id="M69">
<mml:mtext>Accuracy</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mi>TP</mml:mi>
<mml:mrow>
<mml:mi>TP</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>TN</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>FP</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>FN</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(16)</label></disp-formula>
<disp-formula id="EQ17">
<mml:math id="M70">
<mml:mtext>Precision</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mi>TP</mml:mi>
<mml:mrow>
<mml:mi>TP</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>FP</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(17)</label></disp-formula>
<disp-formula id="EQ18">
<mml:math id="M71">
<mml:mtext>Recall</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mi>TP</mml:mi>
<mml:mrow>
<mml:mi>TP</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>FN</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(18)</label></disp-formula>
<disp-formula id="EQ19">
<mml:math id="M72">
<mml:mtext>Specificity</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mi>TN</mml:mi>
<mml:mrow>
<mml:mi>TN</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>FP</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(19)</label></disp-formula>
<disp-formula id="EQ20">
<mml:math id="M73">
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mspace width="0.25em"/>
<mml:mtext>Score</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#x2217;</mml:mo>
<mml:mtext>Precision</mml:mtext>
<mml:mo>&#x2217;</mml:mo>
<mml:mtext>Recall</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>Precision</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext>Recall</mml:mtext>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(20)</label></disp-formula>
</sec>
<sec id="sec11">
<label>4.4</label>
<title>Training details</title>
<p>The model included a 2 &#x00D7; 2 patch size for the initial image segmentation and processes these patches with 8 attention heads and a 64-dimensional embedding. The attention mechanism is performed within a 2 &#x00D7; 2 window which incorporates a shifted window of size 1. Dropout is applied with 0.03 to avoid overfitting. The training is carried out with a learning rate of 1e-3, batch size of 16, and 30 epochs, utilizing weight decay and label smoothing for improved generalization. The model used 80:20 ratio ensuring a suitable split for evaluating performance during the training process.</p>
</sec>
<sec id="sec12">
<label>4.5</label>
<title>Ablation studies</title>
<p>This study evaluates the efficiency of major components in the proposed architectural framework that helps to optimize the performance. The efficiency of the following levels was evaluated: Swin Transformer, Enhanced Features Representation Block and Population based Optimisation for Feature Selection.</p>
<sec id="sec13">
<label>4.5.1</label>
<title>Analysis of the Swin Transformer</title>
<p>The performance of the Swin Transformer was analyzed as an independent module to evaluate its effectiveness in feature extraction for crack detection. This analysis shows the model&#x2019;s ability to capture long-range dependencies, for identifying fine-grained crack patterns and irregularities in structural images. The Swin Transformer employed a hierarchical architecture with shifted windows, enabling efficient computation while preserving spatial granularity. The self-attention mechanism ensures robust modeling of both local and global contextual relationships, important for distinguishing cracks from background textures. <xref ref-type="fig" rid="fig6">Figures 6</xref>, <xref ref-type="fig" rid="fig7">7</xref> illustrate the training and performance metrics of the Swin Transformer block achieving a testing accuracy of 91.68%. The Swin Transformer&#x2019;s performance as an independent module was further analyzed to assess its capability in crack detection. The training and validation curves showed rapid convergence within the first few epochs, achieving a stable testing accuracy of 91.68% with minimal overfitting. The confusion matrix indicated a strong balance between precision and recall for both crack and non-crack classes, demonstrating the model&#x2019;s robustness in distinguishing fine-grained crack patterns from background noise. This performance highlights the Swin Transformer&#x2019;s ability to effectively model both local and global contextual features through its hierarchical shifted window mechanism while maintaining computational efficiency, making it a reliable backbone for structural crack detection tasks.</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>Accuracy and loss plot analysis obtained using Swin Transformer.</p>
</caption>
<graphic xlink:href="frai-08-1655091-g006.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Two graphs display model training metrics over epochs. The left graph shows training and validation accuracy, peaking around 0.9 after 30 epochs. The right graph displays training and validation loss, decreasing significantly and stabilizing below 0.5. Both graphs illustrate performance improvements, with training shown in blue and validation in orange.</alt-text>
</graphic>
</fig>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>Confusion matrix obtained using Swin Transformer.</p>
</caption>
<graphic xlink:href="frai-08-1655091-g007.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Confusion matrix showing actual versus predicted values. True positive: 3772, false positive: 328, false negative: 338, true negative: 3562. A color gradient indicates value intensity.</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec14">
<label>4.5.2</label>
<title>Analysis of the Swin Transformer with enhanced features representation block</title>
<p>The combined performance of the Swin Transformer and the Enhanced Features Representation Block (EFRB) was analyzed to evaluate their collaboration in feature extraction for crack detection. The Swin Transformer is integrated with the EFRB&#x2019;s ability to enhance spatial granularity and channel-wise representation, resulting in a feature extraction framework. The Swin Transformer acts as the initial stage, effectively processing complex input images with its hierarchical architecture and shifted window self-attention mechanism. This enables both global and local contextual relationships critical for detection of subtle and irregular crack patterns. The extracted features are then passed to the Enhanced Features Representation Block, which employs Depthwise Convolutions to independently refine spatial features across channels and Pointwise Convolutions to fuse these features into channel-combined representation. The EFRB&#x2019;s residual connections and Layer Normalization stabilize training, while the GELU activation and Dropout layers prevent overfitting, ensuring robust learning. <xref ref-type="fig" rid="fig8">Figures 8</xref>, <xref ref-type="fig" rid="fig9">9</xref> illustrate the training process and performance evaluation of the Swin Transformer combined with the EFRB. The metrics demonstrate improved convergence rates, stability, and feature extraction efficiency compared to using the Swin Transformer as a standalone component attaining 95.43% as accuracy.</p>
<fig position="float" id="fig8">
<label>Figure 8</label>
<caption>
<p>Accuracy and loss plot obtained using Swin Transformer and the EFRB.</p>
</caption>
<graphic xlink:href="frai-08-1655091-g008.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Two line graphs depicting training and validation metrics over epochs. Left graph shows accuracy increasing from 0 to 1 over 30 epochs for both training and validation. Right graph shows loss decreasing from 2.5 to below 0.5 for both training and validation over the same period.</alt-text>
</graphic>
</fig>
<fig position="float" id="fig9">
<label>Figure 9</label>
<caption>
<p>Confusion matrix obtained using Swin Transformer and the EFRB.</p>
</caption>
<graphic xlink:href="frai-08-1655091-g009.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Confusion matrix showing predicted versus actual values. The true positives are 3,972 and true negatives are 3,762. The false positives are 128 and false negatives are 138. A color scale is present on the right.</alt-text>
</graphic>
</fig>
<p>The integration of the Swin Transformer with the Enhanced Features Representation Block (EFRB) demonstrated improved feature extraction performance for crack detection. The Swin Transformer efficiently modeled both global and local dependencies through its hierarchical shifted window mechanism, while the EFRB enhanced spatial granularity and channel-wise representation using Depthwise and Pointwise Convolutions. Residual connections, Layer Normalization, and GELU activation stabilized training and reduced overfitting, ensuring robust learning. The training and validation curves showed faster convergence and higher stability compared to the Swin Transformer alone, while the confusion matrix confirmed significant improvement in classification accuracy, achieving 95.43%, highlighting the combined model&#x2019;s effectiveness for precise crack detection.</p>
</sec>
<sec id="sec15">
<label>4.5.3</label>
<title>Performance analysis of the proposed model</title>
<p>The proposed model integrates the Swin Transformer, the Enhanced Features Representation Block (EFRB), and Population-based Optimization for Feature Selection, resulting an enhanced framework for crack detection. Each component contributes distinct strengths enhancing the network&#x2019;s overall performance in extracting, refining, and selecting discriminative features. The Swin Transformer efficiently captures both global and local dependencies through its hierarchical architecture and shifted window mechanism. The Enhanced Features Representation Block (EFRB) refines and enhances the extracted features. The Depthwise Convolutions within the EFRB specialize in spatial feature extraction by independently processing each channel, while the Pointwise Convolutions integrate these spatially refined features across channels. Residual connections, along with GELU activation, Layer Normalization, and Dropout, ensure stable training, robust gradient flow, and generalization, resulting in a rich and discriminative representation tailored for crack detection. Population-based Optimization for Feature Selection ensures that the most relevant and informative features are prioritized while redundant or non-contributory features are minimized. By leveraging evolutionary algorithms, this optimization step enhances the model&#x2019;s predictive accuracy while reducing computational overhead, making the network efficient and scalable. <xref ref-type="table" rid="tab2">Table 2</xref> shows the accuracy attained by the ablation studies of each component of the proposed work.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Performance obtained during ablation studies.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Architecture</th>
<th align="center" valign="top">Accuracy</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Swin Transformer</td>
<td align="center" valign="top">91.68%</td>
</tr>
<tr>
<td align="left" valign="top">Swin Transformer with enhanced features representation block</td>
<td align="center" valign="top">95.43%</td>
</tr>
<tr>
<td align="left" valign="top">Proposed work</td>
<td align="center" valign="top">98%</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><xref ref-type="fig" rid="fig10">Figure 10</xref> illustrate the training process and performance of the proposed network. The results demonstrate a good improvement in accuracy, robustness, and convergence compared to individual components analyzed separately. <xref ref-type="fig" rid="fig11">Figure 11</xref> shows the confusion matrix along with the ROC plot obtained for the proposed model. The integration of the Swin Transformer, EFRB, and Population-based Optimization ensures a powerful and balanced approach to crack detection, achieving high precision and generalization across diverse datasets.</p>
<fig position="float" id="fig10">
<label>Figure 10</label>
<caption>
<p>Accuracy and loss plot of the proposed model.</p>
</caption>
<graphic xlink:href="frai-08-1655091-g010.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Two line graphs display neural network training metrics over 30 epochs. The left graph shows train and validation losses, both decreasing sharply and stabilizing below 0.5. The right graph depicts train and validation accuracies, both increasing towards and oscillating around 0.95.</alt-text>
</graphic>
</fig>
<fig position="float" id="fig11">
<label>Figure 11</label>
<caption>
<p>Performance metrics showing the confusion matrix along with the ROC plot.</p>
</caption>
<graphic xlink:href="frai-08-1655091-g011.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Confusion matrix heatmap and ROC curve are shown side by side. The confusion matrix indicates accurate predictions with 3,972 true negatives, 28 false positives, 138 false negatives, and 3,862 true positives. The ROC curve displays near-perfect classification with areas under the curve for both classes equal to 1.0.</alt-text>
</graphic>
</fig>
<p>The proposed network achieves enhanced performance with an overall accuracy of 98%, demonstrating precision, recall, and F1-scores of 0.97, 0.99, and 0.98 for crack detection, respectively, highlighting its robustness and effectiveness for real-world applications as shown in <xref ref-type="table" rid="tab3">Table 3</xref>.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Performance metrics of the proposed work.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Class</th>
<th align="center" valign="top">Precision</th>
<th align="center" valign="top">Recall</th>
<th align="center" valign="top">F1-Score</th>
<th align="center" valign="top">Sensitivity</th>
<th align="center" valign="top">Specificity</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Crack</td>
<td align="center" valign="top">0.97</td>
<td align="center" valign="top">0.99</td>
<td align="center" valign="top">0.98</td>
<td align="center" valign="top">0.99</td>
<td align="center" valign="top">0.96</td>
</tr>
<tr>
<td align="left" valign="top">No Crack</td>
<td align="center" valign="top">0.99</td>
<td align="center" valign="top">0.97</td>
<td align="center" valign="top">0.98</td>
<td align="center" valign="top">0.96</td>
<td align="center" valign="top">0.99</td>
</tr>
<tr>
<td align="left" valign="top">Accuracy</td>
<td align="center" valign="top" colspan="5">98%</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The model, integrating the Swin Transformer, Enhanced Features Representation Block (EFRB), and Population-based Optimization, achieved a testing accuracy of 98%, significantly outperforming the Swin Transformer alone (91.68%) and its combination with EFRB (95.43%). The training and validation curves demonstrated rapid convergence and stable performance across epochs, while the confusion matrix confirmed high classification accuracy with minimal misclassification between crack and non-crack classes. Furthermore, the ROC curves for both positive and negative classes achieved an AUC of 1.0, indicating excellent discriminative capability.</p>
</sec>
</sec>
<sec id="sec16">
<label>4.6</label>
<title>Performance comparison with existing works</title>
<p><xref ref-type="bibr" rid="ref5">Elghaish et al. (2022)</xref> evaluated AlexNet, GoogleNet, and two others for highway crack identification and classification, and proposed a model optimized with diverse learning rates achieving 97.62% accuracy using a dataset of 4,663 crack images grouped into three categories, outperforming GoogleNet&#x2019;s 89.08% and AlexNet&#x2019;s 87.82%. <xref ref-type="bibr" rid="ref1">Ahmadi et al. (2022)</xref> proposed a comprehensive approach combining image segmentation, noise reduction, heuristic-based feature extraction, and the Hough transform with crack classification using six classifiers. The hybrid model achieved the highest accuracy at 93.86%, surpassing individual classifiers. <xref ref-type="bibr" rid="ref16">Liu et al. (2022b)</xref> employed infrared thermography and CNNs to classify asphalt pavement fatigue crack severity into four levels, using three image types. Thirteen CNN models, including EfficientNet-B4, were trained, with accuracy surpassing 0.95 across all image types, particularly on infrared images. Grad-CAM and Guided Grad-CAM analyses indicated fusion images are highly effective for reliable fatigue crack classification.</p>
<p><xref ref-type="bibr" rid="ref13">Liu et al. (2023)</xref> introduced a tunnel crack detection method using image processing with deep learning, comparing SVM and AlexNet-based models. AlexNet achieved 96.7% test accuracy, indicating deep CNN models&#x2019; superior performance for identifying structural flaws in subway tunnel. <xref ref-type="bibr" rid="ref2">Chen et al. (2023)</xref> explored the role of deep learning, specifically transfer learning, in automating the detection of building facade cracks. Addressing the need for efficient large-scale inspections, transfer learning significantly improved CNN performance, increasing accuracy from 89% to 94%, demonstrating its efficacy in image classification with limited data, aligning with national Smart Nation goals for intelligent technology in construction.</p>
<p><xref ref-type="bibr" rid="ref40">Zhang et al. (2023a)</xref> presented a lightweight broad learning system for concrete crack detection, named MobileNetV3-BLS, which overcomes the challenges of complex architectures and high computational requirements. This method improved feature extraction by integrating MobileNetV3&#x2019;s inverted residual structure as a convolutional module, employing random mapping and enhancement nodes to train the model. MobileNetV3-BLS exhibits enhanced accuracy and training speed, facilitating dynamic updates for incremental learning with new data and nodes. <xref ref-type="table" rid="tab4">Table 4</xref> provides a comparative analysis between the proposed model existing state-of-the-art architectures in crack detection.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>Comparison with state-of-the-art architectures.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Sl. No</th>
<th align="left" valign="top">Reference</th>
<th align="left" valign="top">Methodology</th>
<th align="center" valign="top">Accuracy in %</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">1</td>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref5">Elghaish et al. (2022)</xref>
</td>
<td align="left" valign="middle">Optimized CNN model with diverse learning rates vs. GoogleNet and AlexNet on 3-category crack dataset</td>
<td align="center" valign="middle">97.62</td>
</tr>
<tr>
<td align="left" valign="middle">2</td>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref1">Ahmadi et al. (2022)</xref>
</td>
<td align="left" valign="middle">Image segmentation + noise reduction + heuristic feature extraction + hybrid model (6 classifiers)</td>
<td align="center" valign="middle">93.86</td>
</tr>
<tr>
<td align="left" valign="middle">3</td>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref16">Liu et al. (2022b)</xref>
</td>
<td align="left" valign="middle">CNNs with infrared thermography and fused image types for fatigue crack severity classification</td>
<td align="center" valign="middle">&#x003E;95.00</td>
</tr>
<tr>
<td align="left" valign="middle">4</td>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref13">Liu et al. (2023)</xref>
</td>
<td align="left" valign="middle">Image processing + deep learning (SVM vs. AlexNet) for subway tunnel crack detection</td>
<td align="center" valign="middle">96.70</td>
</tr>
<tr>
<td align="left" valign="middle">5</td>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref2">Chen et al. (2023)</xref>
</td>
<td align="left" valign="middle">Transfer learning for building fa&#x00E7;ade crack detection</td>
<td align="center" valign="middle">94.00</td>
</tr>
<tr>
<td align="left" valign="middle">6</td>
<td align="left" valign="middle">
<xref ref-type="bibr" rid="ref40">Zhang et al. (2023a)</xref>
</td>
<td align="left" valign="middle">MobileNetV3-BLS: lightweight broad learning with inverted residual structure and enhancement nodes</td>
<td align="center" valign="middle">Not specified</td>
</tr>
<tr>
<td align="left" valign="middle">8</td>
<td align="left" valign="middle">Proposed</td>
<td align="left" valign="middle">Swin+EFRP+Population based Optimization</td>
<td align="center" valign="middle">98</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="sec17">
<label>5</label>
<title>Conclusion and future work</title>
<p>The proposed a crack detection framework integrating the Swin Transformer with an Enhanced Features Representation Block (EFRB) efficiently captured long-range dependencies and process complex images, while the EFRB improves spatial feature extraction and channel representation through depthwise and pointwise convolutions. Population-based feature selection optimizes the process, resulting an robust performance through effective exploration of the feature space. The proposed model achieved an accuracy of 98%, with precision, recall, and F1-scores of 0.97, 0.99, and 0.98, respectively, highlighting the model&#x2019;s robustness in detecting cracks in real-world structural images. The results demonstrate the potential of combining advanced transformers with convolutional blocks for high-precision tasks in image analysis. The proposed framework can significantly enhance the accuracy and efficiency of crack detection systems, providing a valuable tool for structural monitoring and maintenance.</p>
<sec id="sec18">
<label>5.1</label>
<title>Future work</title>
<p>While the proposed model has shown strong performance, several avenues for future research can be explored. First, improving the model&#x2019;s efficiency for real-time crack detection in large-scale datasets would be beneficial, potentially through model pruning, quantization, or more advanced techniques like knowledge distillation. Furthermore, exploring multi-modal crack detection by incorporating data from different sensors (e.g., thermal, acoustic) could improve the model&#x2019;s robustness under diverse environmental conditions. Future work could also focus on extending the framework to 3D crack detection, enabling the model to handle complex, three-dimensional structural scans.</p>
</sec>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec19">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author. <ext-link xlink:href="https://www.kaggle.com/datasets/arnavr10880/concrete-crack-images-for-classification" ext-link-type="uri">https://www.kaggle.com/datasets/arnavr10880/concrete-crack-images-for-classification</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="sec20">
<title>Author contributions</title>
<p>NA: Project administration, Visualization, Formal analysis, Resources, Validation, Methodology, Supervision, Writing &#x2013; review &#x0026; editing, Writing &#x2013; original draft, Investigation, Software, Conceptualization. LA: Visualization, Validation, Project administration, Formal analysis, Supervision, Writing &#x2013; original draft, Methodology, Software, Writing &#x2013; review &#x0026; editing, Investigation, Conceptualization, Resources.</p>
</sec>

<sec sec-type="COI-statement" id="sec22">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec23">
<title>Generative AI statement</title>
<p>The authors declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="sec24">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ahmadi</surname> <given-names>A.</given-names></name> <name><surname>Khalesi</surname> <given-names>S.</given-names></name> <name><surname>Golroo</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>An integrated machine learning model for automatic road crack detection and classification in urban areas</article-title>. <source>Int. J. Pavement Eng.</source> <volume>23</volume>, <fpage>3536</fpage>&#x2013;<lpage>3552</lpage>. doi: <pub-id pub-id-type="doi">10.1080/10298436.2021.1905808</pub-id></mixed-citation></ref>
<ref id="ref2"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Zhu</surname> <given-names>Z.</given-names></name> <name><surname>Lin</surname> <given-names>Z.</given-names></name> <name><surname>Zhou</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <article-title>Building surface crack detection using deep learning technology</article-title>. <source>Buildings</source> <volume>13</volume>:<fpage>1814</fpage>. doi: <pub-id pub-id-type="doi">10.3390/buildings13071814</pub-id></mixed-citation></ref>
<ref id="ref3"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Dai</surname> <given-names>J.</given-names></name></person-group> (<year>2024</year>). <article-title>Concrete surface crack detection algorithm based on improved YOLOv8</article-title>. <source>Sensors</source> <volume>24</volume>:<fpage>5252</fpage>. doi: <pub-id pub-id-type="doi">10.3390/s24165252</pub-id>, PMID: <pub-id pub-id-type="pmid">39204947</pub-id></mixed-citation></ref>
<ref id="ref4"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>X.</given-names></name> <name><surname>Yuan</surname> <given-names>J.</given-names></name> <name><surname>Dai</surname> <given-names>J.</given-names></name></person-group> (<year>2025</year>). <article-title>Study on lightweight bridge crack detection algorithm based on YOLO11</article-title>. <source>Sensors</source> <volume>25</volume>:<fpage>3276</fpage>. doi: <pub-id pub-id-type="doi">10.3390/s25113276</pub-id>, PMID: <pub-id pub-id-type="pmid">40968797</pub-id></mixed-citation></ref>
<ref id="ref5"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Elghaish</surname> <given-names>F.</given-names></name> <name><surname>Talebi</surname> <given-names>S.</given-names></name> <name><surname>Abdellatef</surname> <given-names>E.</given-names></name> <name><surname>Matarneh</surname> <given-names>S. T.</given-names></name> <name><surname>Hosseini</surname> <given-names>M. R.</given-names></name> <name><surname>Wu</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Developing a new deep learning CNN model to detect and classify highway cracks</article-title>. <source>J. Eng. Des. Technol.</source> <volume>20</volume>, <fpage>993</fpage>&#x2013;<lpage>1014</lpage>. doi: <pub-id pub-id-type="doi">10.1108/JEDT-04-2021-0192</pub-id></mixed-citation></ref>
<ref id="ref6"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fan</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Wei</surname> <given-names>J.</given-names></name> <name><surname>Loprencipe</surname> <given-names>G.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Automatic crack detection on road pavements using encoder-decoder architecture</article-title>. <source>Materials</source> <volume>13</volume>:<fpage>2960</fpage>. doi: <pub-id pub-id-type="doi">10.3390/ma13132960</pub-id>, PMID: <pub-id pub-id-type="pmid">32630713</pub-id></mixed-citation></ref>
<ref id="ref7"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>C.</given-names></name> <name><surname>Gao</surname> <given-names>W.</given-names></name> <name><surname>Zhou</surname> <given-names>D.</given-names></name></person-group> (<year>2024</year>). <article-title>Research on road surface crack detection based on SegNet network</article-title>. <source>J. Eng. Appl. Sci.</source> <volume>71</volume>:<fpage>54</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s44147-024-00391-0</pub-id></mixed-citation></ref>
<ref id="ref8"><mixed-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Rosing</surname> <given-names>T.</given-names></name></person-group> <article-title>Depthwise convolution is all you need for learning multiple visual domains</article-title> <conf-name>Proceedings of the AAAI Conference on Artificial Intelligence</conf-name>. <volume>33</volume>. (<year>2019</year>).</mixed-citation></ref>
<ref id="ref9"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>F.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Xie</surname> <given-names>Q.</given-names></name> <name><surname>Yu</surname> <given-names>H.</given-names></name></person-group> (<year>2024</year>). <article-title>A two-stage framework for pixel-level pavement surface crack detection</article-title>. <source>Eng. Appl. Artif. Intell.</source> <volume>133</volume>:<fpage>108312</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.engappai.2024.108312</pub-id></mixed-citation></ref>
<ref id="ref10"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>He</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Cao</surname> <given-names>S.</given-names></name> <name><surname>Du</surname> <given-names>W.</given-names></name></person-group> (<year>2024</year>). <article-title>Road surface crack detection method based on improved YOLOv5 and vehicle-mounted images</article-title>. <source>Measurement</source> <volume>229</volume>:<fpage>114443</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.measurement.2024.114443</pub-id></mixed-citation></ref>
<ref id="ref11"><mixed-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Hua</surname> <given-names>B.S.</given-names></name> <name><surname>Tran</surname> <given-names>M.K.</given-names></name> <name><surname>Yeung</surname> <given-names>S.K.</given-names></name></person-group> "<article-title>Pointwise convolutional neural networks</article-title>." <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>. (<year>2018</year>).</mixed-citation></ref>
<ref id="ref12"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Karimi</surname> <given-names>N.</given-names></name> <name><surname>Mishra</surname> <given-names>M.</given-names></name> <name><surname>Louren&#x00E7;o</surname> <given-names>P. B.</given-names></name></person-group> (<year>2024</year>). <article-title>Automated surface crack detection in historical constructions with various materials using deep learning-based YOLO network</article-title>. <source>Int. J. Architect. Herit.</source> <volume>19</volume>, <fpage>581</fpage>&#x2013;<lpage>597</lpage>. doi: <pub-id pub-id-type="doi">10.1080/15583058.2024.2376177</pub-id></mixed-citation></ref>
<ref id="ref13"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Hong</surname> <given-names>Z.</given-names></name> <name><surname>Shi</surname> <given-names>W.</given-names></name> <name><surname>Guo</surname> <given-names>X.</given-names></name></person-group> (<year>2023</year>). <article-title>Image-processing-based subway tunnel crack detection system</article-title>. <source>Sensors</source> <volume>23</volume>:<fpage>6070</fpage>. doi: <pub-id pub-id-type="doi">10.3390/s23136070</pub-id>, PMID: <pub-id pub-id-type="pmid">37447919</pub-id></mixed-citation></ref>
<ref id="ref14"><mixed-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Lin</surname> <given-names>Y.</given-names></name> <name><surname>Cao</surname> <given-names>Y.</given-names></name> <name><surname>Hu</surname> <given-names>H.</given-names></name> <name><surname>Wei</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2021</year>) <article-title>Swin Transformer: hierarchical vision transformer using shifted windows</article-title>. <conf-name>Proceedings of the IEEE/CVF International Conference on Computer Vision, Montreal, 10&#x2013;17 October 2021</conf-name>, <fpage>10012</fpage>&#x2013;<lpage>10022</lpage>.</mixed-citation></ref>
<ref id="ref15"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>F.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name></person-group> (<year>2022a</year>). <article-title>Deep learning and infrared thermography for asphalt pavement crack severity classification</article-title>. <source>Autom. Constr.</source> <volume>140</volume>:<fpage>104383</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.autcon.2022.104383</pub-id></mixed-citation></ref>
<ref id="ref16"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>F.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name></person-group> (<year>2022b</year>). <article-title>Asphalt pavement fatigue crack severity classification by infrared thermography and deep learning</article-title>. <source>Autom. Constr.</source> <volume>143</volume>:<fpage>104575</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.autcon.2022.104575</pub-id></mixed-citation></ref>
<ref id="ref17"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>F.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Al-Qadi</surname> <given-names>I. L.</given-names></name></person-group> (<year>2024</year>). <article-title>Multiple-type distress detection in asphalt concrete pavement using infrared thermography and deep learning</article-title>. <source>Autom. Constr.</source> <volume>161</volume>:<fpage>105355</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.autcon.2024.105355</pub-id></mixed-citation></ref>
<ref id="ref18"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>C.</given-names></name> <name><surname>Xu</surname> <given-names>B.</given-names></name></person-group> (<year>2023</year>). <article-title>Weakly-supervised structural surface crack detection algorithm based on class activation map and superpixel segmentation</article-title>. <source>Adv. Bridge Eng.</source> <volume>4</volume>:<fpage>27</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s43251-023-00106-0</pub-id></mixed-citation></ref>
<ref id="ref19"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Yao</surname> <given-names>J.</given-names></name> <name><surname>Lu</surname> <given-names>X.</given-names></name> <name><surname>Xie</surname> <given-names>R.</given-names></name> <name><surname>Li</surname> <given-names>L.</given-names></name></person-group> (<year>2019</year>). <article-title>Deepcrack: a deep hierarchical feature learning architecture for crack segmentation</article-title>. <source>Neurocomputing</source> <volume>338</volume>, <fpage>139</fpage>&#x2013;<lpage>153</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neucom.2019.01.036</pub-id></mixed-citation></ref>
<ref id="ref20"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>Z.</given-names></name> <name><surname>Mei</surname> <given-names>G.</given-names></name></person-group> (<year>2021</year>). <article-title>Deep learning for geological hazards analysis: data, models, applications, and opportunities</article-title>. <source>Earth-Sci. Rev.</source> <volume>223</volume>:<fpage>103858</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.earscirev.2021.103858</pub-id></mixed-citation></ref>
<ref id="ref21"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Malek</surname> <given-names>K.</given-names></name> <name><surname>Mohammadkhorasani</surname> <given-names>A.</given-names></name> <name><surname>Moreu</surname> <given-names>F.</given-names></name></person-group> (<year>2023</year>). <article-title>Methodology to integrate augmented reality and pattern recognition for crack detection</article-title>. <source>Comput. Aided Civ. Inf. Eng.</source> <volume>38</volume>, <fpage>1000</fpage>&#x2013;<lpage>1019</lpage>. doi: <pub-id pub-id-type="doi">10.1111/mice.12932</pub-id></mixed-citation></ref>
<ref id="ref22"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mirjalili</surname> <given-names>S.</given-names></name></person-group> (<year>2016</year>). <article-title>SCA: a sine cosine algorithm for solving optimization problems</article-title>. <source>Knowl. Based Syst.</source> <volume>96</volume>, 120&#x2013;133. doi: <pub-id pub-id-type="doi">10.1016/j.knosys.2015.12.022</pub-id></mixed-citation></ref>
<ref id="ref23"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nguyen</surname> <given-names>N. H. T.</given-names></name> <name><surname>Perry</surname> <given-names>S.</given-names></name> <name><surname>Bone</surname> <given-names>D.</given-names></name> <name><surname>Le</surname> <given-names>H. T.</given-names></name> <name><surname>Nguyen</surname> <given-names>T. T.</given-names></name></person-group> (<year>2021</year>). <article-title>Two-stage convolutional neural network for road crack detection and segmentation</article-title>. <source>Expert Syst. Appl.</source> <volume>186</volume>:<fpage>115718</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.eswa.2021.115718</pub-id></mixed-citation></ref>
<ref id="ref24"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nyathi</surname> <given-names>M. A.</given-names></name> <name><surname>Bai</surname> <given-names>J.</given-names></name> <name><surname>Wilson</surname> <given-names>I. D.</given-names></name></person-group> (<year>2023</year>). <article-title>Concrete crack width measurement using a laser beam and image processing algorithms</article-title>. <source>Appl. Sci.</source> <volume>13</volume>:<fpage>4981</fpage>. doi: <pub-id pub-id-type="doi">10.3390/app13084981</pub-id></mixed-citation></ref>
<ref id="ref25"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Oliveira</surname> <given-names>H.</given-names></name> <name><surname>Correia</surname> <given-names>P. L.</given-names></name></person-group> (<year>2014</year>). &#x201C;<article-title>CrackIT&#x2014;an image processing toolbox for crack detection and characterization</article-title>&#x201D; in <source>2014 IEEE international conference on image processing (ICIP)</source> (<publisher-name>IEEE</publisher-name>).</mixed-citation></ref>
<ref id="ref26"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Oloufa</surname> <given-names>A. A.</given-names></name> <name><surname>Mahgoub</surname> <given-names>H. S.</given-names></name> <name><surname>Ali</surname> <given-names>H.</given-names></name></person-group> (<year>2004</year>). <article-title>Infrared thermography for asphalt crack imaging and automated detection</article-title>. <source>Transp. Res. Rec.</source> <volume>1889</volume>, <fpage>126</fpage>&#x2013;<lpage>133</lpage>. doi: <pub-id pub-id-type="doi">10.3141/1889-14</pub-id></mixed-citation></ref>
<ref id="ref27"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>&#x00D6;zgenel</surname> <given-names>&#x00C7;. F.</given-names></name></person-group> (<year>2019</year>). <source>Concrete crack images for classification</source>: <publisher-name>Mendeley Data, version 2</publisher-name>.</mixed-citation></ref>
<ref id="ref28"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>&#x00D6;zgenel</surname> <given-names>&#x00C7;. F.</given-names></name> <name><surname>Sorgu&#x00E7;</surname> <given-names>A. G.</given-names></name></person-group> (<year>2018</year>). <source>Performance comparison of pretrained convolutional neural networks on crack detection in buildings</source>. <publisher-loc>Berlin</publisher-loc>: <publisher-name>ISARC</publisher-name>.</mixed-citation></ref>
<ref id="ref29"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pham</surname> <given-names>M.-V.</given-names></name> <name><surname>Ha</surname> <given-names>Y.-S.</given-names></name> <name><surname>Kim</surname> <given-names>Y.-T.</given-names></name></person-group> (<year>2023</year>). <article-title>Automatic detection and measurement of ground crack propagation using deep learning networks and an image processing technique</article-title>. <source>Measurement</source> <volume>215</volume>:<fpage>112832</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.measurement.2023.112832</pub-id></mixed-citation></ref>
<ref id="ref30"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Raghaw</surname> <given-names>C. S.</given-names></name> <name><surname>Sharma</surname> <given-names>A.</given-names></name> <name><surname>Bansal</surname> <given-names>S.</given-names></name> <name><surname>Rehman</surname> <given-names>M. Z. U.</given-names></name> <name><surname>Kumar</surname> <given-names>N.</given-names></name></person-group> (<year>2024</year>). <article-title>Cotconet: an optimized coupled transformer-convolutional network with an adaptive graph reconstruction for leukemia detection</article-title>. <source>Comput. Biol. Med.</source> <volume>179</volume>:<fpage>108821</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compbiomed.2024.108821</pub-id>, PMID: <pub-id pub-id-type="pmid">38972153</pub-id></mixed-citation></ref>
<ref id="ref31"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shaoze</surname> <given-names>H.</given-names></name> <name><surname>Qi</surname> <given-names>L.</given-names></name> <name><surname>Chao</surname> <given-names>C.</given-names></name> <name><surname>Yuhang</surname> <given-names>C.</given-names></name></person-group> (<year>2025</year>). <article-title>A real-time concrete crack detection and segmentation model based on YOLOv11</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2508.11517</pub-id></mixed-citation></ref>
<ref id="ref32"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>J.</given-names></name> <name><surname>Feng</surname> <given-names>A.</given-names></name> <name><surname>Korkhov</surname> <given-names>V.</given-names></name> <name><surname>Pu</surname> <given-names>Y.</given-names></name></person-group> (<year>2024</year>). <article-title>Enhancing road crack detection accuracy with BsS-YOLO: optimizing feature fusion and attention mechanisms</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2412.10902</pub-id></mixed-citation></ref>
<ref id="ref33"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tran</surname> <given-names>T. S.</given-names></name> <name><surname>Nguyen</surname> <given-names>S. D.</given-names></name> <name><surname>Lee</surname> <given-names>H. J.</given-names></name> <name><surname>Tran</surname> <given-names>V. P.</given-names></name></person-group> (<year>2023</year>). <article-title>Advanced crack detection and segmentation on bridge decks using deep learning</article-title>. <source>Constr. Build. Mater.</source> <volume>400</volume>:<fpage>132839</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.conbuildmat.2023.132839</pub-id></mixed-citation></ref>
<ref id="ref34"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vivekananthan</surname> <given-names>V.</given-names></name> <name><surname>Vignesh</surname> <given-names>R.</given-names></name> <name><surname>Vasanthaseelan</surname> <given-names>S.</given-names></name> <name><surname>Joel</surname> <given-names>E.</given-names></name> <name><surname>Kumar</surname> <given-names>K. S.</given-names></name></person-group> (<year>2023</year>). <article-title>Concrete bridge crack detection by image processing technique by using the improved OTSU method</article-title>. <source>Mater Today Proc</source> <volume>74</volume>, <fpage>1002</fpage>&#x2013;<lpage>1007</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.matpr.2022.11.356</pub-id></mixed-citation></ref>
<ref id="ref35"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>G.</given-names></name> <name><surname>Yue</surname> <given-names>Q.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name></person-group> (<year>2023</year>). <article-title>Deep learning algorithm for real-time automatic crack detection, segmentation, qualification</article-title>. <source>Eng. Appl. Artif. Intell.</source> <volume>126</volume>:<fpage>107085</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.engappai.2023.107085</pub-id></mixed-citation></ref>
<ref id="ref36"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yamaguchi</surname> <given-names>T.</given-names></name> <name><surname>Hashimoto</surname> <given-names>S.</given-names></name></person-group> (<year>2010</year>). <article-title>Fast crack detection method for large-size concrete surface images using percolation-based image processing</article-title>. <source>Mach. Vis. Appl.</source> <volume>21</volume>, <fpage>797</fpage>&#x2013;<lpage>809</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s00138-009-0189-8</pub-id></mixed-citation></ref>
<ref id="ref37"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yin</surname> <given-names>D.</given-names></name> <name><surname>Zhang</surname> <given-names>B.</given-names></name> <name><surname>Yan</surname> <given-names>J.</given-names></name> <name><surname>Luo</surname> <given-names>Y.</given-names></name> <name><surname>Zhou</surname> <given-names>T.</given-names></name> <name><surname>Qin</surname> <given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>CoWNet: a correlation weighted network for geological hazard detection</article-title>. <source>Knowl. Based Syst.</source> <volume>275</volume>:<fpage>110684</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.knosys.2023.110684</pub-id></mixed-citation></ref>
<ref id="ref38"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>H.</given-names></name> <name><surname>Deng</surname> <given-names>Y.</given-names></name> <name><surname>Guo</surname> <given-names>F.</given-names></name></person-group> (<year>2024</year>). <article-title>Real-time pavement surface crack detection based on lightweight semantic segmentation model</article-title>. <source>Transp. Geotech.</source> <volume>48</volume>:<fpage>101335</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.trgeo.2024.101335</pub-id></mixed-citation></ref>
<ref id="ref39"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zadeh</surname> <given-names>S. S.</given-names></name> <name><surname>Khorshidi</surname> <given-names>M.</given-names></name> <name><surname>Kooban</surname> <given-names>F.</given-names></name></person-group> (<year>2024</year>). <article-title>Concrete surface crack detection with convolutional-based deep learning models</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.5281/zenodo.10061654</pub-id></mixed-citation></ref>
<ref id="ref40"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Cai</surname> <given-names>Y.-Y.</given-names></name> <name><surname>Yang</surname> <given-names>D.</given-names></name> <name><surname>Yuan</surname> <given-names>Y.</given-names></name> <name><surname>He</surname> <given-names>W.-Y.</given-names></name> <name><surname>Wang</surname> <given-names>Y.-J.</given-names></name></person-group> (<year>2023a</year>). <article-title>Mobilenetv3-BLS: a broad learning approach for automatic concrete surface crack detection</article-title>. <source>Constr. Build. Mater.</source> <volume>392</volume>:<fpage>131941</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.conbuildmat.2023.131941</pub-id></mixed-citation></ref>
<ref id="ref42"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Qian</surname> <given-names>S.</given-names></name> <name><surname>Tan</surname> <given-names>C.</given-names></name></person-group> (<year>2023b</year>). <article-title>Automated bridge crack detection method based on lightweight vision models</article-title>. <source>Complex Intell. Syst.</source> <volume>9</volume>, <fpage>1639</fpage>&#x2013;<lpage>1652</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s40747-022-00876-6</pub-id></mixed-citation></ref>
<ref id="ref43"><mixed-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Yang</surname> <given-names>F.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.D.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.J.</given-names></name></person-group> (<year>2016</year>). <article-title>Road crack detection using deep convolutional neural network</article-title>. In <conf-name>2016 IEEE International Conference on Image Processing (ICIP)</conf-name>.</mixed-citation></ref>
<ref id="ref44"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>M.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Guo</surname> <given-names>B.</given-names></name> <name><surname>Gu</surname> <given-names>W.</given-names></name></person-group> (<year>2025</year>). <article-title>Review of crack depth detection technology for engineering structures: from physical principles to artificial intelligence</article-title>. <source>Appl. Sci.</source> <volume>15</volume>:<fpage>9120</fpage>. doi: <pub-id pub-id-type="doi">10.3390/app15169120</pub-id></mixed-citation></ref>
<ref id="ref45"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>R.</given-names></name> <name><surname>Zheng</surname> <given-names>K.</given-names></name> <name><surname>Wei</surname> <given-names>X.</given-names></name> <name><surname>Jia</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>State-of-the-art and annual progress of bridge engineering in 2021</article-title>. <source>Adv. Bridge Eng.</source> <volume>3</volume>:<fpage>29</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s43251-022-00070-1</pub-id></mixed-citation></ref>
<ref id="ref46"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>J.</given-names></name> <name><surname>Sheng</surname> <given-names>J.</given-names></name> <name><surname>Cai</surname> <given-names>Q.</given-names></name></person-group> (<year>2025</year>). <article-title>FD<sup>2</sup>-YOLO: a frequency-domain dual-stream network based on YOLO for crack detection</article-title>. <source>Sensors</source> <volume>25</volume>:<fpage>3427</fpage>. doi: <pub-id pub-id-type="doi">10.3390/s25113427</pub-id>, PMID: <pub-id pub-id-type="pmid">40968931</pub-id></mixed-citation></ref>
</ref-list>
<fn-group><fn id="fn0001" fn-type="custom" custom-type="edited-by"><p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2014753/overview">Seyed Jalaleddin Mousavirad</ext-link>, Mid Sweden University, Sweden</p></fn>
<fn id="fn0002" fn-type="custom" custom-type="reviewed-by"><p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2670366/overview">Masoud Salar</ext-link>, Chabahar Maritime University, Iran</p><p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3135369/overview">Xin Bi</ext-link>, Northeastern University, China</p></fn></fn-group>
</back>
</article>