<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1632052</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>DUNet: a novel dehazing model based on outdoor images</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zhao</surname>
<given-names>Wei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2885586/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Qiusheng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2890238/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Mingliang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2625265/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ye</surname>
<given-names>Guanshi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Zichen</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Qi</surname>
<given-names>Mingyang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2957290/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Yu</surname>
<given-names>Helong</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Tang</surname>
<given-names>You</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2625340/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Electrical and Information Engineering, Jilin Agricultural Science and Technology University</institution>, <addr-line>Jilin</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>College of Underwater Acoustic Engineering, Harbin Engineering University</institution>, <addr-line>Harbin</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>College of Information Technology, Jilin Agricultural University</institution>, <addr-line>Changchun</addr-line>,&#xa0;<country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1348206/overview">Aichen Wang</ext-link>, Jiangsu University, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1406117/overview">Baohua Zhang</ext-link>, Nanjing Agricultural University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1545352/overview">Chung-Huang Yeh</ext-link>, National Central University, Taiwan</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3074161/overview">Chenyang Li</ext-link>, Xidian University, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Mingyang Qi, <email xlink:href="mailto:qimingyang@jlnku.edu.cn">qimingyang@jlnku.edu.cn</email>; Helong Yu, <email xlink:href="mailto:yuhelong@jlau.edu.cn">yuhelong@jlau.edu.cn</email>; You Tang, <email xlink:href="mailto:tangyou9000@163.com">tangyou9000@163.com</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>07</day>
<month>10</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1632052</elocation-id>
<history>
<date date-type="received">
<day>20</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>16</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Zhao, Zhang, Li, Ye, Liu, Qi, Yu and Tang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Zhao, Zhang, Li, Ye, Liu, Qi, Yu and Tang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Image dehazing technology is widely utilized in outdoor environments, especially in precision agriculture, where it enhances image quality and monitoring accuracy. However, conventional dehazing methods have exhibited limited performance in complex outdoor conditions, necessitating the development of more advanced models to address these challenges. This paper proposes DUNet, a high-performance image dehazing model that is well-suited for outdoor smart agriculture applications. In this study, we first introduce a novel hybrid convolution block, MixConv, designed to fully extract detailed feature information from images. Secondly, by incorporating the atmospheric scattering model, we propose a dehazing feature extraction unit, DFEU, integrated between the encoder and decoder, to establish a mapping relationship between hazy and haze-free images in the feature space. Finally, the SK fusion mechanism dynamically fuses feature maps extracted from multiple paths. To evaluate the dehazing performance of DUNet, we constructed a dataset consisting of 1,978 pairs of hazy UAV images of paddy fields. DUNet achieved a PSNR of 36.0206 and an SSIM of 0.9946 on this dataset. We further validated DUNet&#x2019;s performance on a remote sensing dataset, achieving a PSNR of 37.2887 and an SSIM of 0.9933. Experimental results demonstrate that, compared to other well-established image dehazing models, DUNet offers superior performance, confirming its potential and feasibility for outdoor smart agriculture dehazing tasks.</p>
</abstract>
<kwd-group>
<kwd>image dehazing</kwd>
<kwd>deep learning</kwd>
<kwd>image processing</kwd>
<kwd>smart agriculture</kwd>
<kwd>outdoor images</kwd>
</kwd-group>
<counts>
<fig-count count="9"/>
<table-count count="3"/>
<equation-count count="23"/>
<ref-count count="53"/>
<page-count count="17"/>
<word-count count="9299"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Sustainable and Intelligent Phytoprotection</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>With the continuous advancement of social and technological progress, smart agriculture has emerged as a critical direction for modern agricultural development, rapidly gaining popularity and widespread application. In modern smart agriculture, drone technology serves as an essential tool for precision farming, widely employed in areas such as farmland monitoring, crop growth evaluation, pest detection, and soil moisture analysis (<xref ref-type="bibr" rid="B15">Guo et&#xa0;al., 2025</xref>; <xref ref-type="bibr" rid="B37">Su et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B50">Zhao et&#xa0;al., 2025</xref>). UAVs, equipped with high-resolution cameras, infrared sensors, and multispectral imaging devices, can efficiently cover extensive agricultural areas and capture real-time images of farmland, thereby providing precise decision support for agricultural management (<xref ref-type="bibr" rid="B43">Yang et&#xa0;al., 2025</xref>). By integrating deep learning and computer vision technologies, drone systems can not only assess crop growth conditions with high accuracy but also detect issues such as pests, water scarcity, or nutrient deficiencies in a timely manner, thereby significantly enhancing the intelligence and automation of farmland management (<xref ref-type="bibr" rid="B41">Wen et&#xa0;al., 2024</xref>). The widespread adoption of this technology has transformed agricultural production from traditional, experience-based management to data-driven, precision-based models, enhancing crop yields, reducing production costs, and minimizing pesticide use, thereby promoting sustainable agricultural development. However, drones still face several challenges in practical applications, particularly under complex weather conditions such as haze, which degrades image quality. Specifically, haze conditions result in issues such as insufficient contrast and blurred details in images captured by drones, hindering the accurate assessment of crop health and timely identification of pests and diseases. This imposes substantial limitations on the effectiveness of drone applications, particularly in large-scale farmland monitoring, where low-quality images can lead to decreased recognition accuracy and even undermine the intelligence of agricultural production. Therefore, it is essential to restore or enhance the captured blurred images to ensure the usability and reliability of the data, facilitating the smooth execution of subsequent detection and identification tasks. This will not only improve the accuracy and system stability of drone-based farmland monitoring but also play a crucial role in advancing the intelligent development of agriculture (<xref ref-type="bibr" rid="B22">Joshi et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B46">Yu et&#xa0;al., 2024b</xref>).</p>
<p>Image dehazing technology is a crucial task in computer vision, aimed at restoring hazy images to clear and visible ones. The presence of haze or other factors often leads to the loss of image details and reduced contrast, which in turn affects subsequent image analysis and processing. Dehazing technology can effectively mitigate or eliminate these degradations, restoring image clarity and detail and making the image more vivid and informative. This not only enhances the visual experience but also provides more accurate data for subsequent advanced visual tasks, such as object detection and target recognition, thereby aiding in the extraction of more valuable information from images (<xref ref-type="bibr" rid="B14">Goyal et&#xa0;al., 2024</xref>).In the early stages of image dehazing research, the physical processes behind haze formation were not yet understood. Most methods relied on image enhancement techniques to achieve deblurring, such as histogram equalization (<xref ref-type="bibr" rid="B8">Dale-Jones and Tjahjadi, 1993</xref>; <xref ref-type="bibr" rid="B23">Kim et&#xa0;al., 2001</xref>; <xref ref-type="bibr" rid="B53">Zhu et&#xa0;al., 2013</xref>), color correctio (<xref ref-type="bibr" rid="B19">Huang et&#xa0;al., 2014</xref>), and others. Following the introduction of the atmospheric scattering model (<xref ref-type="bibr" rid="B31">Nayar and Narasimhan, 1999</xref>), researchers recognized that hazy images result from the degradation of clear images and began developing dehazing algorithms based on this model. The atmospheric scattering model is represented by <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>J</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Here, <inline-formula>
<mml:math display="inline" id="im1">
<mml:mi>I</mml:mi>
</mml:math>
</inline-formula> is the hazy image, <inline-formula>
<mml:math display="inline" id="im2">
<mml:mi>J</mml:mi>
</mml:math>
</inline-formula> is the clear image, <inline-formula>
<mml:math display="inline" id="im3">
<mml:mi>A</mml:mi>
</mml:math>
</inline-formula> is the global atmospheric light, <inline-formula>
<mml:math display="inline" id="im4">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula> is the transmission map, and <inline-formula>
<mml:math display="inline" id="im5">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula> is the pixel index. When the global atmospheric light <inline-formula>
<mml:math display="inline" id="im6">
<mml:mi>A</mml:mi>
</mml:math>
</inline-formula> is uniform, <inline-formula>
<mml:math display="inline" id="im7">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula> is expressed as <xref ref-type="disp-formula" rid="eq2">Equation 2</xref>:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im8">
<mml:mi>&#x3b2;</mml:mi>
</mml:math>
</inline-formula> is the scattering coefficient of the atmosphere and <inline-formula>
<mml:math display="inline" id="im9">
<mml:mi>d</mml:mi>
</mml:math>
</inline-formula> is the depth information.</p>
<p>Early image dehazing methods based on the atmospheric scattering model, predominantly relied on prior knowledge. In 2009, He et&#xa0;al. proposed the classic dark channel prior (DCP) algorithm (<xref ref-type="bibr" rid="B18">He et&#xa0;al., 2009</xref>), which posits that most local regions in haze-free outdoor images contain pixels with very low intensity in at least one color channel. By integrating the atmospheric scattering model, DCP can generate effective depth maps and restore clear images. However, the DCP method has limitations, primarily due to its reliance on statistical priors, which may not be suitable for images where the atmospheric light closely resembles the scene objects. In 2015, Zhu et&#xa0;al. proposed the color attenuation prior (CAP) (<xref ref-type="bibr" rid="B52">Zhu et&#xa0;al., 2015</xref>), which models the scene depth in hazy images using a linear model, recovers depth information through supervised learning methods, and then estimates the transmission rate using the atmospheric scattering model to obtain the dehazed image. However, due to the limited learning capacity of the model, many prior methods still exhibit shortcomings in representation and accuracy.</p>
<p>With the rapid development of deep learning technologies, many end-to-end convolutional neural networks (CNNs) have been employed by researchers for image dehazing tasks (<xref ref-type="bibr" rid="B21">Jackson et&#xa0;al., 2024</xref>). CNNs reduce the reliance on manually designed priors by automatically learning useful image features. For example, Cai et&#xa0;al. proposed DehazeNet (<xref ref-type="bibr" rid="B3">Cai et&#xa0;al., 2016</xref>), which takes hazy images as input, outputs the medium transmission map, and then uses the atmospheric scattering model to recover the clear image. Li et&#xa0;al. proposed AOD-Net (<xref ref-type="bibr" rid="B25">Li et&#xa0;al., 2017</xref>), which is based on a redesigned atmospheric scattering model and directly generates clear images through CNNs, eliminating the need to estimate the transmission matrix and atmospheric light. Song et&#xa0;al. proposed a compact dehazing model, gUNet (<xref ref-type="bibr" rid="B36">Song et&#xa0;al., 2022</xref>), which introduces minimal modifications to U-Net (<xref ref-type="bibr" rid="B34">Ronneberger et&#xa0;al., 2015</xref>) and incorporates residual blocks with a gating mechanism. This not only reduces the model&#x2019;s parameter count effectively but also yields good dehazing results. The DCPDN dehazing model (<xref ref-type="bibr" rid="B49">Zhang and Patel, 2018</xref>) proposed by Zhang et&#xa0;al. integrates the atmospheric scattering model into the network to optimize the learning of the transmission map, atmospheric light, and dehazed image, and introduces Generative Adversarial Networks (<xref ref-type="bibr" rid="B13">Goodfellow et&#xa0;al., 2014</xref>) to enhance details, significantly improving dehazing performance. Chen et&#xa0;al. proposed the end-to-end gated context aggregation dehazing network GCANet (<xref ref-type="bibr" rid="B4">Chen et&#xa0;al., 2019</xref>), which introduces a smooth dilation technique to eliminate grid artifacts and utilizes a gating network to fuse features across different levels, achieving better dehazing results. Although these methods have advanced image dehazing technology, they still face challenges in handling extreme weather conditions, low-quality images, and real-time scenes. Nevertheless, the introduction of end-to-end networks has undeniably accelerated the progress of image dehazing research.</p>
<p>In recent years, researchers have further enhanced the dehazing performance of models by incorporating attention mechanisms. For example, Xu et&#xa0;al. proposed a novel feature attention (FA) mechanism that combines channel and pixel attention, which was applied to CNNs to expand the network&#x2019;s representational capacity. Through local residual learning, the FFA-Net (<xref ref-type="bibr" rid="B32">Qin et&#xa0;al., 2020</xref>) network can better focus on learning effective information. Chen et&#xa0;al. proposed the DEA-Net (<xref ref-type="bibr" rid="B5">Chen et&#xa0;al., 2023</xref>), which is based on detail-enhancing convolution and content-guided attention. This model introduces differential convolution to enhance representational capacity and integrates three attention mechanisms to comprehensively extract image detail features. With the widespread application of Transformers (<xref ref-type="bibr" rid="B40">Vaswani et&#xa0;al., 2017</xref>) in computer vision, and inspired by the Vision Transformer (<xref ref-type="bibr" rid="B11">Dosovitskiy et&#xa0;al., 2020</xref>), Song et&#xa0;al. proposed DehazeFormer (<xref ref-type="bibr" rid="B35">Song et&#xa0;al., 2023</xref>). This model employs a shifted window partitioning scheme with reflective padding and integrates a convolutional spatial information aggregation scheme parallel to attention. Experimental results demonstrate its excellent performance in dehazing tasks. Guo et&#xa0;al. combined Transformers with CNNs to propose the Dehamer (<xref ref-type="bibr" rid="B17">Guo et&#xa0;al., 2022</xref>) model, which retains the advantages of Transformer in global context modeling while preserving CNN&#x2019;s capability in local representations, thereby significantly improving dehazing performance. However, some color bias persists in its color restoration, causing the dehazed images to differ in color from the clear images.</p>
<p>Although end-to-end networks have improved performance within physical model constraints, they are often confined to the original image space and fail to fully utilize the physical information in the feature space. Therefore, Dong et&#xa0;al. proposed a physics-based dehazing network, PFDN (<xref ref-type="bibr" rid="B9">Dong and Pan, 2020</xref>), which explicitly utilizes the physical model in the feature space by introducing a key component, the ASM-based Feature Dehazing Unit (FDU), learning the required useful information to achieve more effective dehazing. However, the FDU overlooks the fact that atmospheric light and transmission maps are not always uniform, and their features cannot be approximated similarly. To accurately implement the physical model in the deep network feature space, Zheng et&#xa0;al. proposed a physically-aware dual-branch unit (PDU) (<xref ref-type="bibr" rid="B51">Zheng et&#xa0;al., 2023</xref>), which separately captures features corresponding to atmospheric light and transmission maps in two branches, considering the physical properties of each factor. This allows for more precise synthesis of potential clear image features based on the physical model and facilitates information transfer and feature extraction in the feature space.</p>
<p>In recent years, dehazing research based on deep learning has garnered increasing attention. Li et&#xa0;al. proposed an efficient dehazing method applicable to both outdoor and remote sensing images, which integrates the strengths of image enhancement and image restoration techniques (<xref ref-type="bibr" rid="B27">Li et&#xa0;al., 2023</xref>). Experimental results on both synthetic and real-world datasets demonstrated that this method outperformed existing approaches. After that, Li et&#xa0;al. further introduced UAVD-Net, a novel dehazing framework tailored for drone-based remote sensing images affected by spatially varying haze (<xref ref-type="bibr" rid="B28">Li et&#xa0;al., 2025</xref>). UAVD-Net employs both global and local feature extraction mechanisms to effectively eliminate non-uniform haze across spatial regions, consistently achieving superior performance compared to state-of-the-art methods on diverse datasets. Similarly, Cui et&#xa0;al. proposed an image dehazing network called EENet, which aims to achieve image dehazing through enhanced spatial-spectral learning (<xref ref-type="bibr" rid="B7">Cui et&#xa0;al., 2025</xref>). This method works through the coordinated efforts of three modules: frequency processing, spatial processing, and dual-domain interaction. Based on modelling global dependencies and multi-scale features, it achieves information fusion between the frequency domain and the spatial domain to improve image dehazing effects, and has achieved state-of-the-art performance on synthetic and real-world image dehazing datasets. However, due to the domain gap between synthetic and real images, models trained solely on synthetic data often lack generalization in real-world scenarios. To overcome this limitation, Su et&#xa0;al. proposed DNMGDT, a dehazing network that integrates multi-prior guidance with domain transfer mechanisms (<xref ref-type="bibr" rid="B38">Su et&#xa0;al., 2025</xref>). By leveraging pseudo-label supervision, adaptive weighting, and physically guided domain transfer strategies, DNMGDT significantly improves performance on real-world hazy images. Collectively, these deep learning&#x2013;based dehazing approaches offer valuable insights and advancements for the field of image restoration.</p>
<p>Smart agriculture plays a vital role in modern society, and farmland monitoring, as a key component, significantly contributes to its advancement through efficient and intelligent management. In farmland monitoring, the images collected are often influenced by the complexity of outdoor weather conditions, such as fog, causing drones to capture blurry images with visible haze. This impacts subsequent evaluation and recognition tasks, making it challenging to accurately identify and analyze targets. Hazy images not only reduce the accuracy of drone-based farmland monitoring systems but may also adversely impact agricultural decision-making, thereby affecting the responsiveness and efficiency of farmland management (<xref ref-type="bibr" rid="B33">Qiu et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B48">Zhang et&#xa0;al., 2022</xref>). To address this issue, image dehazing technology plays a crucial role in smart agriculture by effectively enhancing the clarity and detail of images, thus providing reliable visual support for tasks such as crop monitoring and pest detection. However, existing image dehazing algorithms continue to suffer from poor performance, primarily due to their inability to directly establish the relationship between hazy and clear images in the feature space, leading to insufficient utilization of physical image information and poor restoration quality. To address this, we propose a novel image dehazing model, DUNet, based on the atmospheric scattering model, designed to fully extract dehazing features and effectively restore visual information affected by haze and other environmental factors. Specifically, we utilize the classic U-Net architecture with residual connections as the backbone to extract multi-scale information. Next, we employ the hybrid convolution module, MixConv, which incorporates depthwise separable convolution and multi-scale gated convolution, to thoroughly extract detailed feature information. Furthermore, we integrate a dehazing feature extraction unit based on the atmospheric scattering model into the network, which predicts atmospheric light and transmission maps through dual paths, establishing the relationship between hazy and clear images in the feature space. Finally, we utilize the SK fusion module (<xref ref-type="bibr" rid="B36">Song et&#xa0;al., 2022</xref>) to dynamically merge the feature maps extracted from different paths. Our main contributions are as follows:</p>
<list list-type="order">
<list-item>
<p>Based on real UAV-collected rice field image data, a haze-affected paddy field image dataset was synthesized using the atmospheric scattering model.</p>
</list-item>
<list-item>
<p>A new end-to-end dehazing model, DUNet, for smart agriculture is proposed based on the atmospheric scattering model.</p>
</list-item>
<list-item>
<p>A hybrid convolution module, MixConv, containing depthwise separable convolution and multi-scale gated convolution, is proposed to enhance the model&#x2019;s ability to extract multi-scale information.</p>
</list-item>
<list-item>
<p>A dehazing feature extraction unit (DFEU) is proposed to establish the relationship between hazy and clear images in the feature space.</p>
</list-item>
<list-item>
<p>Experimental results show that DUNet performs well in dehazing tasks on the remote sensing haze dataset and rice field haze image dataset, demonstrating good robustness and providing a new strategy for image dehazing.</p>
</list-item>
</list>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Datasets</title>
<p>This study employed two datasets, one of which is the publicly available remote sensing dataset, RSHaze (<xref ref-type="bibr" rid="B29">Lihe et&#xa0;al., 2024</xref>). Due to the highly uneven distribution of haze in remote sensing images, haze removal is commonly considered a classic non-uniform image dehazing problem. Therefore, the dehazing performance of the proposed model was evaluated on the remote sensing dataset. The RSHaze dataset comprises 1330 pairs of remote sensing images, each resized to 512x512 pixels. As per the official split, 1000 pairs are designated for training, and the remaining 330 pairs are allocated for testing. The second dataset is derived from two paddy field datasets, URC (<xref ref-type="bibr" rid="B1">Bai et&#xa0;al., 2023</xref>) and DPRD (<xref ref-type="bibr" rid="B44">Ye et&#xa0;al., 2024</xref>). The images are cropped to 512x512 pixels, resulting in a total of 1978 clear paddy field images. According to <xref ref-type="disp-formula" rid="eq1">Equations 1</xref>, <xref ref-type="disp-formula" rid="eq2">2</xref>, after accurately estimating the depth information of the image, the blurred image can be synthesized using the atmospheric scattering model. Thus, we first estimate the depth information of the clear paddy field images using Monodepth2 (<xref ref-type="bibr" rid="B12">Godard et&#xa0;al., 2018</xref>). Next, following <xref ref-type="disp-formula" rid="eq2">Equation 2</xref>, the scattering coefficient is set to 2.0 to compute the transmission map. Finally, based on <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>, the atmospheric light is set to 170 to obtain the synthesized blurred image. Ultimately, we constructed a haze image dataset for paddy fields, named Paddydata. Paddydata consists of 1978 pairs of paddy field images, each resized to 512x512 pixels. The dataset is randomly split into training, validation, and test sets with a 7:1:2 ratio: 1385 pairs for training, 197 pairs for validation, and 396 pairs for testing. The paired images from the two datasets are shown in <xref ref-type="fig" rid="f1">
<bold>Figures&#xa0;1A, B</bold>
</xref> are randomly selected images with varying haze concentrations from RSHaze, while <xref ref-type="fig" rid="f1">
<bold>Figures&#xa0;1C, D</bold>
</xref> are paired images from different paddy fields. During training, to increase data diversity, we applied five data augmentation techniques: random cropping, random horizontal flipping, random rotation, aligned cropping, and pixel normalization. Specifically, the images are randomly cropped to 256x256 pixels, with a 50% chance of horizontal flipping. The rotation angles are randomly chosen from 0&#xb0;, 90&#xb0;, 180&#xb0;, or 270&#xb0;. Center cropping is applied to align the image size, and pixel values are normalized to the range [-1,1]. These data augmentation techniques effectively enhanced data diversity, mitigated overfitting to specific sample features, and improved the robustness and generalization ability of the model. The dataset is publicly available at (<ext-link ext-link-type="uri" xlink:href="https://github.com/MaiheZHao/data">https://github.com/MaiheZHao/data</ext-link>).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Images of RSHaze and Paddydata datasets. <bold>(A)</bold> and <bold>(B)</bold> depict images of varying haze concentrations from RSHaze, while <bold>(C)</bold> and <bold>(D)</bold> show haze images of different rice paddies from Paddydata.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632052-g001.tif">
<alt-text content-type="machine-generated">Satellite images show two groups labeled RSHaze and Paddydata. Group A displays a forested area with a clear and a hazy version. Group B depicts an urban area with the same clear and hazy effect. Group C shows rice plants in a field in both clear and hazy conditions, while Group D displays rows of another crop under similar conditions.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Network architecture</title>
<p>The DUNet model utilizes the MixConv module for feature extraction at every stage. A dehazing feature extraction unit (DFEU), based on the atmospheric scattering model, is inserted between the encoder and decoder to extract fog-free features. The decoder employs the SK fusion module to dynamically combine feature representations from multiple paths. The overall architecture of DUNet is depicted in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>. The blurry image <inline-formula>
<mml:math display="inline" id="im10">
<mml:mtext>I</mml:mtext>
</mml:math>
</inline-formula> first passes through the MixConv module and a downsampling encoder to extract multi-scale blurry image features <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> within the feature space, where <inline-formula>
<mml:math display="inline" id="im12">
<mml:mtext>i</mml:mtext>
</mml:math>
</inline-formula> is 1, 2, 3, or 4. Subsequently, the dehazing feature extraction unit extracts fog-free features <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>J</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im14">
<mml:mtext>i</mml:mtext>
</mml:math>
</inline-formula> is 1, 2, 3, or 4. Subsequently, the SK fusion module combines the feature maps extracted by the encoder&#x2019;s downsampling and those upsampled and restored by the decoder. The MixConv module decodes the fog-free features into a final dehazed image. Finally, a global residual operation is applied to the blurry image <inline-formula>
<mml:math display="inline" id="im15">
<mml:mtext>I</mml:mtext>
</mml:math>
</inline-formula> to produce the final dehazed image <inline-formula>
<mml:math display="inline" id="im16">
<mml:mtext>J</mml:mtext>
</mml:math>
</inline-formula>.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Overall structure of DUNet.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632052-g002.tif">
<alt-text content-type="machine-generated">Flowchart of a neural network architecture showing processing stages. Input \(I\) with dimensions \(H \times W \times C\) passes through a 3x3 convolution. It undergoes multiple MixConv blocks, each paired with DFEU and SK Fusion layers. Dimensions are halved at each stage: \(H/2 \times W/2 \times 2C\), \(H/4 \times W/4 \times 4C\), and \(H/8 \times W/8 \times 8C\). Arrows indicate skip connections (blue), downsampling (green), and upsampling (red), leading to output \(J\) with dimensions \(H \times W \times C\).</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>MixConv block</title>
<p>The MixConv Block primarily utilizes depthwise separable convolutions (<xref ref-type="bibr" rid="B6">Chollet, 2017</xref>) and a multi-scale gated fusion mechanism,and the structure of the MixConv Block is depicted in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>. First, the input feature <inline-formula>
<mml:math display="inline" id="im17">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula> is normalised via BatchNorm (<xref ref-type="bibr" rid="B20">Ioffe and Szegedy, 2015</xref>) to enhance network efficiency and stability, yielding feature <inline-formula>
<mml:math display="inline" id="im18">
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula>. Subsequently, <inline-formula>
<mml:math display="inline" id="im19">
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> undergoes dual-path convolution processing.One branch employs deep separable convolution to efficiently extract local features, while the other branch incorporates a gating mechanism using the Sigmoid function to generate channel weights for the features, thereby enhancing the network&#x2019;s expressive capacity. This results in the intermediate feature <inline-formula>
<mml:math display="inline" id="im20">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Second, the intermediate feature <inline-formula>
<mml:math display="inline" id="im21">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> undergoes dual-path convolution processing. One branch employs deep separable convolution to further extract local features <inline-formula>
<mml:math display="inline" id="im22">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, while the other branch utilises dilated convolution (<xref ref-type="bibr" rid="B45">Yu and Koltun, 2015</xref>) and deep convolution to extract features <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> within a larger receptive field. Both branches incorporate InstanceNorm (<xref ref-type="bibr" rid="B39">Ulyanov et&#xa0;al., 2016</xref>) and the ReLU activation function to achieve normalisation and non-linear enhancement. Finally, a gating mechanism fuses the features from the dual-path convolutions. The outputs <inline-formula>
<mml:math display="inline" id="im24">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im25">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are concatenated, then convolved with the Sigmoid function to generate channel weights <inline-formula>
<mml:math display="inline" id="im26">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im27">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> for each feature. These weighted features are subsequently fused, followed by a pointwise convolution for channel mapping to unify dimensions. The result is connected via residual connections to the original input feature <inline-formula>
<mml:math display="inline" id="im28">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula>, yielding the output feature <inline-formula>
<mml:math display="inline" id="im29">
<mml:mi>y</mml:mi>
</mml:math>
</inline-formula>. With its computational process described by <xref ref-type="disp-formula" rid="eq3">Equations 3</xref>&#x2013;<xref ref-type="disp-formula" rid="eq8">8</xref>. Here, PWConv denotes pointwise convolution, DWConv denotes depthwise convolution, and DConv denotes dilated convolution.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Structure of MixConv Block.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632052-g003.tif">
<alt-text content-type="machine-generated">Flowchart of a neural network architecture. It starts with batch normalization (BN), splits into two paths with pointwise convolutions (PWConv), and diverges further into a sigmoid and depthwise convolution (DWConv). Outputs merge and pass through more PWConv, dilated convolution (DConv), instance normalization (IN), rectified linear unit (ReLU), and DWConv layers. Paths converge, continue with a three-by-three convolution, sigmoid, and PWConv, ending with an addition node.</alt-text>
</graphic>
</fig>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mi>B</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>*</mml:mo>
<mml:mi>D</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>D</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>D</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>*</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>*</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Dehazing feature extraction unit</title>
<p>The atmospheric scattering model is commonly employed to describe the transition of a clear image to a hazy image. Due to the uncertainties in atmospheric light and the transmission map, haze removal from real hazy images remains a central focus for researchers. Methods that directly estimate atmospheric light and the transmission map in the original space may result in error accumulation. Inspired by FDU and PDU, incorporating physical priors into the feature space ensures the model aligns with the atmospheric scattering model, thereby enhancing the interpretability of the dehazing process while mitigating the impact of estimation errors in atmospheric light and the transmission map. This study introduces a novel dehazing feature extraction unit, DFEU, which predicts atmospheric light and the transmission map through a dual-path mechanism, establishing a relationship between hazy and dehazed images in the feature space, and synthesizing the features of the potential clear image with greater accuracy based on the physical model. First, we redefine the atmospheric scattering model, as shown in <xref ref-type="disp-formula" rid="eq9">Equations 9</xref>, <xref ref-type="disp-formula" rid="eq10">10</xref>.</p>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mi>J</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mo>+</mml:mo>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mi>J</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mo>+</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Then, features are extracted through the kernel K, and <xref ref-type="disp-formula" rid="eq9">Equation&#xa0;9</xref> can be expressed as <xref ref-type="disp-formula" rid="eq11">Equation 11</xref>:</p>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x229b;</mml:mo>
<mml:mi>J</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x229b;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2299;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>t</mml:mi>
</mml:mfrac>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x229b;</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im30">
<mml:mo>&#x229b;</mml:mo>
</mml:math>
</inline-formula> represents the convolution operator, and <inline-formula>
<mml:math display="inline" id="im31">
<mml:mo>&#x2299;</mml:mo>
</mml:math>
</inline-formula> represents the Hadamard product. We then introduce <inline-formula>
<mml:math display="inline" id="im32">
<mml:mi mathvariant="bold-italic">K</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im33">
<mml:mi mathvariant="bold-italic">J</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im34">
<mml:mi mathvariant="bold-italic">I</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im35">
<mml:mi mathvariant="bold-italic">A</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im36">
<mml:mi mathvariant="bold-italic">D</mml:mi>
</mml:math>
</inline-formula> as matrix-vector forms of <inline-formula>
<mml:math display="inline" id="im37">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im38">
<mml:mi>J</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im39">
<mml:mi>I</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im40">
<mml:mi>A</mml:mi>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math display="inline" id="im41">
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>t</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>, as shown in <xref ref-type="disp-formula" rid="eq12">Equation 12</xref>. We can compute the formula through algebraic operations. Additionally, the diagonal vectors of the diagonal matrix <inline-formula>
<mml:math display="inline" id="im42">
<mml:mi>D</mml:mi>
</mml:math>
</inline-formula> correspond to the vectorization of <inline-formula>
<mml:math display="inline" id="im43">
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>t</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<disp-formula id="eq12">
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">J</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">I</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="bold-italic">A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mi mathvariant="bold-italic">I</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">D</mml:mi>
<mml:mi mathvariant="bold-italic">A</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">A</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Next, we decompose the matrix <inline-formula>
<mml:math display="inline" id="im44">
<mml:mrow>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> into the product of two matrices <inline-formula>
<mml:math display="inline" id="im45">
<mml:mi mathvariant="bold-italic">K</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im46">
<mml:mi mathvariant="bold-italic">Q</mml:mi>
</mml:math>
</inline-formula>, as indicated in <xref ref-type="disp-formula" rid="eq13">Equation 13</xref>.</p>
<disp-formula id="eq13">
<label>(13)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">J</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">I</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">A</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>We can denote <inline-formula>
<mml:math display="inline" id="im47">
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> as an approximation of the atmospheric light corresponding to the feature <inline-formula>
<mml:math display="inline" id="im48">
<mml:mrow>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im49">
<mml:mover accent="true">
<mml:mi>t</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> as an approximation of the transmission map corresponding to the feature <inline-formula>
<mml:math display="inline" id="im50">
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="bold-italic">Q</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula>
<mml:math display="inline" id="im51">
<mml:mrow>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im52">
<mml:mrow>
<mml:mi mathvariant="bold-italic">K</mml:mi>
<mml:mi mathvariant="bold-italic">J</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> can be considered the extracted features <inline-formula>
<mml:math display="inline" id="im53">
<mml:mover accent="true">
<mml:mi>I</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im54">
<mml:mover accent="true">
<mml:mi>J</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> of the hazy image<inline-formula>
<mml:math display="inline" id="im55">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and its corresponding clear image <inline-formula>
<mml:math display="inline" id="im56">
<mml:mi>J</mml:mi>
</mml:math>
</inline-formula>, respectively. Therefore, based on <xref ref-type="disp-formula" rid="eq12">Equation 12</xref>, we can calculate the physically perceived features, as shown in <xref ref-type="disp-formula" rid="eq14">Equation 14</xref>.</p>
<disp-formula id="eq14">
<label>(14)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>J</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>I</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2299;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mover accent="true">
<mml:mi>t</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mfrac>
<mml:mo>+</mml:mo>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The structure of the DFEU is illustrated in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>. The DFEU employs a dual-path design to predict atmospheric light and the transmission map, with one branch generating the atmospheric light <inline-formula>
<mml:math display="inline" id="im57">
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula>. First, local features are extracted via convolution. Subsequently, two shallow striped convolutions with reduced parameters approximate the effect of standard large-kernel depth convolutions, capturing broader contextual information. Next, global context is fused and nonlinear expression enhanced through convolution, BN, and the ReLU activation function. Finally, atmospheric light is extracted via convolutional layer with Sigmoid activation function (<xref ref-type="bibr" rid="B2">Cai et&#xa0;al., 2024</xref>). In the other branch, the transmission map <inline-formula>
<mml:math display="inline" id="im58">
<mml:mover accent="true">
<mml:mi>t</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> is generated. According to prior research, the transmission map is non-uniform. First, we employ a spatial pyramid structure for multi-scale feature extraction, wherein the structure adaptively and uniformly pools the input features across three scales to capture additional feature representations and structural information. Subsequently, we adjust the dimensions of the three outputs and concatenate them to form a one-dimensional attention map (<xref ref-type="bibr" rid="B16">Guo et&#xa0;al., 2020</xref>). Next, we employ two points to perform dimensionality reduction and enhance non-linear expression through the convolutional layer and the ReLU activation function. Finally, we utilise the Sigmoid activation function to extract transmission graph features (<xref ref-type="bibr" rid="B47">Yu et&#xa0;al., 2024a</xref>), as shown in <xref ref-type="disp-formula" rid="eq15">Equations 15</xref>&#x2013;<xref ref-type="disp-formula" rid="eq17">17</xref>. Here, <inline-formula>
<mml:math display="inline" id="im59">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msup>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>11</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> denotes a convolution kernel of (1,11), which emphasizes feature extraction in the horizontal direction, while <inline-formula>
<mml:math display="inline" id="im60">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msup>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>11</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> denotes a convolution kernel of (11,1), which emphasizes feature extraction in the vertical direction.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Structure of Dehaze Feature Extraction Unit.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632052-g004.tif">
<alt-text content-type="machine-generated">Diagram of a neural network structure with two main branches. The top branch processes input \( \overline{I} \) using a sequence of convolutions, batch normalization, ReLU activation, and sigmoid function to output \( \overline{A} \). The bottom branch applies average pooling, resizing, and pointwise convolutions, also followed by ReLU and sigmoid functions, to output \( \overline{t} \). Outputs are combined through operations including subtraction, multiplication, and addition to produce the final output \( \overline{J} \).</alt-text>
</graphic>
</fig>
<disp-formula id="eq15">
<label>(15)</label>
<mml:math display="block" id="M15">
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>B</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>D</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msup>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>11</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>D</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msup>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>11</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover accent="true">
<mml:mi>I</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<disp-formula id="eq16">
<label>(16)</label>
<mml:math display="block" id="M16">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>t</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover accent="true">
<mml:mi>I</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover accent="true">
<mml:mi>I</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>4</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover accent="true">
<mml:mi>I</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq17">
<label>(17)</label>
<mml:math display="block" id="M17">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>t</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>t</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The proposed DFEU generates dehazed features <inline-formula>
<mml:math display="inline" id="im61">
<mml:mover accent="true">
<mml:mi>J</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> from the input features <inline-formula>
<mml:math display="inline" id="im62">
<mml:mover accent="true">
<mml:mi>I</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula>, which are subsequently utilized by the decoder to produce dehazed images. DFEU predicts atmospheric light and the transmission map using a dual-path approach, establishing the relationship between hazy and dehazed images in the feature space and synthesizing more accurate features for potential dehazed images.</p>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>SK fusion</title>
<p>To fuse the dehazed features extracted by DFEU with those from the MixConv module, this study introduces SK Fusion, which is based on the SK module (<xref ref-type="bibr" rid="B26">Li et&#xa0;al., 2019</xref>). The structural diagram is presented in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>. The two input features consist of the feature map <inline-formula>
<mml:math display="inline" id="im63">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> from the skip connection and the feature map <inline-formula>
<mml:math display="inline" id="im64">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> from the main path. Initially, the input features <inline-formula>
<mml:math display="inline" id="im65">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im66">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are added, followed by global average pooling to extract global information for each channel. Next, the MLP module F comprising two PWConv layers and a ReLU activation function, is introduced to generate a more compact feature representation, thus improving the accuracy of adaptive selection. The two PWConv layers perform dimensionality reduction and expansion, enhancing the efficiency of the MLP module. Finally, the obtained fusion weights are processed using the softmax function and a segmentation operation, enabling the weighted selection of different information, as shown in <xref ref-type="disp-formula" rid="eq18">Equations 18</xref>, <xref ref-type="disp-formula" rid="eq19">19</xref>. Ultimately, the fused output feature <inline-formula>
<mml:math display="inline" id="im67">
<mml:mi>y</mml:mi>
</mml:math>
</inline-formula> is obtained.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Structure of SK Fusion.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632052-g005.tif">
<alt-text content-type="machine-generated">Flowchart of a neural network process includes components: GAP, PWConv, ReLU, dual PWConvs, and Softmax. Connectors indicate data flow with operations like addition and multiplication, illustrating processing steps.</alt-text>
</graphic>
</fig>
<disp-formula id="eq18">
<label>(18)</label>
<mml:math display="block" id="M18">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq19">
<label>(19)</label>
<mml:math display="block" id="M19">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>*</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>*</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s2_6">
<label>2.6</label>
<title>Loss function</title>
<p>In this study, the L1 loss function is employed, which quantifies the absolute difference between the predicted and true values, also referred to as Least Absolute Deviation or Absolute Error Loss. In general, it minimizes the sum of the absolute differences between the target values <inline-formula>
<mml:math display="inline" id="im68">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the model&#x2019;s predicted values <inline-formula>
<mml:math display="inline" id="im69">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Specifically, let the target value be <inline-formula>
<mml:math display="inline" id="im70">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the model&#x2019;s predicted value be <inline-formula>
<mml:math display="inline" id="im71">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The loss function calculation is given by <xref ref-type="disp-formula" rid="eq20">Equation 20</xref>.</p>
<disp-formula id="eq20">
<label>(20)</label>
<mml:math display="block" id="M20">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results and analysis</title>
<sec id="s3_1">
<label>3.1</label>
<title>Experimental environment</title>
<p>In this study, to ensure the objectivity and reliability of the experimental results, all experiments were conducted under a consistent setup. The experiments were conducted on Ubuntu 20.04, utilizing an Intel(R) Xeon(R) Platinum 8352V CPU @ 2.10GHz, paired with an NVIDIA GTX 4090 GPU and 24GB of VRAM. The programming language used was Python 3.8.10, with PyTorch 1.11.0 as the deep learning framework and CUDA 11.3 for GPU acceleration. The training batch size was set to 24, with 1000 epochs. The optimizer used was AdamW (<xref ref-type="bibr" rid="B24">Kingma and Ba, 2014</xref>), with an initial learning rate of 0.0002 and a decay factor of 0.01.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Evaluation metrics</title>
<p>This paper employs commonly used image dehazing evaluation metrics, namely Peak Signal-to-Noise Ratio (PSNR) and Structural Similarity (SSIM), to comprehensively evaluate the model&#x2019;s dehazing performance. PSNR is a metric for image quality that measures the ratio between the maximum signal and background noise. For a grayscale image I of size m&#xd7;n and a noisy image K, the PSNR calculation formula is provided in <xref ref-type="disp-formula" rid="eq21">Equations 21</xref>, <xref ref-type="disp-formula" rid="eq22">22</xref>.</p>
<disp-formula id="eq21">
<label>(21)</label>
<mml:math display="block" id="M21">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq22">
<label>(22)</label>
<mml:math display="block" id="M22">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>10</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>V</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>u</mml:mi>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Here, <inline-formula>
<mml:math display="inline" id="im72">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the Mean Squared Error between two images, and <inline-formula>
<mml:math display="inline" id="im73">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>V</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> refers to the maximum possible pixel value in the image. The minimum value of PSNR is 0, with higher values indicating smaller differences between the two images and less image distortion.</p>
<p>SSIM is a metric used to quantify the structural similarity between two images, based on the human visual system&#x2019;s sensitivity to changes in local image structures. SSIM evaluates image properties such as brightness, contrast, and structure. Brightness is estimated using the mean, contrast through variance, and structural similarity via covariance. Given two images, <inline-formula>
<mml:math display="inline" id="im74">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im75">
<mml:mi>y</mml:mi>
</mml:math>
</inline-formula>, the SSIM calculation formula is provided in <xref ref-type="disp-formula" rid="eq23">Equation 23</xref>.</p>
<disp-formula id="eq23">
<label>(23)</label>
<mml:math display="block" id="M23">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Here, <inline-formula>
<mml:math display="inline" id="im76">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the mean of <inline-formula>
<mml:math display="inline" id="im77">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im78">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> denotes the variance of <inline-formula>
<mml:math display="inline" id="im79">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im80">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the mean of <inline-formula>
<mml:math display="inline" id="im81">
<mml:mi>y</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im82">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the variance of <inline-formula>
<mml:math display="inline" id="im83">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im84">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the covariance between <inline-formula>
<mml:math display="inline" id="im85">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im86">
<mml:mi>y</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im87">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im88">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are constants to maintain stability and avoid division by zero, and <inline-formula>
<mml:math display="inline" id="im89">
<mml:mi>L</mml:mi>
</mml:math>
</inline-formula> refers to the pixel value range. Typically, <inline-formula>
<mml:math display="inline" id="im90">
<mml:mrow>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> = 0.01 and <inline-formula>
<mml:math display="inline" id="im91">
<mml:mrow>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> = 0.03. The minimum value of SSIM is 0, with higher SSIM values indicating greater similarity between the two images.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Ablation experiment</title>
<p>In this study, due to the significant non-uniform distribution of haze in remote sensing images, haze removal is frequently regarded as a classic non-uniform image dehazing problem. Therefore, this study assessed the performance of the proposed model in haze removal using the RSHaze remote sensing dataset. Additionally, the model&#x2019;s dehazing performance in outdoor agricultural settings was validated using the Paddydata paddy field haze dataset. Through a series of ablation experiments, the performance of each module in the network was tested on both datasets, gradually adding modules to observe their specific effect on network performance. Additionally, we compared the dehazing performance of the proposed DFEU with that of the FDU and PDU models.</p>
<sec id="s3_3_1">
<label>3.3.1</label>
<title>Ablation experiment of modules</title>
<p>Based on gUNet, we named the resulting model Model1. We then introduced the MixConv module alone, naming it Model2, followed by the introduction of DFEU alone, resulting in Model3. Finally, both MixConv and DFEU were combined, naming it Model4. The experimental results are presented in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. The &#x201c;&#x221a;&#x201d; in the table indicates the inclusion of the module. From <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>, it can be observed that Model1 achieved a PSNR of 33.2461 and an SSIM of 0.9894 on the RSHaze remote sensing dataset, and a PSNR of 34.7628 and an SSIM of 0.9935 on the Paddydata haze image dataset. To enhance the model&#x2019;s ability to extract multi-scale information, Model2 incorporated the MixConv module, replacing the convolution module of Model1, which leads to a significant improvement in dehazing performance. The number of parameters increased by 2.2371M. For RSHaze, the PSNR increased by 3.4682, and SSIM increased by 0.0035. For Paddydata, PSNR increased by 0.3235, and SSIM increased by 0.0004. Model3 integrated DFEU between the encoder and decoder, directly establishing the relationship between hazy and clear images in the feature space based on the atmospheric scattering model. This further extracts dehazing features from the image and improves the model&#x2019;s dehazing performance. Compared to Model1, the number of parameters increased by 1.2328M. For RSHaze, the PSNR increased by 1.3134, and the SSIM increased by 0.0015. For Paddydata, PSNR increased by 0.5689, and SSIM increased by 0.0001. Finally, Model4 incorporated both MixConv and DFEU, combining the advantages of both to further improve the model&#x2019;s performance. Compared to Model1, the number of parameters increased by 3.3493M, and FLOPs increased by 8.4398G. The PSNR and SSIM on RSHaze increased by 4.0426 and 0.0039, reaching values of 37.2887 and 0.9933, respectively. On Paddydata, PSNR and SSIM increased by 1.2578 and 0.0011, reaching values of 36.0206 and 0.9946.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Ablation experiment results for different modules.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Model</th>
<th valign="middle" rowspan="2" align="center">MixConv</th>
<th valign="middle" rowspan="2" align="center">DFEU</th>
<th valign="middle" colspan="2" align="center">RSHaze</th>
<th valign="middle" colspan="2" align="center">Paddydata</th>
<th valign="middle" rowspan="2" align="center">Parameters/M</th>
<th valign="middle" rowspan="2" align="center">FLOPs/G</th>
</tr>
<tr>
<th valign="middle" align="center">PSNR(dB)</th>
<th valign="middle" align="center">SSIM</th>
<th valign="middle" align="center">PSNR(dB)</th>
<th valign="middle" align="center">SSIM</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Model1</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">33.2461</td>
<td valign="middle" align="center">0.9894</td>
<td valign="middle" align="center">34.7628</td>
<td valign="middle" align="center">0.9935</td>
<td valign="middle" align="center">0.8432</td>
<td valign="middle" align="center">2.8347</td>
</tr>
<tr>
<td valign="middle" align="center">Model2</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">36.7143</td>
<td valign="middle" align="center">0.9929</td>
<td valign="middle" align="center">35.0863</td>
<td valign="middle" align="center">0.9939</td>
<td valign="middle" align="center">3.0803</td>
<td valign="middle" align="center">10.2486</td>
</tr>
<tr>
<td valign="middle" align="center">Model3</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">34.5595</td>
<td valign="middle" align="center">0.9909</td>
<td valign="middle" align="center">35.3317</td>
<td valign="middle" align="center">0.9936</td>
<td valign="middle" align="center">2.0760</td>
<td valign="middle" align="center">3.8639</td>
</tr>
<tr>
<td valign="middle" align="center">Model4(ours)</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">
<bold>37.2887</bold>
</td>
<td valign="middle" align="center">
<bold>0.9933</bold>
</td>
<td valign="middle" align="center">
<bold>36.0206</bold>
</td>
<td valign="middle" align="center">
<bold>0.9946</bold>
</td>
<td valign="middle" align="center">4.1925</td>
<td valign="middle" align="center">11.2745</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The best results are marked bold.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>This paper selected one representative image sample from each of the RSHaze and Paddydata datasets for ablation experiments, with results presented in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>. Model2 incorporates MixConv Blocks, demonstrating superior performance to Model1 in distant scenes and hazy regions by preserving greater structural and textural detail. However, residual blurring persists in denser fog areas. Model3, featuring DFEU, exhibits enhanced stability when processing unevenly distributed haze, with more natural transitions at object boundaries. Nevertheless, clarity remains slightly compromised on minute distant structures. Model 4, combining both MixConv Block and DFEU, achieves the most balanced overall performance. It not only effectively reduces extensive haze veils but also demonstrates significant improvements in detail recovery. The image&#x2019;s colour fidelity and contrast are closer to the real scene, demonstrating the effective integration of multi-scale feature extraction and adaptive path selection, thereby validating the efficacy of module combination.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Experimental results of ablation in different modules.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632052-g006.tif">
<alt-text content-type="machine-generated">A side-by-side comparison of aerial images showing the effects of four models on haze removal. The top row shows a cityscape (RSHaze), and the bottom row shows a paddy field (Paddydata). Each model's result is displayed, with SSIM (Structural Similarity Index Measure) values indicating image quality improvement. Model 1 (SSIM: 0.976), Model 2 (SSIM: 0.983), Model 3 (SSIM: 0.982), and Model 4 (ours) with the highest SSIMs: 0.984 and 0.986.</alt-text>
</graphic>
</fig>
<p>The experimental results demonstrate that the MixConv Block increases the receptive field by combining depthwise separable convolution with dilated convolution. Without a significant increase in computational costs, it captures subtle differences between distant haze features and clear images. Depthwise separable convolution reduces the parameter count, ensuring computational efficiency and meeting the requirements for large-scale image processing. Meanwhile, dilated convolution, by expanding the receptive field, is better suited to handle deeper or more extensive haze layers, thus improving the precision of information in the dehazing process. The incorporation of the gating mechanism further enhances the model&#x2019;s adaptability by dynamically selecting the output from different convolution paths, thereby effectively adjusting the fusion of multi-scale information. For images with varying haze intensities and distributions, the model can adaptively select the optimal feature path, thereby improving the accuracy of image detail recovery affected by haze. The application of this module in image dehazing not only enhances the ability to extract multi-scale features but also optimizes the restoration effect through an efficient weighting mechanism. Particularly in complex haze environments, it more effectively restores the true color and structure of the image. This design enhances the quality of dehazing while avoiding excessive computation and parameter redundancy introduced by traditional convolution layers. It ensures the efficiency and robustness of the entire dehazing process. DFEU predicts atmospheric light and transmission maps using dual pathways, where one path processes image information in different directions via horizontal and vertical convolutions, thereby capturing spatial feature dependencies. It adaptively learns the importance of each channel in the dehazing task, enhancing features from channels that carry critical information while suppressing irrelevant or redundant channels, thereby improving dehazing performance. The second path extracts spatial information at different scales through multi-scale adaptive pooling, incorporating a weighting mechanism to dynamically adjust feature importance at different spatial locations. Through the fusion and learning of multi-scale information, it ensures the precise restoration of haze regions of various sizes during the dehazing process, thereby improving the model&#x2019;s ability to recover image details. MixConv Block extracts multi-scale haze features while retaining structural details. These enhanced features are then utilised by DFEU, which estimates transmitted light and atmospheric light through dual-branch estimation, forming a natural transition from multi-scale feature extraction to haze component estimation. The incorporation of the MixConv Block and DFEU allows for a better capture of multi-scale information in images, further enhancing the model&#x2019;s ability to detect dehazing features, in the current environment of abundant computational resources, ensuring model efficiency while significantly enhancing its dehazing performance.</p>
</sec>
<sec id="s3_3_2">
<label>3.3.2</label>
<title>Ablation experiment of DFEU</title>
<p>Based on the atmospheric scattering model, we proposed a Dehazing Feature Extraction Unit (DFEU) that predicts atmospheric light and transmission maps through dual pathways, establishing the relationship between hazy and clear images in feature space, and synthesizing the potential clear image features more accurately according to the physical model. To evaluate the effectiveness of DFEU, we conducted experiments comparing the FDU and PDU. Starting with gUNet+MixConv, the FDU was introduced and named Model1, the PDU was added and named Model2, and finally, the DFEU was introduced and named Model3. The experimental results are presented in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. The &#x201c;&#x221a;&#x201d; in the table indicates the addition of the module. From <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>, it is evident that Model3, our proposed DUNet, achieved a PSNR of 37.2887 and an SSIM of 0.9933 on the RSHaze dataset. Compared to Model1 and Model2, the PSNR improved by 0.5021 and 0.2183, respectively, while the SSIM increased by 0.0005 and 0.0002, respectively. On the Paddydata foggy image dataset, the PSNR was 36.0206, and the SSIM was 0.9946. Compared to Model1 and Model2, the PSNR increased by 0.7896 and 0.2176, respectively, while the SSIM increased by 0.0005 and 0.0002, respectively.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Ablation experiment results for DFEU.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Model</th>
<th valign="middle" rowspan="2" align="center">FDU</th>
<th valign="middle" rowspan="2" align="center">PDU</th>
<th valign="middle" rowspan="2" align="center">DFEU</th>
<th valign="middle" colspan="2" align="center">RSHaze</th>
<th valign="middle" colspan="2" align="center">Paddydata</th>
<th valign="middle" rowspan="2" align="center">Parameters/M</th>
<th valign="middle" rowspan="2" align="center">FLOPs/G</th>
</tr>
<tr>
<th valign="middle" align="center">PSNR(dB)</th>
<th valign="middle" align="center">SSIM</th>
<th valign="middle" align="center">PSNR(dB)</th>
<th valign="middle" align="center">SSIM</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Model1</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">36.7866</td>
<td valign="middle" align="center">0.9928</td>
<td valign="middle" align="center">35.2310</td>
<td valign="middle" align="center">0.9941</td>
<td valign="middle" align="center">3.5389</td>
<td valign="middle" align="center">10.2435</td>
</tr>
<tr>
<td valign="middle" align="center">Model2</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">37.0704</td>
<td valign="middle" align="center">0.9931</td>
<td valign="middle" align="center">35.8030</td>
<td valign="middle" align="center">0.9944</td>
<td valign="middle" align="center">4.6154</td>
<td valign="middle" align="center">10.6710</td>
</tr>
<tr>
<td valign="middle" align="center">Model3(ours)</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">
<bold>37.2887</bold>
</td>
<td valign="middle" align="center">
<bold>0.9933</bold>
</td>
<td valign="middle" align="center">
<bold>36.0206</bold>
</td>
<td valign="middle" align="center">
<bold>0.9946</bold>
</td>
<td valign="middle" align="center">4.1925</td>
<td valign="middle" align="center">11.2745</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The best results are marked bold.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>This paper selected one representative image sample each from the RSHaze and Paddydata datasets for ablation experiments, with results presented in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>. Model1, failing to adequately account for the spatial non-uniformity of the transmission map, often exhibits residual haze veils in distant regions and blurred boundaries within the test images. Model2 employs dual-branch paths to model atmospheric light and transmission maps separately, yielding more natural overall colouration and improved edge clarity for foreground objects compared to FDU. However, it still exhibits insufficient detail in localised areas and extensive dense fog zones. Model 3, the proposed model in this paper, further incorporates an adaptive mechanism. In the test images, it not only restores colours more accurately in complex regions such as terrain boundaries but also maintains good clarity in fine textures and distant structures. Simultaneously, it avoids over-restoration in clear areas, achieving the optimal overall image depth and naturalness.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>DFEU module ablation experiment results.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632052-g007.tif">
<alt-text content-type="machine-generated">Comparison of image dehazing models on two datasets: RSHaze and Paddydata. The top row shows four images of a hazy landscape, improving in clarity from the original to Models 1, 2, and 3. Model 3 provides the clearest image with an SSIM of 0.951. The bottom row shows images of paddy fields under similar progression, with Model 3 offering the clearest result and an SSIM of 0.992.</alt-text>
</graphic>
</fig>
<p>The experimental results show that FDU overlooks the fact that transmission maps are not uniform like atmospheric light, and using the same method to extract features for both atmospheric light and transmission maps does not lead to accurate feature representations. PDU employs dual pathways to separately extract features corresponding to atmospheric light and transmission maps, more accurately synthesizing potential clear image features and promoting information transfer and feature extraction in feature space. DFEU further enhances feature detail extraction in the dual pathways, not only adaptively adjusting the importance of feature maps but also dynamically adjusting haze intensity in various regions of the image. This results in more accurate extraction of features corresponding to atmospheric light and transmission maps, thereby improving the dehazing effect. DFEU enables the model to restore details in hazy areas more effectively, while preventing over-restoration of clear regions, thereby preserving the naturalness of the image. Compared to FDU and PDU, DFEU exhibited excellent dehazing performance on both datasets, demonstrating the success of the proposed module.</p>
</sec>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Comparison of experimental results of various dehazing models</title>
<p>To validate the performance and effectiveness of the model, we conducted a comprehensive comparison experiment using several representative models, including AODNet, DehazeNet, gUNet, AECRNet (<xref ref-type="bibr" rid="B42">Wu et&#xa0;al., 2021</xref>), GridDehazeNet (<xref ref-type="bibr" rid="B30">Liu et&#xa0;al., 2019</xref>), GCANet, PFDN, MSBDN (<xref ref-type="bibr" rid="B10">Dong et&#xa0;al., 2020</xref>), and Dehazeformer.</p>
<sec id="s3_4_1">
<label>3.4.1</label>
<title>Quantitative analysis</title>
<p>To comprehensively assess the dehazing performance of the DUNet model developed in this study, we conducted comparative experiments using existing popular models in the same experimental environment. The results of the comparative experiments are presented in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>. As presented in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>, the PSNR of DUNet on the RSHaze remote sensing dataset is 37.2887, and the SSIM is 0.9933. On the Paddydata foggy image dataset, the PSNR is 36.0206, and the SSIM is 0.9946. DUNet achieved the highest PSNR and SSIM on both datasets. Compared to Dehazeformer, the model with the best dehazing performance among the other models, the number of parameters in the model increased by 3.5051M, and FLOPs increased by 4.8337G. DUNet improved the PSNR and SSIM by 1.3505 and 0.003, respectively, on the RSHaze dataset. On the Paddydata dataset, DUNet improved the PSNR and SSIM by 0.8459 and 0.0009, respectively. Compared with other models, DUNet has obvious advantages in terms of PSNR and SSIM, proving that our model has good application potential and scalability without significantly increasing computational overhead while improving dehazing performance. The comparison and analysis of the above results clearly demonstrate that, in the image dehazing task, DUNet achieves better evaluation metrics compared to other popular models, indicating superior dehazing performance.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Comparative experimental results of different models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Model</th>
<th valign="middle" colspan="2" align="center">RSHaze</th>
<th valign="middle" colspan="2" align="center">Paddydata</th>
<th valign="middle" rowspan="2" align="center">Parameters/M</th>
<th valign="middle" rowspan="2" align="center">FLOPs/G</th>
</tr>
<tr>
<th valign="middle" align="center">PSNR(dB)</th>
<th valign="middle" align="center">SSIM</th>
<th valign="middle" align="center">PSNR(dB)</th>
<th valign="middle" align="center">SSIM</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">gUNet</td>
<td valign="middle" align="center">33.2464</td>
<td valign="middle" align="center">0.9894</td>
<td valign="middle" align="center">34.7626</td>
<td valign="middle" align="center">0.9935</td>
<td valign="middle" align="center">0.8432</td>
<td valign="middle" align="center">2.8304</td>
</tr>
<tr>
<td valign="middle" align="center">MSBDN</td>
<td valign="middle" align="center">35.4667</td>
<td valign="middle" align="center">0.9908</td>
<td valign="middle" align="center">34.1449</td>
<td valign="middle" align="center">0.9903</td>
<td valign="middle" align="center">28.7117</td>
<td valign="middle" align="center">24.6672</td>
</tr>
<tr>
<td valign="middle" align="center">DehazeNet</td>
<td valign="middle" align="center">22.6331</td>
<td valign="middle" align="center">0.9262</td>
<td valign="middle" align="center">20.9062</td>
<td valign="middle" align="center">0.7736</td>
<td valign="middle" align="center">0.0092</td>
<td valign="middle" align="center">0.5915</td>
</tr>
<tr>
<td valign="middle" align="center">AOD-Net</td>
<td valign="middle" align="center">22.4631</td>
<td valign="middle" align="center">0.9291</td>
<td valign="middle" align="center">19.1063</td>
<td valign="middle" align="center">0.7531</td>
<td valign="middle" align="center">0.0023</td>
<td valign="middle" align="center">0.1164</td>
</tr>
<tr>
<td valign="middle" align="center">GCANet</td>
<td valign="middle" align="center">33.8754</td>
<td valign="middle" align="center">0.9885</td>
<td valign="middle" align="center">31.8586</td>
<td valign="middle" align="center">0.9878</td>
<td valign="middle" align="center">0.7021</td>
<td valign="middle" align="center">18.5027</td>
</tr>
<tr>
<td valign="middle" align="center">GridDehazeNet</td>
<td valign="middle" align="center">33.0615</td>
<td valign="middle" align="center">0.9895</td>
<td valign="middle" align="center">34.3725</td>
<td valign="middle" align="center">0.9929</td>
<td valign="middle" align="center">0.9588</td>
<td valign="middle" align="center">21.5598</td>
</tr>
<tr>
<td valign="middle" align="center">AECRNet</td>
<td valign="middle" align="center">31.8469</td>
<td valign="middle" align="center">0.9847</td>
<td valign="middle" align="center">32.1510</td>
<td valign="middle" align="center">0.9831</td>
<td valign="middle" align="center">2.5906</td>
<td valign="middle" align="center">42.9314</td>
</tr>
<tr>
<td valign="middle" align="center">PFDN</td>
<td valign="middle" align="center">34.9240</td>
<td valign="middle" align="center">0.9900</td>
<td valign="middle" align="center">33.5383</td>
<td valign="middle" align="center">0.9869</td>
<td valign="middle" align="center">11.2742</td>
<td valign="middle" align="center">50.5294</td>
</tr>
<tr>
<td valign="middle" align="center">Dehazeformer</td>
<td valign="middle" align="center">35.9382</td>
<td valign="middle" align="center">0.9903</td>
<td valign="middle" align="center">35.1747</td>
<td valign="middle" align="center">0.9937</td>
<td valign="middle" align="center">0.6874</td>
<td valign="middle" align="center">6.4408</td>
</tr>
<tr>
<td valign="middle" align="center">DUNet(ours)</td>
<td valign="middle" align="center">
<bold>37.2887</bold>
</td>
<td valign="middle" align="center">
<bold>0.9933</bold>
</td>
<td valign="middle" align="center">
<bold>36.0206</bold>
</td>
<td valign="middle" align="center">
<bold>0.9946</bold>
</td>
<td valign="middle" align="center">4.1925</td>
<td valign="middle" align="center">11.2745</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The best results are marked bold.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_4_2">
<label>3.4.2</label>
<title>Qualitative analysis</title>
<p>We selected three representative samples from the RSHaze dataset, which cover different haze concentrations in remote sensing images. Additionally, we chose three representative samples from the Paddydata dataset, containing images with varying haze concentrations collected from rice fields. These samples were qualitatively analyzed to assess the performance differences of various methods in handling segmentation accuracy, robustness, and adaptability to complex scenarios. In the selected test images, we focused on representative areas, such as color-rich regions and heavily hazy zones, and observed the dehazing performance of different models. <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref> and <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref> provide a qualitative comparison between DUNet and other dehazing models.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Performance evaluation of different dehazing models on the RSHaze dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632052-g008.tif">
<alt-text content-type="machine-generated">A grid of satellite images demonstrating the effectiveness of various dehazing algorithms, named AODNet, DehazeNet, gUNet, AECRNet, GridDehazeNet, GCANet, PFDN, MSBDN, Dehazeformer, DUNet, and GT. Each set of images has a corresponding SSIM score underneath, indicating the structural similarity index, with higher scores reflecting better image clarity and resemblance to the ground truth (GT).</alt-text>
</graphic>
</fig>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Performance evaluation of different dehazing models on the Paddydata dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632052-g009.tif">
<alt-text content-type="machine-generated">Comparison of images from various dehazing models: AODNet, DehazeNet, gUNet, AECRNet, GridDehazeNet, GCANet, PFDN, MSBDN, Dehazeformer, DUNet, and GT. Each model's output showcases improvements in image clarity and detail with SSIM scores indicating similarity. GUNet and DUNet achieve high SSIM scores of 0.992.</alt-text>
</graphic>
</fig>
<p>For the RSHaze dataset, early end-to-end image dehazing models, such as AODNet and DehazeNet, exhibit poor performance on remote sensing datasets. These models tend to distort when processing detailed and complex scenes, leaving significant haze in the resulting images and failing to achieve effective haze removal. Models that neglect physical feature spaces, such as gUNet, AECRNet, GridDehazeNet, and GCANet, display similar dehazing performance but still struggle with haze detail processing. As shown in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>, these models leave residual haze in the dehazed regions, resulting in slightly hazy images with color discrepancies compared to haze-free images. This indicates that neglecting physical features impacts both haze removal and color restoration in blurred areas. PFDN and MSBDN demonstrate improved dehazing performance, however, they still suffer from artifacts at object edges and slight visual blur in the restored images. Due to its feature extraction unit, PFDN stands out in color restoration and is one of the few models that emphasize color details in the tests. The introduction of an improved Transformer Block in Dehazeformer significantly advances feature extraction, particularly at object edges, making it the second-best model after our DUNet. However, it struggles to accurately restore the original colors in bright regions obscured by haze, leading to noticeable color errors. As shown in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>, our DUNet model outperforms all other models in dehazing, including color restoration and feature detail recovery. It not only recovers the bright regions obscured by haze but also excels in processing object details.</p>
<p>For the Paddydata dataset, AODNet and DehazeNet encounter the same issue. The processed paddy field images still retain some haze, which is easily noticeable, and fail to achieve effective haze removal. gUNet, AECRNet, and GCANet exhibit similar dehazing performance, however, they still show significant shortcomings in handling the edges of object details, such as the colored flag markers in the paddy fields. As shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>, after processing, the edges of the flag markers still display haze features, resulting in blurry details in the output image. Furthermore, due to the residual haze, color recovery is also insufficient. GridDehazeNet, PFDN, and MSBDN further improve dehazing performance. However, they still exhibit artifacts at object edges, and the bright regions of the restored images remain slightly obscured by haze. The introduction of an improved Transformer Block in Dehazeformer leads to significant advancements in feature handling, making it second only to DUNet in terms of dehazing performance. Nevertheless, color compensation remains slightly skewed, showing deviations from the original clear images, and slight haze still lingers in certain details, such as the paddy stalks. Finally, as shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>, DUNet outperforms all other models in haze removal, excelling in both object edge details and vibrant color regions. It restores sharper edges and handles detailed elements like paddy stalks and colored flags with the best performance.</p>
</sec>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Failure analysis and discussion</title>
<p>In <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref> and <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>, it can be observed that the model still exhibits minor residual haze in certain areas with colour tones similar to those of haze, such as road surfaces and reflective areas of paddy fields. In such areas, where the colours of the objects are highly similar to those of the haze itself, the model may struggle to distinguish between the actual scene content and the haze components, leading to incomplete haze removal. The fundamental reason lies in the fact that, in these low-contrast areas, the model finds it difficult to accurately differentiate between the foreground and the background information obscured by the haze, thereby reducing the effectiveness of the haze removal process. This phenomenon also reflects that the current model still has room for improvement in its representation capabilities when dealing with areas with blurred edges and weakened details, particularly in terms of modelling accuracy for colour separation and structural preservation. In future research, we will enhance the model&#x2019;s perception capabilities in low-contrast regions and improve its ability to distinguish between haze and background details. For example, more refined feature enhancement mechanisms or region-adaptive dehazing methods based on visual attention can be introduced to improve the model&#x2019;s dehazing accuracy in regions with similar colours.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>Deep learning has demonstrated significant advantages in image restoration tasks, offering an effective approach to the problem of fog removal. However, existing methods remain prone to feature loss and edge blurring under extreme weather conditions and complex scenes. Consequently, developing efficient fog removal techniques holds considerable importance for enhancing the accuracy and stability of drone-based agricultural field monitoring.</p>
<p>This paper proposed a dehazing method and develops a new dehazing network to remove blurriness from remote sensing datasets and foggy paddy field image datasets. DUNet aims to fully extract clear features from the images and effectively recover visual information affected by blurriness. Specifically, the backbone network extracts multi-scale feature information from blurry images, while the MixConv convolution module captures useful information more comprehensively, improving the model&#x2019;s feature representation ability when handling complex blurry images. The DFEU based on the atmospheric scattering model, establishes a mapping between the blurry and clear images in feature space through dual-path predictions, providing more precise information for the dehazing process and yielding finer dehazing features. Finally, the dynamic characteristics of the SK module enable it to flexibly adjust the feature fusion strategy under different input conditions, enhancing the model&#x2019;s adaptability and robustness.</p>
<p>The research on image dehazing based on remote sensing and foggy paddy field image datasets demonstrates that DUNet holds significant potential in addressing challenges such as haze and blurring. The PSNR on the RSHaze remote sensing dataset is 37.2887, and the SSIM is 0.9933. On the foggy paddy field image dataset, Paddydata, the PSNR is 36.0206 and the SSIM is 0.9946. Experimental results demonstrate that, compared to other popular image dehazing models, DUNet directly establishes relationships between hazy and clear images within the feature space. This enables the model to fully leverage image physical information to extract dehazing features and effectively restore visual information impaired by factors such as haze. DUNet offers superior performance, confirming its potential and feasibility for outdoor smart agriculture dehazing tasks.</p>
<p>However, similar to most deep convolutional network models, DUNet relies on paired images for training. While it demonstrates strong dehazing performance on the two synthetic blurry datasets used in this study, experiments with unpaired foggy images have not yet been explored. Moreover, DUNet does not adequately balance the number of parameters with computational efficiency. Although it performs well, the inclusion of complex units increases both the number of parameters and computational demands. Furthermore, the lack of real-world foggy image datasets has long been a challenge in the image dehazing field. Most existing open-source datasets are based on clear images with synthetic haze, which undermines the authenticity of foggy datasets and negatively impacts the model&#x2019;s performance.</p>
<p>In future research, we plan to address these limitations from three aspects. First, regarding datasets, we will collaborate with professional organizations or large laboratories to collect real foggy data, capturing paired images from the same area under both clear and foggy conditions. This will also include unpaired real foggy and clear images to ensure the authenticity and effectiveness of the dataset, which is foundational in deep learning. Second, in terms of models, developing a lightweight, high-performance image dehazing model will be a key direction, as image dehazing is primarily used as preprocessing for subsequent visual tasks. Thus, further research on model deployment and computational efficiency is necessary, focusing on lightweight yet high-performance model architectures. Additionally, incorporating better attention mechanisms and innovative feature fusion strategies will enhance the model&#x2019;s adaptability to complex environmental conditions. Finally, we aim to explore unpaired image dehazing, enabling experiments to be conducted entirely based on real-world foggy images, independent of datasets. This will involve not only convolutional neural networks but also the integration of generative adversarial networks and diffusion models for future development. Through these efforts, we seek to advance image dehazing technology to meet real-world needs for high-quality image restoration, providing more accurate technical support for research and contributing to the progress of fields such as intelligent monitoring, remote sensing, and smart agriculture.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: <uri xlink:href="https://github.com/MaiheZHao/data">https://github.com/MaiheZHao/data</uri>.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>WZ: Data curation, Methodology, Software, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. QZ: Investigation, Writing &#x2013; review &amp; editing. ML: Conceptualization, Writing &#x2013; review &amp; editing. GY: Writing &#x2013; review &amp; editing. ZL: Writing &#x2013; review &amp; editing. MQ: Supervision, Writing &#x2013; review &amp; editing. HY: Supervision, Writing &#x2013; review &amp; editing. YT: Funding acquisition, Resources, Supervision, Validation, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research and/or publication of this article. This work was supported by the Science and Technology Development Plan Project of Jilin Province, No.20240302074GX.</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If&#xa0;you identify any issues, please contact us.</p>
</sec>
<sec id="s10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bai</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Rice plant counting, locating, and sizing method based on high-throughput UAV RGB images</article-title>. <source>Plant Phenomics</source> <volume>5</volume>, <fpage>20</fpage>&#x2013;<lpage>20</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.34133/plantphenomics.0020</pub-id>, PMID: <pub-id pub-id-type="pmid">37040495</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Cai</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Lai</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Yao</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Poly kernel inception network for remote sensing detection</article-title>,&#x201d; in <conf-name>2024 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>. <fpage>27706</fpage>&#x2013;<lpage>27716</lpage>.</citation></ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Qing</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Tao</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>DehazeNet: an end-to-end system for single image haze removal</article-title>. <source>IEEE Trans. Image Process.</source> <volume>25</volume>, <fpage>5187</fpage>&#x2013;<lpage>5198</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIP.2016.2598681</pub-id>, PMID: <pub-id pub-id-type="pmid">28873058</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>D.</given-names>
</name>
<name>
<surname>He</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Hou</surname> <given-names>D.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). &#x201c;<article-title>Gated context aggregation network for image dehazing and deraining</article-title>,&#x201d; in <conf-name>2019 IEEE Winter Conference on Applications of Computer Vision (WACV)</conf-name>. <fpage>1375</fpage>&#x2013;<lpage>1383</lpage>.</citation></ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>He</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>Z.-m.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>DEA-net: single image dehazing based on detail-enhanced convolution and content-guided attention</article-title>. <source>IEEE Trans. Image Process.</source> <volume>33</volume>, <fpage>1002</fpage>&#x2013;<lpage>1015</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIP.2024.3354108</pub-id>, PMID: <pub-id pub-id-type="pmid">38252568</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chollet</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Xception: deep learning with depthwise separable convolutions</article-title>,&#x201d; in <conf-name>2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>. <fpage>1800</fpage>&#x2013;<lpage>1807</lpage>.</citation></ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cui</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Knoll</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>EENet: An effective and efficient network for single image dehazing</article-title>. <source>Pattern Recognition</source> <volume>158</volume>, <fpage>111074</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.patcog.2024.111074</pub-id>
</citation></ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dale-Jones</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Tjahjadi</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>1993</year>). <article-title>A study and modification of the local histogram equalization algorithm</article-title>. <source>Pattern Recognition</source> <volume>26</volume>, <fpage>1373</fpage>&#x2013;<lpage>1381</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/0031-3203(93)90143-K</pub-id>
</citation></ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Pan</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Physics-based feature Dehazing networks</article-title>. <source>Comput. Vision &#x2013; ECCV</source> <volume>2020</volume>, <fpage>188</fpage>&#x2013;<lpage>204</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-030-58577-8_12</pub-id>
</citation></ref>
<ref id="B10">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Dong</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Pan</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Xiang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). &#x201c;<article-title>Multi-scale boosted dehazing network with dense feature fusion</article-title>,&#x201d; in <conf-name>2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>. <fpage>2154</fpage>&#x2013;<lpage>2164</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-030-58577-8_12</pub-id>
</citation></ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dosovitskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Beyer</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Kolesnikov</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Weissenborn</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Zhai</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Unterthiner</surname> <given-names>T.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>An image is worth 16x16 words: transformers for image recognition at scale</article-title>. <source>ArXiv abs/2010.11929</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2010.11929</pub-id>
</citation></ref>
<ref id="B12">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Godard</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Aodha</surname> <given-names>O. M.</given-names>
</name>
<name>
<surname>Brostow</surname> <given-names>G. J.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Digging into self-supervised monocular depth estimation</article-title>,&#x201d; in <conf-name>2019 IEEE/CVF International Conference on Computer Vision (ICCV)</conf-name>. <fpage>3827</fpage>&#x2013;<lpage>3837</lpage>.</citation></ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goodfellow</surname> <given-names>I. J.</given-names>
</name>
<name>
<surname>Pouget-Abadie</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Mirza</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Warde-Farley</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Ozair</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2014</year>). <article-title>Generative adversarial nets</article-title>. <source>Neural Inf. Process. Syst</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.5555/2969033.2969125</pub-id>
</citation></ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goyal</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Dogra</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Lepcha</surname> <given-names>D. C.</given-names>
</name>
<name>
<surname>Goyal</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Alkhayyat</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chohan</surname> <given-names>J. S.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Recent advances in image dehazing: Formal analysis to automated approaches</article-title>. <source>Inf. Fusion</source> <volume>104</volume>, <fpage>102151</fpage>&#x2013;<lpage>102151</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.inffus.2023.102151</pub-id>
</citation></ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Research on unmanned aerial vehicle (UAV) rice field weed sensing image segmentation method based on CNN-transformer</article-title>. <source>Comput. Electron. Agric.</source> <volume>229</volume>, <fpage>109719</fpage>&#x2013;<lpage>109719</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.109719</pub-id>
</citation></ref>
<ref id="B16">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Guo</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Sansom</surname> <given-names>A.</given-names>
</name>
<name>
<surname>McGuire</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Kalaani</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Q.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). &#x201c;<article-title>Spanet: spatial pyramid attention network for enhanced image recognition</article-title>,&#x201d; in <conf-name>2020 IEEE International Conference on Multimedia and Expo (ICME)</conf-name>. <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</citation></ref>
<ref id="B17">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Guo</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Anwar</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Cong</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Image dehazing transformer with transmission-aware 3D position embedding</article-title>,&#x201d; in <conf-name>2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>. <fpage>5802</fpage>&#x2013;<lpage>5810</lpage>.</citation></ref>
<ref id="B18">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>Single image haze removal using dark channel prior</article-title>,&#x201d; in <conf-name>2009 IEEE Conference on Computer Vision and Pattern Recognition</conf-name>. <fpage>1956</fpage>&#x2013;<lpage>1963</lpage>., PMID: <pub-id pub-id-type="pmid">20820075</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>S.-C.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>B.-H.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>W.-J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Visibility restoration of single hazy images captured in real-world weather conditions</article-title>. <source>IEEE Trans. Circuits Syst. Video Technol.</source> <volume>24</volume>, <fpage>1814</fpage>&#x2013;<lpage>1824</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TCSVT.2014.2317854</pub-id>
</citation></ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ioffe</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Szegedy</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Batch normalization: accelerating deep network training by reducing internal covariate shift</article-title>. <source>ArXiv abs/1502.03167</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1502.03167</pub-id>
</citation></ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jackson</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Agyekum</surname> <given-names>K. O.</given-names>
</name>
<name>
<surname>Kwabena</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ukwuoma</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Patamia</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Hazy to hazy free: A comprehensive survey of multi-image, single-image, and CNN-based algorithms for dehazing</article-title>. <source>Comput. Sci. Rev.</source> <volume>54</volume>, <fpage>100669</fpage>&#x2013;<lpage>100669</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cosrev.2024.100669</pub-id>
</citation></ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Joshi</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Sandhu</surname> <given-names>K. S.</given-names>
</name>
<name>
<surname>Singh Dhillon</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Bohara</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Detection and monitoring wheat diseases using unmanned aerial vehicles (UAVs)</article-title>. <source>Comput. Electron. Agric.</source> <volume>224</volume>, <fpage>109158</fpage>&#x2013;<lpage>109158</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.109158</pub-id>
</citation></ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname> <given-names>J.-Y.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>L.-S.</given-names>
</name>
<name>
<surname>Hwang</surname> <given-names>S.-H.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>An advanced contrast enhancement using partially overlapped sub-block histogram equalization</article-title>. <source>IEEE Trans. Circuits Syst. Video Technol.</source> <volume>11</volume>, <fpage>475</fpage>&#x2013;<lpage>484</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/76.915354</pub-id>
</citation></ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname> <given-names>D. P.</given-names>
</name>
<name>
<surname>Ba</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Adam: A method for stochastic optimization</article-title>. <source>CoRR abs/1412.6980</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1412.6980</pub-id>
</citation></ref>
<ref id="B25">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>AOD-net: all-in-one dehazing network</article-title>,&#x201d; in <conf-name>2017 IEEE International Conference on Computer Vision (ICCV)</conf-name>. <fpage>4780</fpage>&#x2013;<lpage>4788</lpage>.</citation></ref>
<ref id="B26">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Selective kernel networks</article-title>,&#x201d; in <conf-name>2019 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>. <fpage>510</fpage>&#x2013;<lpage>519</lpage>.</citation></ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yin</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Efficient dehazing method for outdoor and remote sensing images</article-title>. <source>IEEE J. Selected Topics Appl. Earth Observations Remote Sens.</source> <volume>16</volume>, <fpage>4516</fpage>&#x2013;<lpage>4528</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/JSTARS.2023.3274779</pub-id>
</citation></ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2025</year>). <article-title>A dehazing method for UAV remote sensing based on global and local feature collaboration</article-title>. <source>Remote Sens.</source> <volume>17</volume>, <fpage>1688 17</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs17101688</pub-id>
</citation></ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lihe</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>He</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>PhDnet: A novel physic-aware dehazing network for remote sensing images</article-title>. <source>Inf. Fusion</source> <volume>106</volume>, <fpage>102277</fpage>&#x2013;<lpage>102277</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.inffus.2024.102277</pub-id>
</citation></ref>
<ref id="B30">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>GridDehazeNet: attention-based multi-scale network for image dehazing</article-title>,&#x201d; in <conf-name>2019 IEEE/CVF International Conference on Computer Vision (ICCV)</conf-name>. <fpage>7313</fpage>&#x2013;<lpage>7322</lpage>.</citation></ref>
<ref id="B31">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Nayar</surname> <given-names>S. K.</given-names>
</name>
<name>
<surname>Narasimhan</surname> <given-names>S. G.</given-names>
</name>
</person-group> (<year>1999</year>). &#x201c;<article-title>Vision in bad weather</article-title>,&#x201d; in <conf-name>Proceedings of the Seventh IEEE International Conference on Computer Vision 2</conf-name>, Vol. <volume>822</volume>. <fpage>820</fpage>&#x2013;<lpage>827</lpage>.</citation></ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qin</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Bai</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>FFA-net: feature fusion attention network for single image dehazing</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>34</volume>, <fpage>11908</fpage>&#x2013;<lpage>11915</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1609/aaai.v34i07.6865</pub-id>
</citation></ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Gong</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Cong</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Bai</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Perception-oriented UAV image dehazing based on super-pixel scene prior</article-title>. <source>IEEE Trans. Geosci. Remote Sens.</source> <volume>62</volume>, <fpage>1</fpage>&#x2013;<lpage>19</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TGRS.2024.3393751</pub-id>
</citation></ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ronneberger</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Fischer</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Brox</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>U-net: convolutional networks for biomedical image segmentation</article-title>. <source>Med. Image Computing Computer-Assisted Intervention &#x2013; MICCAI</source> <volume>2015</volume>, <fpage>234</fpage>&#x2013;<lpage>241</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-319-24574-4_28</pub-id>
</citation></ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>He</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Qian</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Du</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Vision transformers for single image dehazing</article-title>. <source>IEEE Trans. Image Process.</source> <volume>32</volume>, <fpage>1927</fpage>&#x2013;<lpage>1941</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIP.2023.3256763</pub-id>, PMID: <pub-id pub-id-type="pmid">37030760</pub-id></citation></ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Qian</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Du</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Rethinking performance gains in image dehazing networks</article-title>. <source>ArXiv abs/2209.11448</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2209.11448</pub-id>
</citation></ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Nian</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Shaghaleh</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Hamad</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Yue</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Combining features selection strategy and features fusion strategy for SPAD estimation of winter wheat based on UAV multispectral imagery</article-title>. <source>Front. Plant Sci.</source> <volume>15</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2024.1404238</pub-id>, PMID: <pub-id pub-id-type="pmid">38799101</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Cui</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>He</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Real scene single image dehazing network with multi-prior guidance and domain transfer</article-title>. <source>IEEE Trans. Multimedia</source> <volume>27</volume>, <fpage>5492</fpage>&#x2013;<lpage>5506</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TMM.2025.3543063</pub-id>
</citation></ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ulyanov</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Vedaldi</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Lempitsky</surname> <given-names>V. S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Instance normalization: the missing ingredient for fast stylization</article-title>. <source>ArXiv abs/1607.08022</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1607.08022</pub-id>
</citation></ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname> <given-names>N. M.</given-names>
</name>
<name>
<surname>Parmar</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname> <given-names>A. N.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). <article-title>Attention is all you need</article-title>. <source>Neural Inf. Process. Syst</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1706.03762</pub-id>
</citation></ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Detecting rice straw burning based on infrared and visible information fusion with UAV remote sensing</article-title>. <source>Comput. Electron. Agric.</source> <volume>222</volume>, <fpage>109078</fpage>&#x2013;<lpage>109078</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.109078</pub-id>
</citation></ref>
<ref id="B42">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Qu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Qiao</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). &#x201c;<article-title>Contrastive learning for compact single image dehazing</article-title>,&#x201d; in <conf-name>2021 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>. <fpage>10546</fpage>&#x2013;<lpage>10555</lpage>.</citation></ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>M.-D.</given-names>
</name>
<name>
<surname>Hsu</surname> <given-names>Y.-C.</given-names>
</name>
<name>
<surname>Tseng</surname> <given-names>W.-C.</given-names>
</name>
<name>
<surname>Tseng</surname> <given-names>H.-H.</given-names>
</name>
<name>
<surname>Lai</surname> <given-names>M.-H.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Precision assessment of rice grain moisture content using UAV multispectral imagery and machine learning</article-title>. <source>Comput. Electron. Agric.</source> <volume>230</volume>, <fpage>109813</fpage>&#x2013;<lpage>109813</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.109813</pub-id>
</citation></ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Vision foundation model for agricultural applications with efficient layer aggregation network</article-title>. <source>Expert Syst. Appl.</source> <volume>257</volume>, <fpage>124972</fpage>&#x2013;<lpage>124972</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eswa.2024.124972</pub-id>
</citation></ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Koltun</surname> <given-names>V.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Multi-scale context aggregation by dilated convolutions</article-title>. <source>CoRR abs/1511.07122</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1511.07122</pub-id>
</citation></ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Dai</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Mao</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>N.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>b). <article-title>Aquatic plants detection in crab ponds using UAV hyperspectral imagery combined with transformer-based semantic segmentation model</article-title>. <source>Comput. Electron. Agric.</source> <volume>227</volume>, <fpage>109656</fpage>&#x2013;<lpage>109656</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.109656</pub-id>
</citation></ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2024</year>a). <article-title>Multi-scale spatial pyramid attention mechanism for image recognition: An effective approach</article-title>. <source>Eng. Appl. Artif. Intell.</source> <volume>133</volume>, <fpage>108261</fpage>&#x2013;<lpage>108261</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.engappai.2024.108261</pub-id>
</citation></ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>UAV remote sensing image dehazing based on double-scale transmission optimization strategy</article-title>. <source>IEEE Geosci. Remote Sens. Lett.</source> <volume>19</volume>, <fpage>1</fpage>&#x2013;<lpage>5</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/LGRS.2022.3206205</pub-id>
</citation></ref>
<ref id="B49">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Patel</surname> <given-names>V. M.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Densely connected pyramid dehazing network</article-title>,&#x201d; in <conf-name>2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>. <fpage>3194</fpage>&#x2013;<lpage>3203</lpage>.</citation></ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jiao</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2025</year>). <article-title>MOSSNet: multiscale and oriented sorghum spike detection and counting in UAV images</article-title>. <source>Front. Plant Sci.</source> <volume>16</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2025.1526142</pub-id>, PMID: <pub-id pub-id-type="pmid">40949571</pub-id></citation></ref>
<ref id="B51">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhan</surname> <given-names>J.</given-names>
</name>
<name>
<surname>He</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Du</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Curricular contrastive regularization for physics-aware single image dehazing</article-title>,&#x201d; in <conf-name>2023 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>. <fpage>5785</fpage>&#x2013;<lpage>5794</lpage>.</citation></ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Mai</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Shao</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>A fast single image haze removal algorithm using color attenuation prior</article-title>. <source>IEEE Trans. Image Process.</source> <volume>24</volume>, <fpage>3522</fpage>&#x2013;<lpage>3533</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIP.2015.2446191</pub-id>, PMID: <pub-id pub-id-type="pmid">26099141</pub-id></citation></ref>
<ref id="B53">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Shuai</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>An improved single image haze removal algorithm based on dark channel prior and histogram specification</article-title>,&#x201d; in <conf-name>International Conference on Model Transformation</conf-name>.</citation></ref>
</ref-list>
</back>
</article>