<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Phys.</journal-id>
<journal-title>Frontiers in Physics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Phys.</abbrev-journal-title>
<issn pub-type="epub">2296-424X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1626220</article-id>
<article-id pub-id-type="doi">10.3389/fphy.2025.1626220</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Mul-material decomposition method for sandstone spectral CT images based on I-MultiEncFusion-Net</article-title>
<alt-title alt-title-type="left-running-head">Wu et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphy.2025.1626220">10.3389/fphy.2025.1626220</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wu</surname>
<given-names>Yanfang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1503541/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Ran</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kong</surname>
<given-names>Huihua</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3062997/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Ping</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zou</surname>
<given-names>Yu</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2908416/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Mathematics</institution>, <institution>North University of China</institution>, <addr-line>Taiyuan</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>National Key Laboratory of Photoelectric Dynamic Testing Technology and Instrument for Extreme Environments</institution>, <institution>North University of China</institution>, <addr-line>Taiyuan</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Shanxi Key Laboratory of Signal Capturing and Processing</institution>, <institution>North University of China</institution>, <addr-line>Taiyuan</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>School of Information and Communication Engineering</institution>, <addr-line>Taiyuan</addr-line>, <country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>State Key Laboratory of Lithospheric and Environmental Coevolution, Institute of Geology and Geophysics</institution>, <institution>Chinese Academy of Sciences</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1647743/overview">Jian Dong</ext-link>, Central South University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2372079/overview">Wenchao Zheng</ext-link>, Hubei University of Technology, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2814675/overview">Chengwang Xiao</ext-link>, Central South University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Huihua Kong, <email>huihuak@163.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>13</volume>
<elocation-id>1626220</elocation-id>
<history>
<date date-type="received">
<day>10</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Wu, Zhang, Kong, Chen and Zou.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wu, Zhang, Kong, Chen and Zou</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Material analysis in sandstone is essential for oil and gas extraction. Energy spectrum Computed Tomography (CT) can acquire various spectrally distinct datasets and reconstruct energy-selective images. Additionally, deep learning significantly improves the accuracy of material decomposition by establishing a nonlinear mapping relationship between multi-energy channel reconstructed images and their corresponding multi-material reconstructed images. However, traditional convolutional neural networks (CNNs) demonstrate limited effectiveness in capturing non-local features. In this paper, we present a multi-encoder single-decoder network architecture named I-MultiEncFusion-Net, designed for material decomposition. In this framework, multiple encoders concentrate on the distinctive features of reconstructed images from different energy spectrum channels, while a single decoder enables feature fusion. The encoder incorporates Inception_B modules that utilize three parallel branches to comprehensively capture image features, while integrating a Local-Nonlocal Feature Aggregation (LNFA) module to fuse both local and global characteristics. The non-local feature extraction module constructs non-local neighborhood relationships and employs Euclidean distance metrics to extract global contextual features from images, thereby enhancing the material decomposition process. To further enhance model accuracy, the decoder computes Huber loss between each output and its corresponding label, while simultaneously incorporating correlations of base material images extracted by a High-Resolution Network (HRNet) as an auxiliary loss constraint for material decomposition. Validation experiments using spectral CT data of sandstone demonstrate the method&#x2019;s efficacy. Both simulated and practical results indicate that I-MultiEncFusion-Net exhibits superior generalization capability, preserves internal image details, and produces decomposed images with sharper edges.</p>
</abstract>
<kwd-group>
<kwd>i-MultiEncFusion-net</kwd>
<kwd>high-resolution network</kwd>
<kwd>multi-material decomposition</kwd>
<kwd>layer normalization and feature aggregation</kwd>
<kwd>energy spectrum CT</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Radiation Detectors and Imaging</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Computed tomography (CT) has been extensively employed for cross-sectional analysis and three-dimensional structural characterization [<xref ref-type="bibr" rid="B1">1</xref>, <xref ref-type="bibr" rid="B2">2</xref>]. In reserve forecasting research, the applications of CT have shown sustained growth over the past decade, evolving into an indispensable tool for critical geological investigations, including reservoir rock analysis and mineral analysis [<xref ref-type="bibr" rid="B3">3</xref>&#x2013;<xref ref-type="bibr" rid="B5">5</xref>]. Materials analysis is crucial for determining oil content; however, conventional CT reveals tissue morphology but does not provide information about the elemental composition of the tissues. Spectral CT can acquire X-ray data at different energy levels, allowing for the precise analysis of characteristic attenuation curves to effectively identify and separate material components.</p>
<p>Rocks consist of heterogeneous mixtures of mineral constituents, matrix materials, pore networks, and fracture systems, each exhibiting distinct X-ray attenuation coefficients [<xref ref-type="bibr" rid="B6">6</xref>, <xref ref-type="bibr" rid="B7">7</xref>]. The attenuation of photons is influenced by both the material properties and the energy of the X-rays, resulting from a combination of photoelectric absorption and Compton scattering. Spectral CT captures two projection datasets using different energetic spectra, enabling the determination of the electron density and effective atomic number of various materials [<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B9">9</xref>]. This critical physical information is vital for characterizing material mixtures and differentiating between tissue types.</p>
<p>There are several mainstream material decomposition technologies. Alvarez and Macovski [<xref ref-type="bibr" rid="B10">10</xref>] firstly presents the theory of a technique for obtaining essentially complete energy dependent information in a computerized tomography system by making simple, low-resolution, energy spectrum measurements. This technique enables material differentiation and constituent identification by leveraging the energy-dependent attenuation properties of substances across distinct photon energy spectra. Building upon these foundations, two main algorithms subsequently developed are the one-step method and the two-step method. The one-step method refers to a direct iterative material decomposition approach. Michael [<xref ref-type="bibr" rid="B11">11</xref>] presents a method which makes use of both pre- and post-reconstruction data in an iterative manner to achieve accurate beam hardening correction and decomposition into basis materials. Such methods integrate material decomposition and image reconstruction into a single process, simplifying the workflow but potentially reducing computational speed. The Two-step method includes material decomposition approaches based on both the projection domain and the image domain [<xref ref-type="bibr" rid="B12">12</xref>&#x2013;<xref ref-type="bibr" rid="B14">14</xref>]. Projection domain material decomposition methods can effectively reduce the impact of beam hardening artifacts on reconstructed images [<xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B16">16</xref>]. However, they require precise estimation of the X-ray emission spectrum and have high spatial consistency requirements for energy spectrum CT projection data. In contrast, image domain material decomposition algorithms are more flexible, with relatively simple material decomposition models that are easier to implement [<xref ref-type="bibr" rid="B17">17</xref>&#x2013;<xref ref-type="bibr" rid="B19">19</xref>]. This paper focuses on image-domain decomposition due to its implementational robustness in handling real-world spectral CT datasets making them the focus of this study.</p>
<p>Emerging deep learning (DL) can extract non-linear relationships in a data driven way, enabling the discovery of complex features and representations [<xref ref-type="bibr" rid="B20">20</xref>]. These methods have been widely and successfully utilized in applications such as image classification, super-resolution imaging, and image denoising [<xref ref-type="bibr" rid="B21">21</xref>, <xref ref-type="bibr" rid="B22">22</xref>]. Convolutional Neural Networks (CNNs) can learn the characteristics of complex nonlinear relationships, thereby further improving the accuracy of material decomposition and advancing deep learning-based material decomposition algorithms. Current algorithms, including convolutional networks such as Incept-Net, Fully Convolutional Dense Network (FCDense-Net), and Squeeze-and-Excitation Neural Architecture Search Network (SeNAS-Net), have been widely applied to material decomposition [<xref ref-type="bibr" rid="B23">23</xref>]. By analyzing images across different energy spectra, material decomposition enhances performance by improving robustness, boosting sensitivity to dose variations, and minimizing noise and artifacts. The developed feedforward neural network projection decomposition techniques have successfully achieved multi-material projection decomposition on simulated data [<xref ref-type="bibr" rid="B17">17</xref>, <xref ref-type="bibr" rid="B24">24</xref>&#x2013;<xref ref-type="bibr" rid="B26">26</xref>]. GECCU-Net employs edge-conditioned convolutional layers to aggregate non-local features, effectively mitigating the impact of non-local noise on decomposition results [<xref ref-type="bibr" rid="B27">27</xref>]. MPU-Net proposes an encoder-multi-decoder architecture that utilizes High-Resolution Network (HRNet) to extract correlations between material images, thereby constraining the material decomposition model through these learned correlations [<xref ref-type="bibr" rid="B28">28</xref>]. Transformer-based architecture has demonstrated remarkable capabilities in capturing long-range dependencies and contextual features, particularly in medical image analysis tasks. Wang et al. [<xref ref-type="bibr" rid="B29">29</xref>] proposed framework integrates both convolutional neural networks and a Transformer module to effectively combine local and global information for direct material decomposition from single-energy CT images. The Butterfly Convolutional Neural Network (Butterfly Net) exhibits significant advantages over traditional Fully Convolutional Networks (FCN) in material decomposition [<xref ref-type="bibr" rid="B30">30</xref>]. Previous studies have confirmed that combining traditional model-driven decomposition methods with data-driven CNN to construct hybrid model-driven deep learning approaches results in improved projection decomposition performance of the neural network methods. Using a joint model-driven CNN for DE-CT image processing not only avoids the limitations of traditional reconstruction methods but also reduces image noise and artifacts, thereby improving the accuracy and efficiency of the decomposition. However, existing methods using traditional convolutional neural network (CNN) operators exhibit limited ability to capture non-local and global contextual features, restricting their effectiveness in optimizing material decomposition accuracy. Additionally, deep learning frameworks inadequately exploit cross-spectral correlations and energy-specific differences in multi-channel spectral CT data, leading to suboptimal utilization of spectral information for robust decomposition. Furthermore, existing algorithms have been rarely applied in the decomposition of rock materials.</p>
<p>To address these limitations, this paper proposes I-MultiEncFusion-Net, a multi-encoder single-decoder network designed to exploit spectral differences across energy channels while integrating local and non-local feature representations. The encoder employs Inception_B modules with parallel branches to capture multi-scale features and an LNFA module that aggregates both local textures and global contextual patterns through non-local neighborhood relationships measured by Euclidean distance. To optimize decomposition accuracy, the decoder incorporates Huber loss for each output channel and leverages HR Net-derived material correlations as a constraint within the loss function. Validated on spectral CT data of sandstone samples, I-MultiEncFusion-Net demonstrates superior generalization, enhanced detail preservation, and sharper edge delineation in both simulated and real-world experiments.</p>
</sec>
<sec id="s2">
<title>2 I-MultiEncFusion-net for multi-material decomposition</title>
<p>To comprehensively capture feature representations from reconstructed images across all energy channels in spectral CT imaging, this paper introduces an enhanced encoder-decoder architecture named I-MultiEncFusion-Net, building upon the foundation of Incept-Net (I-Net). The proposed framework employs a multi-encoder-single-decoder configuration to achieve cross-channel feature integration, where encoder and decoder components are interconnected through skip connections. The quantity of encoders precisely corresponds to the spectral dimension of CT energy channels (as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>). Each encoder performs hierarchical downsampling while the unified decoder executes progressive upsampling. Experimental validation utilizes tri-channel spectral CT data, corresponding to the design of three specialized encoder branches aimed at extracting energy-specific characteristics. The encoder section utilizes the Inception_B module structure, employing three parallel branches to comprehensively extract multi-scale and multi-directional features from the images. Finally, the LNFA module is introduced to achieve cross-modal feature fusion and material decomposition tasks through a unified decoder.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>I-MultiEncFusion-Net architecture.</p>
</caption>
<graphic xlink:href="fphy-13-1626220-g001.tif">
<alt-text content-type="machine-generated">Diagram of a neural network architecture featuring input images, convolutional blocks, encoder and decoder blocks, and outputs. Detailed components include input, convolution, batch normalization, and various convolutional layers. Arrows indicate the data flow through the network, with concatenation, convolution, and pooling operations. The decoder blocks end with transposed convolution, outputting final processed images.</alt-text>
</graphic>
</fig>
<sec id="s2-1">
<title>2.1 I-MultiEncFusion-net architecture</title>
<p>For spectral CT image, there exists a nonlinear relationship between the decomposed materials and the reconstructed images across three energy spectra. This study proposes an enhanced feature aggregation network to improve material decomposition accuracy through optimization of an encoder-decoder architecture. Initially, the multiscale feature extraction is implemented through an Inception_B module, a composite convolutional architecture comprising three parallel processing branches. Each branch initiates with a 1 &#xd7; 1 convolution for dimensionality reduction, followed by 3 &#xd7; 3 convolutional kernels of varying depths (0, 1, and 2 layers) to extract multiscale local features, with the convolutional layers configured to have 32, 64, and 128 channels, respectively. Additionally, to address local-nonlocal feature aggregation challenges, the U-Net architecture is enhanced through novel integration of a LNFA as the encoder, constructing fine-grained feature representations to strengthen feature extraction capabilities. The network processes three independent input images, with each input branch employing same structure for feature extraction. Each branch begins with a convolutional layer followed by a batch normalization (BN) layer for preliminary feature extraction, utilizing L<sub>2</sub> regularization to mitigate overfitting. Subsequently, four Inception_B modules are used for deep feature extraction. The outputs from the three branches are concatenated with the original input feature maps, followed by a 3 &#xd7; 3 convolutional layer with 256 channels, a BN layer, and a ReLU activation function. Finally, downsampling is achieved through a MaxPooling layer, while the architecture enhances feature representational capacity through multi-scale convolutional operations and non-local context aggregation. The HRNet multi-resolution supervision mechanism is utilized, incorporating the Huber loss function to enhance noise robustness. This approach effectively regulates the material decomposition process, thereby improving the model&#x2019;s robustness and accuracy.</p>
<p>Since the decomposed materials are correlated with the reconstructed images across three energy levels, a decoder is implemented to enable feature fusion within this framework. The decoder utilizes a deconvolution operation with a kernel size of 2 and a stride of 2 for upsampling, progressively reconstructing the feature maps from the encoder to match the original input resolution. After each upsampling operation, the resulting feature maps are fused with the corresponding feature maps from the encoder layers to obtain multi-scale feature information. Following two convolutional layers with a 3 &#xd7; 3 kernel and a stride of 1, the number of channels in each decoder block is 256, 128, 64, and 32, corresponding to the channel counts of the encoder. After the final decoder block, a convolutional layer with a kernel size of 1 &#xd7; 1 and 3 output channels are employed to perform a linear transformation on the feature maps, effectively aggregating the information within the feature maps, with the output channels aligned to the dimensions of the material decomposition targets.</p>
</sec>
<sec id="s2-2">
<title>2.2 LNFA model</title>
<p>With effective capabilities in the extraction and aggregation of both local and non-local features, the LNFA module is specifically designed to enhance model performance in deep learning. However, in the LNFA module, non-local features are extracted by Edge Convolution (ECC), which emphasizes the representation of local spatial structures. It should be noted that the integration of global information in the decomposition of materials within spectral computed tomography (CT) images enhances the model&#x2019;s representational capacity and decomposition accuracy, thereby improving the precision of material identification and separation. This study proposes a method for extracting local structural features by modeling spatial differences between adjacent points, emphasizing the integration of global information. This approach enables the dynamic synthesis of multi-regional feature representations, thereby enhancing the learning capacity of the network. The local feature extraction branch employs convolutional, normalization, and activation operations to capture detailed information within the local receptive field, facilitating the identification of low-level features such as textures and edges. And the non-local feature extraction branch constructs a non-local neighborhood to capture relationships between distant pixels across a broader spatial extent. It computes dynamic weights based on the similarity in Euclidean space, thereby enabling the weighted integration of non-local features (as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Dual-branch feature extraction architecture of LNFA block.</p>
</caption>
<graphic xlink:href="fphy-13-1626220-g002.tif">
<alt-text content-type="machine-generated">Diagram of an LNFA block illustrating feature extraction. The process includes two main parts: local feature extraction using convolution, batch normalization, and LeakyReLU; and non-local feature extraction. Non-local processing involves selecting non-local neighbors, calculating distances, generating dynamic weights, three-layer fully connected extraction, and aggregating features. Both extract local and non-local features for final output.</alt-text>
</graphic>
</fig>
<p>Firstly, local features are extracted using a 3 &#xd7; 3 convolutional layer, followed by BN and a Leaky ReLU activation function (<xref ref-type="disp-formula" rid="e1">Equation 1</xref>):<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>L</mml:mi>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>L</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>k</mml:mi>
<mml:mi>y</mml:mi>
<mml:mtext>Re</mml:mtext>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where, <italic>x</italic> is input feature map, Conv represents the convolution operation, BN signifies batch normalization, LeakyReLU refers to the Leaky ReLU activation function. The extracted local features are denoted as <inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>L</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, with <italic>i</italic> serving as an index corresponding to a specific position within the feature map.</p>
<p>Then, non-local features are extracted by selecting eight non-local neighborhoods randomly within a range of [-3, 3] centered around <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The Euclidean distance between the center point and the randomly selected non-local neighborhoods is computed and utilized as the weight for the edges:<disp-formula id="e2">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf3">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the feature vector of the <italic>j</italic>th non-local neighborhood point. <inline-formula id="inf4">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the edge weight between the center point <inline-formula id="inf5">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the non-local neighborhood <inline-formula id="inf6">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>A three-layer fully connected neural network is employed to generate dynamic aggregation weights based on these edge weights (<xref ref-type="disp-formula" rid="e3">Equation 3</xref>):<disp-formula id="e3">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>l</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf7">
<mml:math id="m10">
<mml:mrow>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>l</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the three-layer fully connected neural network, <inline-formula id="inf8">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>&#x3c9;</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the trainable parameters that generate the dynamic edge weight (<xref ref-type="disp-formula" rid="e4">Equation 4</xref>). <inline-formula id="inf9">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> corresponding to the <italic>j</italic>th node:<disp-formula id="e4">
<mml:math id="m13">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>ReLU</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>ReLU</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mspace width="-0.50em"/>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf10">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the weights of the fully connected layer, <inline-formula id="inf11">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the biases of the fully connected layer. Subsequently, the dynamic weights are aggregated with the non-local features, followed by batch normalization and activation (<xref ref-type="disp-formula" rid="e5">Equation 5</xref>):<disp-formula id="e5">
<mml:math id="m16">
<mml:mrow>
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf12">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the neighborhood set of the central point <inline-formula id="inf13">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> (non-local neighborhood). <inline-formula id="inf14">
<mml:math id="m19">
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> indicates the number of neighboring points, and <italic>k</italic> denotes the total count of non-local neighborhoods.</p>
<p>Finally, the extracted local features are concatenated with the aggregated non-local features, followed by fusion through a 1 &#xd7; 1 convolutional layer, BN, and Leaky ReLU activation, resulting in the final feature representation (<xref ref-type="disp-formula" rid="e6">Equation 6</xref>):<disp-formula id="e6">
<mml:math id="m20">
<mml:mrow>
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>concat</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>L</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>The LNFA module enhances feature expressiveness by integrating both local and non-local information, enabling the model to learn more robust and informative features when processing data characterized by local and non-local attributes.</p>
</sec>
<sec id="s2-3">
<title>2.3 Loss function</title>
<p>The HRNet architecture is distinguished by its ability to simultaneously capture fine-grained high-resolution features and rich contextual information. This characteristic is particularly relevant for material decomposition, where features of different materials may exhibit substantial variations across scales. In this study, HRNet is employed to extract relevant features from spectral CT images, utilizing the feature extraction module of HRNet to generate high-resolution loss that constrains the training process of the material decomposition model. Specifically, we compare real data with the outputs of the material decomposition model, using this comparative analysis as a basis for constraining the model&#x2019;s loss function, thereby enhancing the accuracy of material decomposition. HRNet effectively maintains the integration of high-resolution and low-resolution features through repeated multi-scale feature fusion across the entire network architecture. To further improve decomposition accuracy, we have enhanced HRNet by two upsampling layers and additional convolutional layers (as shown in <xref ref-type="fig" rid="F3">Figure 3</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Loss function.</p>
</caption>
<graphic xlink:href="fphy-13-1626220-g003.tif">
<alt-text content-type="machine-generated">Diagram of a neural network architecture with labeled components, including convolution layers, convolutional transpose, basic blocks, Layer 1, with arrows representing data flow. Huber Loss and HR Loss are indicated.</alt-text>
</graphic>
</fig>
<p>The Huber loss function and the HR loss function (<xref ref-type="disp-formula" rid="e7">Equation 7</xref>) are employed for training the network. In the presence of outliers within the dataset, traditional loss functions can negatively impact model performance. The Huber loss function effectively combines the benefits of mean squared error (MSE) and mean absolute error (MAE): it behaves as MSE for small errors while transitioning to MAE for larger errors, thus enhancing robustness against outliers:<disp-formula id="e7">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>&#x3b4;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mi>a</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3e;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where, <inline-formula id="inf15">
<mml:math id="m22">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> represents the discrepancy between the true value <italic>y</italic> and the predicted value <inline-formula id="inf16">
<mml:math id="m23">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf17">
<mml:math id="m24">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a hyperparameter that governs the threshold of the loss function. The Huber loss function exhibits a linear growth characteristic similar to that of MAE when the error is significant, thereby reducing the influence of outliers on the model. And the Huber loss function is smooth at the threshold, contributing to greater stability during optimization and facilitating faster convergence. Furthermore, by adjusting the hyperparameter <inline-formula id="inf18">
<mml:math id="m25">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, a flexible balance can be achieved between MSE and MAE, making it suitable for various application scenarios.</p>
<p>During the training process, the Huber loss utilizes the correlations among materials to impose constraints on the model. Additionally, the output from each decoder is computed in conjunction with its corresponding labels to generate the Huber loss. The overall loss of the I-MultiEncFusion-Net (<xref ref-type="disp-formula" rid="e8">Equation 8</xref>) can be expressed as follows:<disp-formula id="e8">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where <inline-formula id="inf19">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf20">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf21">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denote the Huber losses corresponding to the three materials. <inline-formula id="inf22">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the HR loss, <inline-formula id="inf23">
<mml:math id="m31">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> indicates the magnitude of the HR loss. For the HRNet, we employ a self-supervised approach for training and apply the Huber loss function to constrain the model. Dynamic weighting optimizes PSNR/SSIM, effectively addressing the limitations of single-loss functions, as shown in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Ablation of loss.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Loss</th>
<th align="center">CaCO<sub>3</sub> PSNR</th>
<th align="center">SiO<sub>2</sub> SSIM</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Huber</td>
<td align="center">28.5341</td>
<td align="center">0.9125</td>
</tr>
<tr>
<td align="center">HR</td>
<td align="center">26.8263</td>
<td align="center">0.9341</td>
</tr>
<tr>
<td align="center">Total</td>
<td align="center">32.9873</td>
<td align="center">0.9679</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s3">
<title>3 Experiments and result</title>
<p>The proposed I-MultiEncFusion-Net is implemented within the Keras and TensorFlow frameworks on an NVIDIA RTX 4090D GPU. Learning rate is set to 1 &#xd7; 10<sup>&#x2212;5</sup>, and the Adam optimization algorithm is employed to minimize the loss function. The Structural Similarity Index (SSIM) and Peak Signal-to-Noise Ratio (PSNR) are utilized as assessment metrics to quantify the similarity between predicted outcomes and ground truth labels. Notably, both simulated sandstone models and real data are utilized as test datasets, and I-MultiEncFusion-Net exhibits substantial performance advantages in all test datasets.</p>
<sec id="s3-1">
<title>3.1 Simulation experiment</title>
<p>The spectral computed tomography (CT) system is simulated using the Spekpy v2.0 spectral simulation software. The X-ray source is positioned 100 mm from the rotation center, and the detector, which has a length of 20 mm, consists of 256 individual detection elements. For each scan, data is acquired at intervals of 1.4 degrees, resulting in a total of 256 projections over 360 degrees. <xref ref-type="fig" rid="F4">Figure 4</xref> illustrates the structure of the simulated phantom used in this experiment, which consists of air, quartz and sodium feldspar (SiO<sub>2</sub>), calcite (CaCO<sub>3</sub>), and pyrite (FeS<sub>2</sub>), with a reconstruction image resolution of 256 &#xd7; 256 pixels. The simulation experiment selects three energy channels: [20,30) keV, [30,40) keV, and [40,50) keV, resulting in the generation of 800 datasets, of which 700 are designated for the training dataset and 100 for the testing dataset.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>The result of materials decomposition with different Net.</p>
</caption>
<graphic xlink:href="fphy-13-1626220-g004.tif">
<alt-text content-type="machine-generated">Three rows of images show circular patterns for different materials labeled SiO&#x2082;, FeS&#x2082;, and CaCO&#x2083;. Each row includes: an input image, a label image, outputs from Butterfly Net, Incept-Net, GECCU-Net, and a method labeled &#x22;Our Method.&#x22; The patterns show varying levels of detail and clarity.</alt-text>
</graphic>
</fig>
<sec id="s3-1-1">
<title>3.1.1 Comparison of I-MultiEncFusion-net and other net</title>
<p>Four famous net, Butterfly Net, Inecpt-Net, GECC and GECC, are selected to compared with I-MultiEncFusion-Net. These networks are representative of the current state-of-the-art in their respective domains and provide a robust framework for evaluating the performance of our approach. To ensure an objective assessment of each network&#x2019;s performance, we maintained consistency in the datasets utilized and the number of training iterations. <xref ref-type="fig" rid="F4">Figure 4</xref> illustrates the decomposition results for the three networks. And <xref ref-type="table" rid="T1">Table 1</xref> presents the quantitative analysis outcomes.</p>
<p>As illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>, the first column displays the input images, while the second column presents the corresponding label images for SiO<sub>2</sub>, FeS<sub>2</sub>, and CaCO<sub>3</sub>. Columns three through six sequentially showcase the decomposition results for SiO<sub>2</sub>, FeS<sub>2</sub>, and CaCO<sub>3</sub> using various Nets. It can be shown that DishNet, Inept-Net, and GECCU-Net achieve the material identification and decomposition. However, the performance of these networks in decomposing FeS<sub>2</sub> is notably inadequate, characterized by blurred edges and insufficient detail in the internal structure. Additionally, GECCU-Net&#x2019;s results for CaCO<sub>3</sub> decomposition are similarly unsatisfactory, whereas DishNet and Inept-Net demonstrate comparatively superior performance in this task. Notably, our method I-MultiEncFusion-Net significantly enhances both the clarity of internal structures and the accuracy of edge recovery in the decomposition results, thereby demonstrating its exceptional capability in material decomposition.</p>
<p>The cross-channel semantic fusion in I-MultiEncFusion-Net operates through three-stage process: first, encoder outputs from three energy-specific pathways undergo spatial interaction through concatenation followed by 3 &#xd7; 3 convolution and batch normalization to establish local feature correlations; second, these integrated features enter the LNFA block where self-attention mechanisms model global dependencies across energy channels, capturing contextual relationships between materials like CaCO<sub>3</sub> or SiO<sub>2</sub> spatial distributions; finally, learnable dynamic weights based on the similarity in Euclidean space, thereby enabling the weighted integration of non-local features, enabling adaptive fusion optimized for mineral decomposition tasks. As shown in <xref ref-type="table" rid="T2">Table 2</xref>, the results show that our method exhibits exceptional performance in decomposition tasks, particularly demonstrating superior efficacy in the decomposition of SiO<sub>2</sub> and FeS<sub>2</sub>. For SiO<sub>2</sub>, the ILNF-Net achieved a SSIM of 0.9998, effectively reconstructing the original image structure, with a PSNR of 37.6074, significantly outperforming other models. In the decomposition of FeS<sub>2</sub>, our method attained SSIM and PSNR values of 0.9883 and 36.8928, respectively, again surpassing all comparative models. The decomposition performance for CaCO<sub>3</sub> was slightly lower than that of other Net, with a comparatively reduced SSIM and modest PSNR. This difference may arise from the complexity of CaCO<sub>3</sub>&#x2019;s characteristics during component extraction, which resulted in minor overfitting during training. In future work, we will further investigate the introduction of adaptive attention mechanisms and multi-scale feature fusion strategies to optimize the network&#x2019;s ability to extract features from such materials, in order to improve decomposition accuracy. Despite the relatively lower performance of our method in the decomposition of CaCO<sub>3</sub>, as indicated by the SSIM in comparison to DishNet and Incept-Net, and a modest PSNR, it demonstrated significant efficacy in the decomposition of complex materials overall.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>The SSIM and PSNR of different nets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="2" align="center">Materials</th>
<th align="center">Butterfly net</th>
<th align="center">Incept-net</th>
<th align="center">GECCU-net</th>
<th align="center">Our method</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="center">SiO<sub>2</sub>
</td>
<td align="center">SSIM</td>
<td align="center">0.9634</td>
<td align="center">0.9667</td>
<td align="center">0.9962</td>
<td align="center">
<bold>0.9998</bold>
</td>
</tr>
<tr>
<td align="center">PSNR</td>
<td align="center">33.0194</td>
<td align="center">34.6891</td>
<td align="center">30.6676</td>
<td align="center">
<bold>37.6074</bold>
</td>
</tr>
<tr>
<td rowspan="2" align="center">FeS<sub>2</sub>
</td>
<td align="center">SSIM</td>
<td align="center">0.9443</td>
<td align="center">0.9401</td>
<td align="center">0.8897</td>
<td align="center">
<bold>0.9883</bold>
</td>
</tr>
<tr>
<td align="center">PSNR</td>
<td align="center">31.2487</td>
<td align="center">31.8577</td>
<td align="center">28.8101</td>
<td align="center">
<bold>36.8928</bold>
</td>
</tr>
<tr>
<td rowspan="2" align="center">CaCO<sub>3</sub>
</td>
<td align="center">SSIM</td>
<td align="center">0.9846</td>
<td align="center">0.9880</td>
<td align="center">0.944</td>
<td align="center">
<bold>0.9820</bold>
</td>
</tr>
<tr>
<td align="center">PSNR</td>
<td align="center">33.9883</td>
<td align="center">35.3123</td>
<td align="center">34.1991</td>
<td align="center">
<bold>31.4901</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>&#x002A;: the bold values is the result of our method (I-MultiEncFusion-Net).</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-1-2">
<title>3.1.2 Shape generalization</title>
<p>The heterogeneity of rock material structures necessitates a higher generalization capability in models. To evaluate the generalization performance of the proposed network, a newly constructed modeling dataset was utilized as the test set. The previously uniform circular structures were replaced with synthetic geometric structures of varied shapes, including triangles, rectangles, and squares, as illustrated in <xref ref-type="fig" rid="F5">Figure 5</xref>. Experimental results reveal that DishNet, Incept-Net, and GECCU-Net encounter significant challenges in achieving satisfactory outcomes for complex decomposition tasks. Specifically, DishNet was unable to decompose FeS<sub>2</sub>, partially decomposed CaCO<sub>3</sub>, and showed substantial residual CaCO<sub>3</sub> during the decomposition of SiO<sub>2</sub>. Incept-Net could not effectively decompose either CaCO<sub>3</sub> or SiO<sub>2</sub> but managed to partially decompose FeS<sub>2</sub>. GECCU-Net failed to decompose FeS<sub>2</sub> while successfully decomposing both CaCO<sub>3</sub> and SiO<sub>2</sub>. Notable, the proposed I-MultiEncFusion-Net, although exhibiting lower brightness in the decomposition of FeS<sub>2</sub> compared to CaCO<sub>3</sub> and SiO<sub>2</sub>, effectively decomposed all three materials.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>New test dataset with different geometrical shapes, where squares represent SiO<sub>2</sub>, rectangles denote CaCO<sub>3</sub>, and triangles signify FeS<sub>2</sub>.</p>
</caption>
<graphic xlink:href="fphy-13-1626220-g005.tif">
<alt-text content-type="machine-generated">Comparison of visual pattern recognition using different models. The first column shows the initial test image with various gray shapes. Subsequent columns display results from Butterfly Net, Incept-Net, GECCU, and ILNF-Net, illustrating how each model interprets or processes the shapes. Each column depicts two rows of images reflecting differing levels of detail and extraction by the methods.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="table" rid="T3">Table 3</xref> shows the analysis results of SSIM and PSNR for various networks applied to different materials. The results show that the proposed I-MultiEncFusion-Net demonstrates robust adaptability and stability in both SSIM and PSNR metrics. In the SiO<sub>2</sub> decomposition task, I-MultiEncFusion-Net achieved an SSIM of 0.9804, significantly outperforming other models. The PSNR value of 57.6074 is comparable to that of GECCU-Net at 58.6583, while still surpassing other models, thereby highlighting I-MultiEncFusion-Net&#x2019;s robust capabilities in detail preservation and noise management. For the FeS<sub>2</sub> decomposition, I-MultiEncFusion-Net recorded an SSIM of 0.9559, surpassing all competing models, particularly excelling in structural reconstruction. Although the PSNR of 48.0696 is slightly lower than that of the butterfly network and GECCU-Net, it still reflects commendable robustness. In the CaCO<sub>3</sub> decomposition, I-MultiEncFusion-Net achieved an SSIM of 0.953, surpassing GECCU-Net and significantly outperforming both the butterfly network and Incept-Net, with a PSNR of 50.9731 that is comparable to GECCU-Net, thereby demonstrating consistent performance. Overall, I-MultiEncFusion-Net exhibits outstanding generalization capabilities in complex tasks, particularly showing a significant advantage in structural similarity (SSIM) and stable adaptability across various material decomposition challenges. In real scanning experiments, the presence of noise in rock images is influenced by factors such as voltage fluctuations. This study also analyzes the sensitivity to noise by introducing Gaussian noise <inline-formula id="inf24">
<mml:math id="m32">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>&#x3b4;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The resulting images and experimental outcomes are shown in <xref ref-type="app" rid="app1">Appendix A</xref>. The results indicate that the proposed I-MultiEncFusion-Net model demonstrates robust performance under noise interference, effectively facilitating material decomposition. Notably, it demonstrates superior efficacy in the decomposition of CaCO<sub>3</sub>, the primary component of rocks.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>The results of materials decomposition for a new test dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="2" align="center">Materials</th>
<th align="center">Butterfly-net</th>
<th align="center">Incept-net</th>
<th align="center">GECCU-net</th>
<th align="center">ILNF-net</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="center">SiO<sub>2</sub>
</td>
<td align="center">SSIM</td>
<td align="center">0.7789</td>
<td align="center">0.7251</td>
<td align="center">0.9197</td>
<td align="center">
<bold>0.9804</bold>
</td>
</tr>
<tr>
<td align="center">PSNR</td>
<td align="center">39.5608</td>
<td align="center">43.3856</td>
<td align="center">58.6583</td>
<td align="center">
<bold>57.6074</bold>
</td>
</tr>
<tr>
<td rowspan="2" align="center">FeS<sub>2</sub>
</td>
<td align="center">SSIM</td>
<td align="center">0.9154</td>
<td align="center">0.9191</td>
<td align="center">0.9153</td>
<td align="center">
<bold>0.9559</bold>
</td>
</tr>
<tr>
<td align="center">PSNR</td>
<td align="center">57.9010</td>
<td align="center">42.0281</td>
<td align="center">57.5796</td>
<td align="center">
<bold>48.0696</bold>
</td>
</tr>
<tr>
<td rowspan="2" align="center">CaCO<sub>3</sub>
</td>
<td align="center">SSIM</td>
<td align="center">0.8219</td>
<td align="center">0.7573</td>
<td align="center">0.93057</td>
<td align="center">
<bold>0.953</bold>
</td>
</tr>
<tr>
<td align="center">PSNR</td>
<td align="center">45.6127</td>
<td align="center">46.3230</td>
<td align="center">53.1903</td>
<td align="center">
<bold>50.9731</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>&#x002A;: the bold values is the result of our method (I-MultiEncFusion-Net).</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 Ablation of architecture</title>
<p>In the I-MultiEncFusion-Net, the effectiveness of material decomposition is affected by the quantity of Inception_B modules and the sampling structures utilized for local (LF) and non-local (NLF) feature extraction. The increase in the number of Inception_B modules does not lead to an improvement in performance. And a simple combination of LF and NLF structures does not effectively enhance performance. Therefore, this study conducted an ablation analysis regarding the number of Inception_B modules and the LF and NLF structures, with results showed in <xref ref-type="table" rid="T4">Table 4</xref>. As the number of Inception_B modules increases from 2 to 4, the model does not demonstrate a linear improvement in performance. However, a comparative analysis of the SSIM values presented in the table (e.g., 0.9155 vs. 0.8266) reveals that the configuration with 4 modules effectively enhances the network&#x2019;s performance. Further analysis of structural factors reveals that a simple combination of LF and NLF (as indicated in the &#x201c;2 Inception_B&#x201d; column of <xref ref-type="table" rid="T3">Table 3</xref>) does not result in significant performance improvements. The ablation study results regarding the number of Inception_B modules and the implementation of the LF and NLF structures indicate that the configuration comprising 4 Inception_B modules in conjunction with LF and NLF achieves the highest SSIM value. These results indicate that a increase in model depth, combined with the use of local and non-local sampling structures, is effective for materials decomposition, particularly exhibiting notable advantages in the decomposition of SiO<sub>2</sub> materials. The implementation of 4 Inception_B modules, combined with both local and non-local sampling techniques, demonstrates a superior overall balance in performance. However, the results also indicate that the decomposition performance of CaCO<sub>3</sub> does not exhibit significant enhancement under the optimal configuration, suggesting the necessity for further investigation into optimization methodologies in future research.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Results of ablation study on architecture experiment.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Evaluation metrics</th>
<th colspan="2" align="center">2 Inception_B</th>
<th colspan="2" align="center">2 Inception_B</th>
<th colspan="2" align="center">4 Inception_B</th>
<th colspan="3" align="center">Materials</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="7" align="center">SSIM</td>
<td align="center">LF</td>
<td align="center">NLF</td>
<td align="center">LF</td>
<td align="center">NLF</td>
<td align="center">LF</td>
<td align="center">NLF</td>
<td align="center">SiO<sub>2</sub>
</td>
<td align="center">FeS<sub>2</sub>
</td>
<td align="center">CaCO<sub>3</sub>
</td>
</tr>
<tr>
<td align="center">&#x221a;</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="center">0.8266</td>
<td align="center">0.9155</td>
<td align="center">0.8057</td>
</tr>
<tr>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="center">0.7122</td>
<td align="center">0.9425</td>
<td align="center">0.8172</td>
</tr>
<tr>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="center">0.8072</td>
<td align="center">0.9154</td>
<td align="center">0.7310</td>
</tr>
<tr>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="left"/>
<td align="left"/>
<td align="center">0.8604</td>
<td align="center">0.9513</td>
<td align="center">0.8895</td>
</tr>
<tr>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="left"/>
<td align="center">0.8681</td>
<td align="center">0.9439</td>
<td align="center">0.8391</td>
</tr>
<tr>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">
<bold>0.9804</bold>
</td>
<td align="center">
<bold>0.9759</bold>
</td>
<td align="center">
<bold>0.883</bold>
</td>
</tr>
<tr>
<td rowspan="6" align="center">PANR</td>
<td align="center">&#x221a;</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="center">48.3208</td>
<td align="center">57.8798</td>
<td align="center">50.3266</td>
</tr>
<tr>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="center">41.4968</td>
<td align="center">46.6686</td>
<td align="center">40.5508</td>
</tr>
<tr>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="center">51.8042</td>
<td align="center">57.8475</td>
<td align="center">44.1407</td>
</tr>
<tr>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="left"/>
<td align="left"/>
<td align="center">47.6639</td>
<td align="center">47.5220</td>
<td align="center">38.4082</td>
</tr>
<tr>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="left"/>
<td align="center">49.0487</td>
<td align="center">47.1535</td>
<td align="center">45.8702</td>
</tr>
<tr>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">&#x221a;</td>
<td align="center">
<bold>57.6074</bold>
</td>
<td align="center">
<bold>48.0696</bold>
</td>
<td align="center">
<bold>50.9731</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>&#x002A;: the bold values is the final design structure of -MultiEncFusion-Net.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-3">
<title>3.3 Real scan experiment</title>
<p>To evaluate the decomposition performance of the proposed I-MultiEncFusion-Net in practical applications, experiments were conducted using sandstone samples. These samples were made by the Calcite <italic>In-situ</italic> Precipitation System (CIPS). To simulate the natural cementation of reservoir rocks, the samples were primarily composed of calcium carbonate and quartz cement. Medium-grained quartz sand (150&#x2013;300 &#x3bc;m), pre-washed and dried, was uniformly packed into a 20 &#xd7; 20 &#xd7; 10 cm<sup>3</sup> mold, followed by the injection of a water-based chemical solution to finish the final samples within the CIPS system. Medium-grained quartz sand (150&#x2013;300 &#x3bc;m), which was pre-washed and dried, was uniformly packed into a 20 &#xd7; 20 &#xd7; 10 cm<sup>3</sup> mold, after which a water-based chemical solution was injected to complete the final samples within the CIPS system.</p>
<p>The scanning was conducted using the NanoVoxel-3000HX X-ray three-dimensional high-resolution imaging system from Tianjin Sanying Precision Instrument, operating at a tube voltage of 70 kV. Two energy channels were utilized, specifically [25, 35) keV and [45, 55) keV. The distance from the X-ray source to the specimen was 14.6 mm, while the distance from the X-ray source to the detector was 648.9 mm. This model assumes the presence of three components&#x2014;pores, CaCO<sub>3</sub>, and SiO<sub>2</sub>&#x2014;in the artificial sandstone samples, each exhibiting different volumetric attenuation coefficient in the CT images. Furthermore, the X-ray attenuation of each voxel in the artificial sandstone is equivalent to the sum of the X-ray absorption contributions from the pores, CaCO<sub>3</sub>, and SiO<sub>2</sub> within that voxel. A total of 150 CT slices from layers 200 to 349 were designated as the training set, while 20 CT slices from layers 400 to 419 were allocated as the testing set. <xref ref-type="fig" rid="F6">Figure 6a</xref> presents representative images from the training set across various energy ranges, highlighting the challenges of mineral decomposition in rocks, which require simultaneous consideration of diverse mineral compositions. Based on the composition of the rock, the network primarily decomposes CaCO<sub>3</sub> and silicon dioxide SiO<sub>2</sub>, with the remaining voids represented as pores. The results indicate that the Butterfly Network is unable to effectively decompose CaCO<sub>3</sub>, whereas the other networks successfully achieve decomposition of both substances. However, in the enlarged view of the region of interest, it is evident that the edges of the CaCO<sub>3</sub> decomposition produced by GECCU exhibit significant blurriness, while Incept-Net demonstrates suboptimal recovery of the internal structure during the decomposition of SiO<sub>2</sub> (as shown in <xref ref-type="fig" rid="F6">Figure 6b</xref>). These limitations arise primarily because 7 mineral decomposition demands joint modeling of local textures7and non-local contextual features 7, a capability uniquely addressed by our I-MultiEncFusion-Net. Notably, I-MultiEncFusion-Net demonstrates superior performance in preserving the integrity of both the composition and structure of the materials.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>
<bold>(a)</bold> The training data under different energy. <bold>(b)</bold> The materials decomposition results by different Net.</p>
</caption>
<graphic xlink:href="fphy-13-1626220-g006.tif">
<alt-text content-type="machine-generated">Comparison of different methods for interpreting image data. Panel (a) shows two circular grayscale images at different energy levels, labeled [25,35] keV and [55,55] keV. Panel (b) presents multiple rows of circular images, each with a red square indicating a specific region. Methods compared include Label, Butterfly-Net, Incept-Net, GECCU, and Our Method, with corresponding black and white segmented outputs. Insets provide close-ups of specific areas.</alt-text>
</graphic>
</fig>
<p>The performance of different networks in material segmentation was evaluated using SSIM and PSNR, as shown in <xref ref-type="table" rid="T5">Table 5</xref>. The I-MultiEncFusion-Net demonstrates superior performance in the task of material decomposition in the real CT image.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>The results of material decomposition of sandstone based on real CT scanning images.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="2" align="center">Materials</th>
<th align="center">Butterfly- net</th>
<th align="center">Incept-net</th>
<th align="center">GECCU-net</th>
<th align="center">ILNF-net</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="center">SiO<sub>2</sub>
</td>
<td align="center">SSIM</td>
<td align="center">0.2771</td>
<td align="center">0.8892</td>
<td align="center">0.8432</td>
<td align="center">
<bold>0.9679</bold>
</td>
</tr>
<tr>
<td align="center">PSNR</td>
<td align="center">32.3072</td>
<td align="center">33.4696</td>
<td align="center">33.448</td>
<td align="center">
<bold>33.1885</bold>
</td>
</tr>
<tr>
<td rowspan="2" align="center">CaCO<sub>3</sub>
</td>
<td align="center">SSIM</td>
<td align="center">0.9726</td>
<td align="center">0.9741</td>
<td align="center">0.9722</td>
<td align="center">
<bold>0.9801</bold>
</td>
</tr>
<tr>
<td align="center">PSNR</td>
<td align="center">32.3945</td>
<td align="center">33.05</td>
<td align="center">32.8095</td>
<td align="center">
<bold>32.9873</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>&#x002A;: the bold values is the result of our method (I-MultiEncFusion-Net).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>For the decomposition of SiO<sub>2</sub>, I-MultiEncFusion-Net achieved a SSIM of 0.9679, significantly surpassing that of other models. In terms of PSNR, I-MultiEncFusion-Net also exhibited outstanding results, with a value of 33.1885, comparable to those of Incept-Net and GECCU-Net. For the decomposition of CaCO<sub>3</sub>, I-MultiEncFusion-Net achieved an SSIM of 0.9801, slightly surpassing that of other networks, while its PSNR performance remained among the best in the field at 32.9873. Overall, I-MultiEncFusion-Net exhibits enhanced decomposition quality and visual accuracy in image decomposition, particularly demonstrating a notable advantage in structural similarity (SSIM).</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>To address the limitations of traditional CNNs in extracting non-local features from CT images, this study proposes a multi-encoder-single-decoder architecture, named as I-MultiEncFusion-Net, which demonstrates superior material decomposition performance across both simulated and real datasets. Specifically, this architecture integrates parallel encoders designed to capture common features of base material images (through shared parameters for multimodal input) and distinctive features (via differential feature extraction). These features are fused across modalities using a single decoder, enabling the multi-encoder structure to effectively integrate information from multiple input images while emphasizing their differences, with the single decoder promoting feature sharing. Through ablation studies, this research identifies the optimal configuration of the Inception_B structure and the LNFA module, where the Inception_B module serves as an efficient feature extraction unit specifically designed for image-domain material decomposition tasks, capable of effectively capturing multi-scale features, while the LNFA module enhances the aggregation of local and non-local features, thereby optimizing the material decomposition process. Furthermore, this study innovatively employs the correlation between material images as a loss function to impose constraints on the model, resulting in improved stability and accelerated convergence during the optimization process. The superiority of the proposed model&#x2019;s performance is validated by results from both simulation experiments and artificial sandstone tests.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>YW: Writing &#x2013; review and editing, Supervision, Funding acquisition, Writing &#x2013; original draft, Software, Resources, Investigation, Validation, Project administration, Conceptualization, Methodology, Formal Analysis, Data curation, Visualization. RZ: Methodology, Project administration, Investigation, Writing &#x2013; original draft. HK: Software, Writing &#x2013; original draft. PC: Writing &#x2013; original draft, Funding acquisition. YZ: Data curation, Funding acquisition, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This research was funded by National Key Research and Development Program of China under grant 2023YFE0205800; National Nature Science Foundation of China under grants (U23A20285, 42207205); Provincial Natural Science Foundation of Shanxi, China under grants (202403021223006, 202403021211025), Shanxi Key Research and Development Program under grant (202302150401011), Technology Development Fund Project of Shanxi Province under grant (202304021301028), Shanxi Province Overseas Study Program under grants (2023-129).</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cong</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Xi</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Fitzgerald</surname>
<given-names>P</given-names>
</name>
<name>
<surname>De Man</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G</given-names>
</name>
</person-group>. <article-title>Virtual monoenergetic CT imaging via deep learning</article-title>. <source>Patterns</source> (<year>2020</year>) <volume>1</volume>(<issue>100128</issue>):<fpage>100128</fpage>. <pub-id pub-id-type="doi">10.1016/j.patter.2020.100128</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>J</given-names>
</name>
<etal/>
</person-group> <article-title>ShaleSeg: deep-learning dataset and models for practical fracture segmentation of large-scale shale CT images</article-title>. <source>Int. J Rock Mech Mining Sci</source> (<year>2024</year>) <volume>180</volume>:<fpage>105820</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijrmms.2024.105820</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cnudde</surname>
<given-names>V</given-names>
</name>
<name>
<surname>Boone</surname>
<given-names>MN</given-names>
</name>
</person-group>. <article-title>High-resolution X-ray computed tomography in geosciences: a review of the current technology and applications</article-title>. <source>Earth-Science Rev</source> (<year>2013</year>) <volume>123</volume>:<fpage>1</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1016/j.earscirev.2013.04.003</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>XK</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>B</given-names>
</name>
<name>
<surname>He</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>T</given-names>
</name>
</person-group>. <article-title>Study on the progressive fracturing in soil and rock mixture under uniaxial compression conditions by CT scanning</article-title>. <source>Eng Geology</source> (<year>2020</year>) <volume>279</volume>:<fpage>105884</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1016/j.enggeo.2020.105884</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y-H</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>W-L</given-names>
</name>
<name>
<surname>Bai</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Bai</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>YP</given-names>
</name>
<etal/>
</person-group> <article-title>Shale oil resource evaluation with an improved understanding of free hydrocarbons: insights from three-step hydrocarbon thermal desorption</article-title>. <source>Geosci Front</source> (<year>2023</year>) <volume>14</volume>(<issue>6</issue>):<fpage>101677</fpage>. <pub-id pub-id-type="doi">10.1016/j.gsf.2023.101677</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Duan</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>B</given-names>
</name>
<name>
<surname>He</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Hao</surname>
<given-names>J</given-names>
</name>
</person-group>. <article-title>Cracking evolution and failure characteristics of longmaxi shale under uniaxial compression using real-time computed tomography scanning</article-title>. <source>Rock Mech Rock Eng</source> (<year>2019</year>) <volume>52</volume>(<issue>9</issue>):<fpage>3003</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1007/s00603-019-01765-0</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>G</given-names>
</name>
</person-group>. <article-title>Quantitative analysis of anisotropy effect on hydrofracturing efficiency and process in shale using X-ray computed tomography and acoustic emission</article-title>. <source>Rock Mech Rock Eng</source> (<year>2021</year>) <volume>54</volume>:<fpage>5715</fpage>&#x2013;<lpage>30</lpage>. <pub-id pub-id-type="doi">10.1007/s00603-021-02589-7</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Patino</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Prochowski</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Agrawal</surname>
<given-names>MD</given-names>
</name>
<name>
<surname>Simeone</surname>
<given-names>FJ</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Hahn</surname>
<given-names>PF</given-names>
</name>
<etal/>
</person-group> <article-title>Material separation using dual-energy CT: current and emerging applications</article-title>. <source>Radiographics</source> (<year>2016</year>) <volume>36</volume>(<issue>4</issue>):<fpage>1087</fpage>&#x2013;<lpage>105</lpage>. <pub-id pub-id-type="doi">10.1148/rg.2016150220</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>A multienergy computed tomography method without image segmentation or prior knowledge of X-ray spectra or materials</article-title>. <source>Heliyon</source> (<year>2022</year>) <volume>8</volume>(<issue>11</issue>):<fpage>e11584</fpage>. <pub-id pub-id-type="doi">10.1016/j.heliyon.2022.e11584</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alvarez</surname>
<given-names>RE</given-names>
</name>
<name>
<surname>Macovski</surname>
<given-names>A</given-names>
</name>
</person-group>. <article-title>Energy-selective reconstructions in x-ray computerised tomography</article-title>. <source>Phys Med and Biol</source> (<year>1976</year>) <volume>21</volume>(<issue>5</issue>):<fpage>733</fpage>&#x2013;<lpage>44</lpage>. <pub-id pub-id-type="doi">10.1088/0031-9155/21/5/002</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Michael</surname>
<given-names>G</given-names>
</name>
</person-group>. <article-title>Tissue analysis using dual energy CT</article-title>. <source>Australas Phys and Eng Sci Med</source> (<year>1992</year>) <volume>15</volume>(<issue>1</issue>):<fpage>75</fpage>&#x2013;<lpage>87</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="http://europepmc.org/abstract/MED/1575646">http://europepmc.org/abstract/MED/1575646</ext-link>
</comment>
</citation>
</ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brody</surname>
<given-names>WR</given-names>
</name>
<name>
<surname>Butt</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Hall</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Macovski</surname>
<given-names>A</given-names>
</name>
</person-group>. <article-title>A method for selective tissue and bone visualization using dual energy scanned projection radiography</article-title>. <source>Med Phys</source> (<year>1981</year>) <volume>8</volume>(<issue>3</issue>):<fpage>353</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1118/1.594957</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fang</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Kalra</surname>
<given-names>MK</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L</given-names>
</name>
<etal/>
</person-group> <article-title>Iterative material decomposition for spectral CT using self-supervised Noise2Noise prior</article-title>. <source>Phys Med and Biol</source> (<year>2021</year>) <volume>66</volume>(<issue>15</issue>):<fpage>155013</fpage>. <pub-id pub-id-type="doi">10.1088/1361-6560/ac0afd</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ying</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Naidu</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Crawford</surname>
<given-names>CR</given-names>
</name>
</person-group>. <article-title>Dual energy computed tomography for explosive detection</article-title>. <source>J X-ray Sci Technology</source> (<year>2006</year>) <volume>14</volume>(<issue>4</issue>):<fpage>235</fpage>&#x2013;<lpage>56</lpage>. <pub-id pub-id-type="doi">10.3233/xst-2006-00163</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ducros</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Abascal</surname>
<given-names>JFPJ</given-names>
</name>
<name>
<surname>Sixou</surname>
<given-names>B</given-names>
</name>
<name>
<surname>Rit</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Peyrin</surname>
<given-names>F</given-names>
</name>
</person-group>. <article-title>Regularization of nonlinear decomposition of spectral x&#x2010;ray projection images</article-title>. <source>Med Phys</source> (<year>2017</year>) <volume>44</volume>(<issue>9</issue>):<fpage>e174</fpage>&#x2013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1002/mp.12283</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xue</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Lyu</surname>
<given-names>Q</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>C</given-names>
</name>
<etal/>
</person-group> <article-title>Accurate multi-material decomposition in dual-energy CT: a phantom study</article-title>. <source>IEEE Trans Comput Imaging</source> (<year>2019</year>) <volume>5</volume>(<issue>4</issue>):<fpage>515</fpage>&#x2013;<lpage>529</lpage>. <pub-id pub-id-type="doi">10.1109/TCI.2019.2909192</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Niu</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Petrongolo</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>L</given-names>
</name>
</person-group>. <article-title>Iterative image&#x2010;domain decomposition for dual&#x2010;energy CT</article-title>. <source>Med Phys</source> (<year>2014</year>) <volume>41</volume>(<issue>4</issue>):<fpage>041901</fpage>. <pub-id pub-id-type="doi">10.1118/1.4866386</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tao</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Rajendran</surname>
<given-names>K</given-names>
</name>
<name>
<surname>McCollough</surname>
<given-names>CH</given-names>
</name>
<name>
<surname>Leng</surname>
<given-names>S</given-names>
</name>
</person-group>. <article-title>Material decomposition with prior knowledge aware iterative denoising (MD-PKAID)</article-title>. <source>Phys Med and Biol</source> (<year>2018</year>) <volume>63</volume>(<issue>19</issue>):<fpage>195003</fpage>. <pub-id pub-id-type="doi">10.1088/1361-6560/aadc90</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Niu</surname>
<given-names>T</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Elmore</surname>
<given-names>K</given-names>
</name>
<etal/>
</person-group> <article-title>Using edge-preserving algorithm with non-local mean for significantly improved image-domain material decomposition in dual-energy CT</article-title>. <source>Phys Med and Biol</source> (<year>2016</year>) <volume>61</volume>(<issue>3</issue>):<fpage>1332</fpage>&#x2013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1088/0031-9155/61/3/1332</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kalmet</surname>
<given-names>PH</given-names>
</name>
<name>
<surname>Sanduleanu</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Primakov</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Jochems</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Refaee</surname>
<given-names>T</given-names>
</name>
<etal/>
</person-group> <article-title>Deep learning in fracture detection: a narrative review</article-title>. <source>Acta orthopaedica</source> (<year>2020</year>) <volume>91</volume>(<issue>2</issue>):<fpage>215</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1080/17453674.2019.1711323</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H</given-names>
</name>
</person-group>. <article-title>Rock crack recognition technology based on deep learning</article-title>. <source>Sensors</source> (<year>2023</year>) <volume>23</volume>(<issue>12</issue>):<fpage>5421</fpage>. <pub-id pub-id-type="doi">10.3390/s23125421</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>Intelligent identification of coal crack in CT images based on deep learning</article-title>. <source>Comput Intelligence Neurosci</source> (<year>2022</year>) <volume>2022</volume>:<fpage>1</fpage>&#x2013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1155/2022/7092436</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>X</given-names>
</name>
<name>
<surname>He</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Long</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>X</given-names>
</name>
<etal/>
</person-group> <article-title>Multi-material decomposition of spectral CT images via fully convolutional DenseNets</article-title>. <source>J X-Ray Sci Technology</source> (<year>2019</year>) <volume>27</volume>(<issue>3</issue>):<fpage>461</fpage>&#x2013;<lpage>71</lpage>. <pub-id pub-id-type="doi">10.3233/xst-190500</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gong</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Tao</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Rajendran</surname>
<given-names>K</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>W</given-names>
</name>
<name>
<surname>McCollough</surname>
<given-names>CH</given-names>
</name>
<name>
<surname>Leng</surname>
<given-names>S</given-names>
</name>
</person-group>. <article-title>Deep&#x2010;learning&#x2010;based direct inversion for material decomposition</article-title>. <source>Med Phys</source> (<year>2020</year>) <volume>47</volume>(<issue>12</issue>):<fpage>6294</fpage>&#x2013;<lpage>309</lpage>. <pub-id pub-id-type="doi">10.1002/mp.14523</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ji</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y</given-names>
</name>
<name>
<surname>Zhuo</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Kan</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>W</given-names>
</name>
<etal/>
</person-group> <article-title>SeNAS-net: self-supervised noise and artifact suppression network for material decomposition in spectral CT</article-title>. <source>Ieee Trans On Comput Imaging</source> (<year>2024</year>) <volume>10</volume>:<fpage>677</fpage>&#x2013;<lpage>89</lpage>. <pub-id pub-id-type="doi">10.1109/tci.2024.3394772</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>C-W</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>M</given-names>
</name>
<etal/>
</person-group> <article-title>Image-domain material decomposition for dual-energy CT using a conditional diffusion model</article-title>, <volume>12930</volume>. <publisher-loc>SPIE</publisher-loc> (<year>2024</year>). <fpage>517</fpage>&#x2013;<lpage>22</lpage>.<source>Proc Med Imaging 2024: Clin Biomed Imaging2024</source>. <pub-id pub-id-type="doi">10.1117/12.3006941</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Ouyang</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Q</given-names>
</name>
</person-group>. <article-title>Multi-energy CT material decomposition using graph model improved CNN</article-title>. <source>Med and Biol Eng and Comput</source> (<year>2024</year>) <volume>62</volume>(<issue>4</issue>):<fpage>1213</fpage>&#x2013;<lpage>28</lpage>. <pub-id pub-id-type="doi">10.1007/s11517-023-02986-w</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>F</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C</given-names>
</name>
<etal/>
</person-group> <article-title>MPU-net: multi-decoder U-net based on prior information for multi-material decomposition in spectral CT</article-title>. In <source>Proceedings 2023 16th international congress on image and signal processing, BioMedical engineering and informatics (CISP-BMEI)2023</source>. <publisher-name>IEEE</publisher-name> (<year>2023</year>). p. <fpage>1</fpage>&#x2013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>G</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Z</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>L</given-names>
</name>
<etal/>
</person-group> <article-title>Improved GAN: using a transformer module generator approach for material decomposition</article-title>. <source>Comput Biol Med</source> (<year>2022</year>) <volume>149</volume>(<issue>2022</issue>):<fpage>105952</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.105952</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>A</given-names>
</name>
<etal/>
</person-group> <article-title>Image domain dual material decomposition for dual&#x2010;energy CT using butterfly network</article-title>. <source>Med Phys</source> (<year>2019</year>) <volume>46</volume>(<issue>5</issue>):<fpage>2037</fpage>&#x2013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1002/mp.13489</pub-id>
</citation>
</ref>
</ref-list>
<app-group>
<app id="app1">
<title>Appendix</title>
<fig id="FA1" position="float">
<label>FIGURE A1</label>
<caption>
<p>A nosie generalization. <bold>(a)</bold> Images under three energy bands with added noise. <bold>(b)</bold> Decomposed materials of the image under three energy bands with added noise.</p>
</caption>
<graphic xlink:href="FPHY_fphy-2025-1626220_wc_app1.tif">
<alt-text content-type="machine-generated">X-ray fluorescence images depict circular patterns with varying intensities. Panel (a) shows images at energy ranges: [20,30) keV, [30,40) keV, and [40,50) keV. Panel (b) displays elemental distribution images: SiO&#x2082;, FeS&#x2082;, and CaCO&#x2083;, each showing distinct circular arrangements.</alt-text>
</graphic>
</fig>
</app>
</app-group>
</back>
</article>