<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Med. Technol.</journal-id>
<journal-title>Frontiers in Medical Technology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Med. Technol.</abbrev-journal-title>
<issn pub-type="epub">2673-3129</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmedt.2025.1621922</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Medical Technology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A<sc>TTN</sc>F<sc>NET</sc>: feature aware depth-to-pressure translation with cGAN training</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes"><name><surname>Manavar</surname><given-names>Neevkumar</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="cor1">&#x002A;</xref><uri xlink:href="https://loop.frontiersin.org/people/3033286/overview"/><role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/><role content-type="https://credit.niso.org/contributor-roles/investigation/"/><role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/><role content-type="https://credit.niso.org/contributor-roles/visualization/"/><role content-type="https://credit.niso.org/contributor-roles/software/"/><role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/><role content-type="https://credit.niso.org/contributor-roles/validation/"/><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/><role content-type="https://credit.niso.org/contributor-roles/data-curation/"/><role content-type="https://credit.niso.org/contributor-roles/methodology/"/></contrib>
<contrib contrib-type="author"><name><surname>Meyer</surname><given-names>Hanno Gerd</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/></contrib>
<contrib contrib-type="author"><name><surname>Wa&#x00DF;muth</surname><given-names>Joachim</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/><role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/></contrib>
<contrib contrib-type="author"><name><surname>Hammer</surname><given-names>Barbara</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/134498/overview" /><role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/><role content-type="https://credit.niso.org/contributor-roles/supervision/"/><role content-type="https://credit.niso.org/contributor-roles/project-administration/"/><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/><role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/></contrib>
<contrib contrib-type="author"><name><surname>Schneider</surname><given-names>Axel</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref><role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/><role content-type="https://credit.niso.org/contributor-roles/project-administration/"/><role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/><role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/><role content-type="https://credit.niso.org/contributor-roles/supervision/"/><role content-type="https://credit.niso.org/contributor-roles/investigation/"/></contrib>
</contrib-group>
<aff id="aff1"><label><sup>1</sup></label><institution>Faculty of Engineering and Mathematics, Bielefeld University of Applied Sciences</institution>, <addr-line>Bielefeld</addr-line>, <country>Germany</country></aff>
<aff id="aff2"><label><sup>2</sup></label><institution>Faculty of Technology, CITEC, Bielefeld University</institution>, <addr-line>Bielefeld</addr-line>, <country>Germany</country></aff>
<author-notes>
<fn fn-type="edited-by"><p><bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2056188/overview">Seung Kwan Kang</ext-link>, Seoul National University, Republic of Korea</p></fn>
<fn fn-type="edited-by"><p><bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1632117/overview">Hariharan Shanmugasundaram</ext-link>, Vel Tech Rangarajan Dr. Sagunthala R&#x0026;D Institute of Science and Technology, India</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2086858/overview">Joanna Przybek-Mita</ext-link>, University of Rzeszow, Poland</p></fn>
<corresp id="cor1"><label>&#x002A;</label><bold>Correspondence:</bold> Neevkumar Manavar <email>neevkumar&#x005F;hareshbhai.manavar@hsbi.de</email></corresp>
</author-notes>
<pub-date pub-type="epub"><day>16</day><month>09</month><year>2025</year></pub-date>
<pub-date pub-type="collection"><year>2025</year></pub-date>
<volume>7</volume><elocation-id>1621922</elocation-id>
<history>
<date date-type="received"><day>02</day><month>05</month><year>2025</year></date>
<date date-type="accepted"><day>26</day><month>08</month><year>2025</year></date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2025 Manavar, Meyer, Wa&#x00DF;muth, Hammer and Schneider.</copyright-statement>
<copyright-year>2025</copyright-year><copyright-holder>Manavar, Meyer, Wa&#x00DF;muth, Hammer and Schneider</copyright-holder><license license-type="open-access" xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Excessive pressure and shear forces on bedridden patients can lead to pressure injuries, particularly on those with existing ulcers. Monitoring pressure distribution is crucial for preventing such injuries by identifying high-risk areas. To address this challenge, we propose Attention Feature Network (<sc>AttnFnet</sc>), a self-attention-based deep neural network that generates pressure distribution maps from single-depth images using Conditional Generative Adversarial Network (<sc>cGAN</sc>) training. We introduce a mixed-domain SSIML2 loss function, combining structural similarity and pixel-level accuracy, along with adversarial loss, to enhance the prediction of pressure distributions for subjects lying in a bed. Evaluation results from the benchmark dataset demonstrate that the <sc>AttnFnet</sc> outperforms existing methods in terms of Structural Similarity Index Measure (SSIM) and quality analysis, providing accurate pressure distribution estimation from a single depth image.</p>
</abstract>
<kwd-group>
<kwd>patient monitoring</kwd>
<kwd>generative network</kwd>
<kwd>contact pressure prediction</kwd>
<kwd>image translation</kwd>
<kwd>deep neural network</kwd>
<kwd>transformer</kwd>
</kwd-group><counts>
<fig-count count="5"/>
<table-count count="3"/><equation-count count="84"/><ref-count count="37"/><page-count count="11"/><word-count count="0"/></counts><custom-meta-wrap><custom-meta><meta-name>section-at-acceptance</meta-name><meta-value>Medtech Data Analytics</meta-value></custom-meta></custom-meta-wrap>
</article-meta>
</front>
<body><sec id="s1" sec-type="intro"><label>1</label><title>Introduction</title>
<p>Image processing techniques have become integral to advancements in medical diagnostics and patient care. Transformer-based model architectures, such as those foundational in Natural Language Processing (NLP) (<xref ref-type="bibr" rid="B1">1</xref>) have been successfully adapted for image classification and segmentation tasks (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B3">3</xref>). However, these models typically require large datasets and significant computational resources to learn global attention patterns and image encodings. This limitation poses challenges in medical applications, where data availability and computational efficiency are critical.</p>
<p>Alternatively, Fully Convolutional Network (FCN)-based models offer computationally less intensive solutions and can provide superior feature representations in the context of limited resources and datasets (<xref ref-type="bibr" rid="B4">4</xref>). Despite these advancements, learning image representations using Convolutional Neural Network (CNN) remains complex when attempting to capture global context effectively. Incorporating attention mechanisms with CNN can address this challenge by focusing on relevant features across the entire image (<xref ref-type="bibr" rid="B5">5</xref>). Inspired by the original transformer architecture (<xref ref-type="bibr" rid="B1">1</xref>) and conditional adversarial training (<xref ref-type="bibr" rid="B6">6</xref>), we propose the Attention Feature Network (<sc>AttnFnet</sc>), a novel model that leverages a convolutional architecture to project image features while employing transformer-like attention mechanisms to obtain global feature context. Our model processes images through 12 transformer layers to generate encodings in a latent space, followed by deconvolution with skip connections back to the image space.</p>
<p>A specific use case in medical applications is studied using <sc>AttnFnet</sc>. Pressure ulcers pose a significant risk to bedridden patients, often leading to severe complications if not addressed promptly (<xref ref-type="bibr" rid="B7">7</xref>). Conventional monitoring methods can be resource-intensive or lack real-time capabilities. By predicting pressure distributions from depth images captured by an overhead camera, our approach offers a non-invasive, efficient tool for continuous patient monitoring. Our experimental results demonstrate that <sc>AttnFnet</sc> effectively predicts pressure distributions, potentially aiding in timely interventions to reposition patients and prevent pressure injuries.</p>
<p>The motivation for <sc>AttnFnet</sc> is to capture contextual features in depth images, particularly around pressure-sensitive areas at risk of developing pressure ulcers. This architecture is designed to balance computational efficiency and predictive performance, addressing the limitations of large-scale transformer-based sequence-to-sequence models, which are resource-intensive, and Fully Convolutional Network (FCN)-based encoder-decoder architectures, which often struggle with capturing long-range dependencies. The proposed Structural Similarity Index Measure L2 norm (SSIML2) loss function enables the model to minimize Mean Squared Error (MSE) more effectively than standard L2 loss alone. Additionally, the inclusion of cGAN loss constrains the network to generate contextually relevant outputs, enhancing the fidelity of the predicted pressure maps rather than promoting image diversity.</p>
<p>We evaluated our model&#x2019;s performance on depth-to-pressure image translation tasks using a publicly available benchmark dataset (<xref ref-type="bibr" rid="B8">8</xref>), with the U-Net architecture (<xref ref-type="bibr" rid="B9">9</xref>), and previous state-of-the-art BPBnet, and BPWnet (<xref ref-type="bibr" rid="B10">10</xref>) as baselines for comparison. Our results indicate that <sc>AttnFnet</sc> demonstrates promising performance in this specific medical application and shows potential for broader image translation tasks.</p>
<p>This study focuses on critical medical applications, trained on publicly available supine and lateral depth-pressure data, and possibly pinpoint high-risk tissue-loading zones in real time, thereby enabling early off-loading interventions in long-term-care and home settings.</p>
</sec>
<sec id="s2"><label>2</label><title>Related work</title>
<sec id="s2a"><label>2.1</label><title>Image generation</title>
<p>Since the introduction of Generative Adversarial Network (GAN)s by Goodfellow et al. (<xref ref-type="bibr" rid="B11">11</xref>), image generation has gained significant attention in the research community. FCN have emerged as foundational architectures for many GAN-based generation tasks due to their ability to effectively capture spatial hierarchies. Over the years, numerous GAN variants have been proposed for image generation, each enhancing different aspects of the model&#x2019;s capabilities. Noteable examples include CycleGAN (<xref ref-type="bibr" rid="B12">12</xref>), StarGAN (<xref ref-type="bibr" rid="B13">13</xref>), Least Squares GAN (<xref ref-type="bibr" rid="B14">14</xref>), StyleGAN (<xref ref-type="bibr" rid="B15">15</xref>), DCGAN (<xref ref-type="bibr" rid="B16">16</xref>), and cGAN (<xref ref-type="bibr" rid="B17">17</xref>).</p>
<p>These advancements have paved the way for more sophisticated image translation tasks. For instance, Isola et al. (<xref ref-type="bibr" rid="B6">6</xref>) demonstrated the effectiveness of conditional GANs for image-to-image translation tasks. Our proposed model builds upon these foundations by leveraging transformer-based conditional GAN training with a mixed-domain loss function to translate depth images into pressure distribution maps.</p>
</sec>
<sec id="s2b"><label>2.2</label><title>CNN architecture</title>
<p>CNNs are foundational models for vision tasks, first introduced by Lecun et al. (<xref ref-type="bibr" rid="B18">18</xref>). Their ability to learn hierarchical visual features established them as state-of-the-art for a wide range of vision applications. Prominent models such as ImageNet (<xref ref-type="bibr" rid="B19">19</xref>), VGGNet (<xref ref-type="bibr" rid="B20">20</xref>), ResNet (<xref ref-type="bibr" rid="B21">21</xref>), and MobileNet (<xref ref-type="bibr" rid="B22">22</xref>) have employed FCN architectures to capture fine-grained image features, becoming foundational in tasks like image recognition and object detection. In the domain of semantic segmentation, the work by Ronneberger et al. (<xref ref-type="bibr" rid="B9">9</xref>) made a significant contribution to FCN-based architectures. The success of U-Net in semantic segmentation and image translation has rapidly established it as a state-of-the-art model.</p>
<p>Our proposed model builds upon a CNN-based architecture and utilizes CNNs to upscale latent representations to pixel space. It leverages the computational efficiency of CNNs in vision tasks to provide an effective and efficient mechanism for upscaling latent features. This study compares the performance of the proposed method with the FCN based U-Net model.</p>
</sec>
<sec id="s2c"><label>2.3</label><title>Vision transformer</title>
<p>The introduction of transformers by Vaswani et al. (<xref ref-type="bibr" rid="B1">1</xref>) marked a paradigm shift in Natural Language Processing (NLP). The success of transformers in sequence-to-sequence tasks inspired their adaptation to computer vision, leading to the development of Vision Transformer (<sc>ViT</sc>s) (<xref ref-type="bibr" rid="B2">2</xref>). ViTs utilize self-attention mechanisms to capture long-range dependencies in images, proving particularly effective in global feature extraction (<xref ref-type="bibr" rid="B23">23</xref>). Subsequent works have explored transformer architectures for various image processing tasks, including image generation and segmentation (<xref ref-type="bibr" rid="B24">24</xref>&#x2013;<xref ref-type="bibr" rid="B26">26</xref>).</p>
<p>Kirillov et al. (<xref ref-type="bibr" rid="B3">3</xref>) and Zheng et al. (<xref ref-type="bibr" rid="B27">27</xref>) extended the transformer capabilities by combining a transformer capabilities with CNNs for segmentation tasks. The proposed model leverages a hybrid transformer-CNN architecture, utilizing CNN layers both in patch projection and as part of the feed-forward network. Additionally, it incorporates skip connections between the encoder and decoder, enhancing information flow and feature retention across the network.</p>
<p>As shown by Raghu et al. (<xref ref-type="bibr" rid="B23">23</xref>), <sc>ViT</sc>s maintain robust feature representations through attention, and transfer learning can significantly accelerate training. In line with these findings, our model employs pre-trained weights from the Segment Anything Model (SAM) (<xref ref-type="bibr" rid="B3">3</xref>) to initialize training and hence facilitating efficient convergence and improved performance.</p>
</sec>
<sec id="s2d"><label>2.4</label><title>Inferring pressure distribution</title>
<p>Several studies have focused on predicting pressure injuries in hospitalized patients. These studies have utilized statistical models and machine learning techniques to identify risk factors such as body mass index, age, gender, and comorbidities that influence the likelihood of developing pressure injuries (<xref ref-type="bibr" rid="B28">28</xref>&#x2013;<xref ref-type="bibr" rid="B31">31</xref>). While effective in risk stratification, these approaches do not provide spatially resolved information on when or where a pressure injury might occur. Hence, understanding body pressure distribution offers deeper insights into the specific locations at risk of pressure ulcer development. Clever et al. (<xref ref-type="bibr" rid="B10">10</xref>) utilized BPBnet and BPWnet to predict body pressure distribution using a depth camera, demonstrating the potential of non-invasive monitoring techniques.</p>
<p>Building upon this concept, our approach leverages a transformer-based GAN architecture trained on real-world data with various human poses (<xref ref-type="bibr" rid="B8">8</xref>) to predict pressure distributions from depth images. Unlike prior methods, our model incorporates attention mechanisms to improve results on pressure-sensitive areas and adversarial training to enhance prediction accuracy and spatial distribution.</p>
</sec>
</sec>
<sec id="s3" sec-type="methods"><label>3</label><title>Methods</title>
<p>This section provides a detailed description of the proposed <sc>AttnFnet</sc> architecture, training objectives, training strategy, and evaluation metrics. We begin by outlining the structure of the <sc>AttnFnet</sc> model, including its image encoder, bottleneck layer, and decoder, and then explain how the model is trained using a conditional GAN framework. We also describe the metrics used to evaluate its performance in terms of both pixel-level accuracy and perceptual quality.</p>
<sec id="s3a"><label>3.1</label><title><sc>AttnFnet</sc> architecture</title>
<p>The <sc>AttnFnet</sc> architecture is designed to translate depth into pressure distribution maps. <xref ref-type="fig" rid="F1">Figure 1</xref> describes overall architecture and it consist of three primary components: 1. an <italic>image encoder</italic> that encodes the image into a latent space, 2. a <italic>bottleneck layer</italic> that reduces computational complexity while preserving crucial features, and 3. a <italic>decoder</italic> that reconstructs the image encodings back into the original image space. Additionally, skip connections are introduced from the encoder to the decoder to preserve contextual features during the reconstruction process.</p>
<fig id="F1" position="float"><label>Figure 1</label>
<caption><p>Schematic representation of the <sc>AttnFnet</sc> model architecture. <bold>(A)</bold> An example architecture for a <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM1"><mml:mn>128</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>54</mml:mn></mml:math></inline-formula> input image. The input image is projected to a 712-dimensional embedding via a convolution operation. Positional embeddings are added to these projections before being processed by the transformer block, and the output is passed through the decoder block with skip connections. <bold>(B)</bold> Transformer block, where the input undergoes a standard multi-head self-attention mechanism followed by convolutional projections. <bold>(C)</bold> Decoder block schematic, where the output from the transformer encoder passes through multiple up-convolution layers, progressively increasing resolution until the desired output size is reached, with skip connections added to the deconvolution blocks. <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM2"><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>6</mml:mn></mml:math></inline-formula>, <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM3"><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>8</mml:mn></mml:math></inline-formula>, and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM4"><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>10</mml:mn></mml:math></inline-formula> indicate the number of transformer blocks whose output is used.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fmedt-07-1621922-g001.tif"><alt-text content-type="machine-generated">Diagram showing three models: (a) A schematic with input images processed through convolutional linear projections, transformer blocks, and a decoder block with skip connections. (b) A transformer block is detailed with layers including self multi-headed attention and layer normalization. (c) A decoder block with multiple convolutional transformations and skip connections, indicating processing stages for an input depth image.</alt-text>
</graphic>
</fig>
<sec id="s3a1"><label>3.1.1</label><title>Image encoder design</title>
<p>In the image encoder, the input image is first divided into patches, which are then processed through convolutional projections. These projections are followed by the addition of sinusoidal positional embeddings to retain spatial information Vaswani et al. (<xref ref-type="bibr" rid="B1">1</xref>). The patched image features are subsequently passed through 12 transformer blocks that perform self-attention and convolution operations to encode the image, capturing both local and global dependencies.</p>
<p>Formally, the self-attention is defined in <xref ref-type="disp-formula" rid="disp-formula1">Equation 1</xref><disp-formula id="disp-formula1"><label>(1)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM1"><mml:mtext>Attention</mml:mtext><mml:mspace width="thickmathspace" /><mml:mo stretchy="false">(</mml:mo><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>Z</mml:mi><mml:mo>+</mml:mo><mml:mtext>softmax</mml:mtext><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>Q</mml:mi><mml:msup><mml:mi>K</mml:mi><mml:mi>T</mml:mi></mml:msup></mml:mrow><mml:msqrt><mml:msub><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:msqrt></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mi>V</mml:mi></mml:math></disp-formula>where <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM5"><mml:mi>Z</mml:mi></mml:math></inline-formula> is the input patch from the previous layer, <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM6"><mml:mi>Q</mml:mi></mml:math></inline-formula>, <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM7"><mml:mi>K</mml:mi></mml:math></inline-formula>, and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM8"><mml:mi>V</mml:mi></mml:math></inline-formula> represent the query, key, and value vectors, and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM9"><mml:msub><mml:mi>d</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:math></inline-formula> is the dimensionality of the key vector.</p>
<p>The outputs of the self-attention mechanism are concatenated to form the multi-head attention (MHA) (as shown in the <xref ref-type="disp-formula" rid="disp-formula2">Equation 2</xref>)<disp-formula id="disp-formula2"><label>(2)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM2"><mml:mtext>MHA</mml:mtext><mml:mspace width="thickmathspace" /><mml:mo stretchy="false">(</mml:mo><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mtext>Concat</mml:mtext><mml:mspace width="thickmathspace" /><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mtext>Head</mml:mtext><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mtext>Head</mml:mtext><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mtext>Head</mml:mtext><mml:mi>n</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>where each <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM10"><mml:msub><mml:mtext>Head</mml:mtext><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> is computed as in <xref ref-type="disp-formula" rid="disp-formula1">Equation 1</xref>.</p>
<p>In the standard transformer block, the MHA output is typically passed through a <italic>multi-layer perceptron</italic> (<italic>MLP</italic>) followed by a residual connection:<disp-formula id="disp-formula3"><label>(3)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM3"><mml:msub><mml:mrow><mml:mi mathvariant="normal">ViT</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">mlp</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="normal">MLP</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="normal">MHA</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">MHA</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mo stretchy="false">(</mml:mo><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>However, in <sc>AttnFnet</sc>, we replace the <italic>MLP</italic> with convolutional projections, allowing the encoder to refine features more quickly while maintaining spatial hierarchies:<disp-formula id="disp-formula4"><label>(4)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM4"><mml:msub><mml:mrow><mml:mi mathvariant="normal">ViT</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">conv</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="normal">Conv</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="normal">MHA</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="normal">MHA</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>Skip connections are introduced between intermediate transformer layers and the decoder block to help retain high-resolution details.</p>
<p>Both model variants were evaluated:</p>
<p>
<list list-type="simple">
<list-item><label>&#x2022;</label>
<p><bold>ViT-mlp</bold>: AttnFnet with an MLP feed-forward network in the transformer block, as shown in <xref ref-type="disp-formula" rid="disp-formula3">Equation 3</xref>.</p></list-item>
<list-item><label>&#x2022;</label>
<p><bold><sc>AttnFnet</sc></bold>: AttnFnet with convolutional projections in the transformer block, as shown in <xref ref-type="disp-formula" rid="disp-formula4">Equation 4</xref>.</p></list-item>
</list></p>
</sec>
<sec id="s3a2"><label>3.1.2</label><title>Image decoder</title>
<p>The decoder reconstructs the encoded image representations by upsampling them through successive deconvolution layers. These layers progressively increase the spatial resolution until the original input size is restored. To preserve critical image details, skip connections from the encoder are incorporated, allowing the decoder to combine low-level feature maps with upsampled features and enhance high-resolution reconstruction. Unlike the encoder, the decoder is designed to be lightweight, focusing on upsampling the encoded features.</p>
</sec>
</sec>
<sec id="s3b"><label>3.2</label><title>Training objective</title>
<p>The training objective is inspired by the Pix2Pix framework (<xref ref-type="bibr" rid="B6">6</xref>), where we employ a conditional GAN (cGAN) architecture with a PatchGAN discriminator. The PatchGAN discriminator distinguishes between real and generated image pairs, ensuring that local image details are accurately predicted while maintaining global consistency in the generated pressure maps.</p>
<p>The total training objective aims to optimize both the discriminator and generator losses. The discriminator loss <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM11"><mml:msub><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mi>D</mml:mi></mml:msub></mml:math></inline-formula> is defined in the <xref ref-type="disp-formula" rid="disp-formula5">Equation 5</xref>.<disp-formula id="disp-formula5"><label>(5)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM5"><mml:mtable columnalign="right left" rowspacing=".5em" columnspacing="thickmathspace" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mi>D</mml:mi></mml:msub></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mtext>real</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>D</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">]</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">[</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mtext>real</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>D</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">]</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd /><mml:mtd><mml:mspace width="1em" /><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mtext>gen</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>D</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>G</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">]</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">[</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mtext>gen</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>D</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>G</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">]</mml:mo><mml:mo stretchy="false">]</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM12"><mml:mi>x</mml:mi></mml:math></inline-formula> is the input depth image, <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM13"><mml:mi>y</mml:mi></mml:math></inline-formula> is the ground truth pressure distribution map, and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM14"><mml:mi>G</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is the generated pressure map from the generator. The first two terms evaluate how well the discriminator identifies real image-label pairs, while the last two terms penalize the discriminator for misclassifying generated pressure distribution maps as real. Here, <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM15"><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mtext>real</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> refers to the label for real pressure maps, and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM16"><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mtext>gen</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> refers to the label for generated pressure maps.</p>
<p>The generator loss <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM17"><mml:msub><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mi>G</mml:mi></mml:msub></mml:math></inline-formula> combines the adversarial loss with perceptual loss (as shown in <xref ref-type="disp-formula" rid="disp-formula6">Equation 6</xref>), encouraging the generated images to be both realistic and similar to the ground truth:<disp-formula id="disp-formula6"><label>(6)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM6"><mml:msub><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mi>G</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">[</mml:mo><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>D</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>G</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">]</mml:mo><mml:mo>+</mml:mo><mml:mi>&#x03BB;</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mi>S</mml:mi><mml:mi>S</mml:mi><mml:mi>I</mml:mi><mml:mi>M</mml:mi><mml:mi>L</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo stretchy="false">]</mml:mo></mml:math></disp-formula>Here, <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM18"><mml:mi>&#x03BB;</mml:mi></mml:math></inline-formula> is a regularization constant that balances the contributions of the adversarial and perceptual losses.</p>
<p>The perceptual similarity L2 loss <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM19"><mml:msub><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mtext>SSIML2</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> combines the Structural Similarity Index Measure (SSIM) loss with the mean squared error (MSE) loss:<disp-formula id="disp-formula7"><label>(7)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM7"><mml:msub><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mtext>SSIML2</mml:mtext><mml:mspace width="thickmathspace" /></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mi>&#x03B1;</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mtext>SSIM</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo>,</mml:mo><mml:mi>G</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mi>y</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>G</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:msubsup><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup></mml:math></disp-formula>where <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM20"><mml:mi>&#x03B1;</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM21"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> are weighting factors for the SSIM and MSE components, respectively.</p>
<p>The SSIM between two images <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM22"><mml:mi>a</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM23"><mml:mi>b</mml:mi></mml:math></inline-formula> is defined as:<disp-formula id="disp-formula8"><label>(8)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM8"><mml:mtext>SSIM</mml:mtext><mml:mspace width="thickmathspace" /><mml:mo stretchy="false">(</mml:mo><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>2</mml:mn><mml:msub><mml:mi>&#x03BC;</mml:mi><mml:mi>a</mml:mi></mml:msub><mml:msub><mml:mi>&#x03BC;</mml:mi><mml:mi>b</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>C</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>2</mml:mn><mml:msub><mml:mi>&#x03C3;</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>b</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>C</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>&#x03BC;</mml:mi><mml:mi>a</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>&#x03BC;</mml:mi><mml:mi>b</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:msub><mml:mi>C</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>&#x03C3;</mml:mi><mml:mi>a</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>&#x03C3;</mml:mi><mml:mi>b</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:msub><mml:mi>C</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula>where:
<list list-type="simple">
<list-item><label>&#x2022;</label>
<p><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM24"><mml:msub><mml:mi>&#x03BC;</mml:mi><mml:mi>a</mml:mi></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM25"><mml:msub><mml:mi>&#x03BC;</mml:mi><mml:mi>b</mml:mi></mml:msub></mml:math></inline-formula> are the mean values of <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM26"><mml:mi>a</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM27"><mml:mi>b</mml:mi></mml:math></inline-formula>, respectively.</p></list-item>
<list-item><label>&#x2022;</label>
<p><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM28"><mml:msubsup><mml:mi>&#x03C3;</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM29"><mml:msubsup><mml:mi>&#x03C3;</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msubsup></mml:math></inline-formula> are the variances of <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM30"><mml:mi>a</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM31"><mml:mi>b</mml:mi></mml:math></inline-formula>.</p></list-item>
<list-item><label>&#x2022;</label>
<p><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM32"><mml:msub><mml:mi>&#x03C3;</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>b</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> represents the covariance between <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM33"><mml:mi>a</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM34"><mml:mi>b</mml:mi></mml:math></inline-formula>.</p></list-item>
<list-item><label>&#x2022;</label>
<p><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM35"><mml:msub><mml:mi>C</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM36"><mml:msub><mml:mi>C</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:math></inline-formula> are constants to stabilize the division when the denominator is small.</p></list-item>
</list>By combining SSIM with pixel-level MSE loss, the model is encouraged to maintain structural similarity while optimizing pixel-wise accuracy, which helps to produce more perceptually faithful reconstructions.</p>
</sec>
<sec id="s3c"><label>3.3</label><title>Training strategy</title>
<sec id="s3c1"><label>3.3.1</label><title>Dataset</title>
<p>The proposed model was evaluated on an open-source dataset from Liu et al. (<xref ref-type="bibr" rid="B8">8</xref>). The dataset includes depth images of 102 healthy subjects (28 female) in 45 unique poses, each lying on a hospital bed. The poses are classified into three primary postures: supine, left-side lateral, and right-side lateral. The data were split into training (data from <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM37"><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>61</mml:mn></mml:math></inline-formula> subjects), validation (data from <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM38"><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>20</mml:mn></mml:math></inline-formula> subjects), and test sets (data from <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM39"><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>21</mml:mn></mml:math></inline-formula> subjects). The training data did not include poses with blanket covers or synthetic data.</p>
<p>The model used only depth information to predict pressure distributions and did not utilize any <xref ref-type="sec" rid="s14">Supplementary Material</xref> from the dataset. However, the model uses Occlusion Free Depth Images (OFDI), which are noise-free, cropped depth images containing all data points from the human surface (<xref ref-type="bibr" rid="B32">32</xref>), and Pre-processed Pressure Distribution (<sc>PPress</sc>). The <sc>PPress</sc> involves reducing the image resolution to <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM40"><mml:mn>27</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>64</mml:mn></mml:math></inline-formula> and applying a Gaussian filter (<inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM41"><mml:mi>&#x03C3;</mml:mi><mml:mo>=</mml:mo><mml:mn>1.4</mml:mn></mml:math></inline-formula>) to diminish noise and smooth the pressure images (<xref ref-type="bibr" rid="B33">33</xref>). This preprocessing step is conducted to assess its impact on prediction accuracy and to facilitate comparison with (<xref ref-type="bibr" rid="B10">10</xref>).</p>
</sec>
<sec id="s3c2"><label>3.3.2</label><title>Training settings</title>
<p>All networks were trained using the same settings, except for the learning rate <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM42"><mml:mi>&#x03B7;</mml:mi></mml:math></inline-formula>. The Adam optimizer (<xref ref-type="bibr" rid="B34">34</xref>) was employed for optimization, using a learning rate of <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM43"><mml:mi>&#x03B7;</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> for the U-Net model and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM44"><mml:mi>&#x03B7;</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> for <sc>AttnFnet</sc>. The initial decay rates (<inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM45"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula>) for the Adam optimizer were set to <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM46"><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>=</mml:mo><mml:mn>0.5</mml:mn></mml:math></inline-formula> and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM47"><mml:msub><mml:mi>&#x03B2;</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>=</mml:mo><mml:mn>0.999</mml:mn></mml:math></inline-formula>. All the optimizer parameters were the same for the discriminator and generator. All models were trained until 90 epochs with a batch size of 1.</p>
<p>For conditional GAN training, a regularization constant <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM48"><mml:mi>&#x03BB;</mml:mi><mml:mo>=</mml:mo><mml:mn>100</mml:mn></mml:math></inline-formula> was used in the generator loss, with weighting factors <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM49"><mml:mi>&#x03B1;</mml:mi><mml:mo>=</mml:mo><mml:mn>300</mml:mn></mml:math></inline-formula> and <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM50"><mml:mi>&#x03B2;</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> in the perceptual similarity L2 loss (<xref ref-type="disp-formula" rid="disp-formula7">Equation 7</xref>). Since image generation tasks are generally more challenging than image classification tasks, label smoothing was applied to reduce the confidence of the discriminator, setting the label for generated pressure distribution maps to <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM51"><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mtext>gen</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>0.1</mml:mn></mml:math></inline-formula> and the label for real distribution maps to <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM52"><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mtext>real</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>0.9</mml:mn></mml:math></inline-formula>.</p>
</sec>
<sec id="s3c3"><label>3.3.3</label><title>Evaluation metrics</title>
<p>
<list list-type="simple">
<list-item><label>&#x2022;</label>
<p><bold>Pixel Prediction Accuracy (PPA)</bold>: Pixel Prediction Accuracy (PPA) is described by the ratio of the total correctly predicted pixels to the total number of pixels <xref ref-type="disp-formula" rid="disp-formula9">Equation 9</xref>.<disp-formula id="disp-formula9"><label>(9)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM9"><mml:mtext>PPA</mml:mtext><mml:mo>=</mml:mo><mml:mrow><mml:mfrac><mml:mtext>Number of True Predictions</mml:mtext><mml:mtext>Number of Total Pixels</mml:mtext></mml:mfrac></mml:mrow></mml:math></disp-formula></p></list-item>
<list-item><label>&#x2022;</label>
<p><bold>Structural Similarity Index Measure (SSIM)</bold>: Structural Similarity Index Measure (SSIM) is defined in <xref ref-type="disp-formula" rid="disp-formula8">Equation 8</xref>.</p></list-item>
<list-item><label>&#x2022;</label>
<p><bold>Fr&#x00E9;chet Inception Distance (FID)</bold>: Defined by Heusel et al. (<xref ref-type="bibr" rid="B35">35</xref>). Fr&#x00E9;chet Inception Distance (FID) is calculated from the features, extracted using the pre-trained inception-V3 model trained on the imagenet dataset.</p></list-item>
<list-item><label>&#x2022;</label>
<p><bold>MSE</bold>: Calculates the average squared difference between the estimated values <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM53"><mml:msub><mml:mrow><mml:mover><mml:mi>Y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> and the actual values <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM54"><mml:msub><mml:mrow><mml:mi>Y</mml:mi></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> across all the data points <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM55"><mml:mi>n</mml:mi></mml:math></inline-formula>, <xref ref-type="disp-formula" rid="disp-formula10">Equation 10</xref>.<disp-formula id="disp-formula10"><label>(10)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM10"><mml:mrow><mml:mi mathvariant="normal">MSE</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mfrac></mml:mrow><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:munderover><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>Y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:math></disp-formula></p></list-item>
<list-item><label>&#x2022;</label>
<p><bold>Peak Signal-to-Noise Ratio (PSNR)</bold>: PSNR Measures the ratio between the maximum possible power of a signal and the power of corrupting noise, defined in (<xref ref-type="bibr" rid="B36">36</xref>).</p></list-item>
<list-item><label>&#x2022;</label>
<p><bold>Posture Intersection Over Union (IOU)</bold>: The largest area of pressure higher than the threshold in actual pressure distribution is <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM56"><mml:msub><mml:mi>A</mml:mi><mml:mi>y</mml:mi></mml:msub></mml:math></inline-formula> and the largest area of pressure exceeding the threshold in predicted pressure distribution is <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM57"><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:msub></mml:math></inline-formula>. posture Intersection Over Union (IOU) is defined by <xref ref-type="disp-formula" rid="disp-formula11">Equation 11</xref>.<disp-formula id="disp-formula11"><label>(11)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM11"><mml:mi>I</mml:mi><mml:mi>O</mml:mi><mml:mi>U</mml:mi><mml:mspace width="thickmathspace" /><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mi>y</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mi>A</mml:mi><mml:mi>y</mml:mi></mml:msub><mml:mo>&#x2229;</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>A</mml:mi><mml:mi>y</mml:mi></mml:msub><mml:mo>&#x222A;</mml:mo><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula></p></list-item>
</list>The metrics&#x2013;Mean Pixel Prediction Accuracy (MPPA), Mean Structural Similarity Index (MSSIM), Mean Fr&#x00E9;chet Inception Distance (MFID), MSE, Mean Peak-Peak Signal-to-Noise Ratio (MPSNR), and Posture Mean Intersection Over Union (MIOU) are the average values across the test data. These metrics provide a comprehensive evaluation of the models in terms of both pixel-level accuracy and perceptual quality.</p>
</sec>
</sec>
</sec>
<sec id="s4" sec-type="results"><label>4</label><title>Results</title>
<p>We evaluated the performance of the proposed AttnFnet model and compared it with implementations of U-Net, BPBnet, and BPWnet (<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B10">10</xref>). The variation of AttnFnet&#x2014;ViT-mlp was also assessed to determine the impact of the convolutional projections in the transformer blocks.</p>
<sec id="s4a"><label>4.1</label><title>Quantitative evaluation</title>
<p><xref ref-type="table" rid="T1">Table&#x00A0;1</xref> compares the MPPA, MSSIM, MFID, MSE, and MPSNR scores calculated on test data from U-Net and AttnFnet model predictions. The results indicate that <sc>AttnFnet</sc> achieves higher MSSIM and MPSNR scores, as well as lower MSE scores, compared to U-Net, ViT-mlp, BPBnet, and BPWnet. Notably, <sc>AttnFnet</sc> outperforms U-Net by 15&#x0025; in terms of MSE.</p>
<table-wrap id="T1" position="float"><label>Table 1</label>
<caption><p>MPPA, MSSIM, MFID, MSE, and MPSNR metrics comparison with the state-of-the-art on the test data.</p></caption>
<table frame="hsides" rules="groups">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th valign="top" align="left">Model</th>
<th valign="top" align="center">MPPA</th>
<th valign="top" align="center">MSSIM</th>
<th valign="top" align="center">MFID</th>
<th valign="top" align="center">MSE</th>
<th valign="top" align="center">MPSNR</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">U-Net</td>
<td valign="top" align="center"><bold>0.6658</bold></td>
<td valign="top" align="center">0.7958</td>
<td valign="top" align="center">0.4615</td>
<td valign="top" align="center">0.000433</td>
<td valign="top" align="center">34.4185</td>
</tr>
<tr>
<td valign="top" align="left"><sc>AttnFnet</sc></td>
<td valign="top" align="center">0.6142</td>
<td valign="top" align="center"><bold>0.8291</bold></td>
<td valign="top" align="center">0.3475</td>
<td valign="top" align="center"><bold>0.000368</bold></td>
<td valign="top" align="center"><bold>35.0508</bold></td>
</tr>
<tr>
<td valign="top" align="left">ViT-mlp</td>
<td valign="top" align="center">0.5112</td>
<td valign="top" align="center">0.7968</td>
<td valign="top" align="center"><bold>0.2393</bold></td>
<td valign="top" align="center">0.000426</td>
<td valign="top" align="center">34.2621</td>
</tr>
<tr>
<td valign="top" align="left">BPBnet (<xref ref-type="bibr" rid="B10">10</xref>)</td>
<td valign="top" align="center">0.0078</td>
<td valign="top" align="center">0.0204</td>
<td valign="top" align="center">160.58</td>
<td valign="top" align="center">0.00567</td>
<td valign="top" align="center">22.5927</td>
</tr>
<tr>
<td valign="top" align="left">BPWnet (<xref ref-type="bibr" rid="B10">10</xref>)</td>
<td valign="top" align="center">0.5244</td>
<td valign="top" align="center">0.6331</td>
<td valign="top" align="center">1.6335</td>
<td valign="top" align="center">0.00405</td>
<td valign="top" align="center">24.1364</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values denote the best score for each metric.</p></fn>
</table-wrap-foot>
</table-wrap>
<p><xref ref-type="fig" rid="F2">Figure&#x00A0;2</xref> presents box plots of the FID, MSE, PPA, and SSIM metrics for the U-Net, <sc>AttnFnet</sc>, ViT-MLP, BPBnet, and BPWnet. <sc>AttnFnet</sc> shows a narrower Interquartile Range (IQR) and lower median values in MSE, indicating more consistent performance. U-Net demonstrates higher median and IQR in PPA, suggesting superior pixel-level accuracy. However, <sc>AttnFnet</sc> achieves better SSIM scores, reflecting higher structural similarity with the actual pressure distributions. <sc>AttnFnet</sc> version of ViT-mlp has a lower MFID score, but <sc>AttnFnet</sc> has a narrower IQR than any other method. The proposed methodology outperforms BPBnet and BPWnet in all metrics.</p>
<fig id="F2" position="float"><label>Figure 2</label>
<caption><p>Box plot representation of FID, MSE, PPA, and SSIM metric scores obtained from test predictions.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fmedt-07-1621922-g002.tif"><alt-text content-type="machine-generated">Box plot comparing five metrics: FID, MSE, PPA, SSIM, and PSNR across five models: Unet, AttnFnet, ViT-mlp, BPBnet, and BPWnet. Each plot shows distribution and variance, with BPWnet generally having higher variability.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s4b"><label>4.2</label><title>Effects of image pre-processing</title>
<p>The pressure distributions were converted to kPa by multiplying calibration factors from the dataset (<xref ref-type="bibr" rid="B8">8</xref>) with the pressure distributions, and the MSE was recalculated. <xref ref-type="table" rid="T2">Table&#x00A0;2</xref> presents the overall MSE across the test dataset for models trained on three cases: 1. raw depth images as input and raw pressure images as ground truth, 2. Occlusion Free Depth Images (OFDI) inputs, and 3. combined OFDI input with <sc>PPress</sc> ground truth.</p>
<table-wrap id="T2" position="float"><label>Table 2</label>
<caption><p>Overall MSE comparison of U-Net, <sc>AttnFnet</sc>, and ViT-mlp model predictions on test subjects, with results compared against BPWnet and BPBnet models proposed by Clever et al. (<xref ref-type="bibr" rid="B10">10</xref>). Models were trained on three different cases: 1. raw depth input with raw pressure ground truth, 2. OFDI input with raw pressure ground truth, and 3. combined OFDI input with <sc>PPress</sc> ground truth. MSE values are derived from rescaled pressure distributions in kPa.</p></caption>
<table frame="hsides" rules="groups">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th valign="top" align="left">Model</th>
<th valign="top" align="center">OFDI</th>
<th valign="top" align="center">PPress</th>
<th valign="top" align="center">MSE <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM58"><mml:mo stretchy="false">&#x2193;</mml:mo></mml:math></inline-formula> (<inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM59"><mml:msup><mml:mtext>kPa</mml:mtext><mml:mn>2</mml:mn></mml:msup></mml:math></inline-formula>)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">U-Net</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">2.7871</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="center"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM60"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula></td>
<td valign="top" align="center"/>
<td valign="top" align="center">2.5694</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="center"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM61"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM62"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula></td>
<td valign="top" align="center">0.7950</td>
</tr>
<tr>
<td valign="top" align="left"><sc>AttnFnet</sc></td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">2.5354</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="center"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM63"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula></td>
<td valign="top" align="center"/>
<td valign="top" align="center">2.3333</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="center"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM64"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM65"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula></td>
<td valign="top" align="center"><bold>0.6884</bold></td>
</tr>
<tr>
<td valign="top" align="left">ViT-mlp</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">2.6614</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="center"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM66"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula></td>
<td valign="top" align="center"/>
<td valign="top" align="center">2.5023</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="center"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM67"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM68"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula></td>
<td valign="top" align="center">0.8091</td>
</tr>
<tr>
<td valign="top" align="left">BPBnet (<xref ref-type="bibr" rid="B10">10</xref>)</td>
<td valign="top" align="center"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM69"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM70"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula></td>
<td valign="top" align="center">0.772</td>
</tr>
<tr>
<td valign="top" align="left">BPWnet (<xref ref-type="bibr" rid="B10">10</xref>)</td>
<td valign="top" align="center"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM71"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM72"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula></td>
<td valign="top" align="center">1.155</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values denote the best score for each metric.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>Using Occlusion Free Depth Images (OFDI) and Pre-processed Pressure Distribution (<sc>PPress</sc>) resulted in a 33&#x0025; greater reduction in error compared to using raw depth images. Notably, <sc>AttnFnet</sc> achieved better results in this scenario.</p>
</sec>
<sec id="s4c"><label>4.3</label><title>Qualitative analysis</title>
<p><xref ref-type="fig" rid="F3">Figure&#x00A0;3</xref> shows the average deviations for three different postures&#x2014;supine, lateral left-side, and lateral right-side -, comparing the U-Net, <sc>AttnFnet</sc>, and ViT-mlp models. Absolute deviations were calculated by taking the absolute pressure difference between the actual and predicted pressure distribution and averaging it over the test dataset.</p>
<fig id="F3" position="float"><label>Figure 3</label>
<caption><p>Visual representation of the pressure deviations in supine, left-side lateral, and right-side lateral postures. The heat map is constrained between pressure deviation values of 0 and 1.5&#x2009;kPa.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fmedt-07-1621922-g003.tif"><alt-text content-type="machine-generated">Three heatmaps compare pressure distribution in different sleeping positions: supine, leftside, and rightside, using Unet, AttnFnet, and ViT-mlp models. The pressure is measured in kilopascals, represented by a color scale from blue (low) to red (high).</alt-text>
</graphic>
</fig>
<p>A visual comparison of the predicted pressure distributions using U-Net, <sc>AttnFnet</sc>, ViT-mlp, BPBnet, and BPWnet models, against the reference pressure images, shown in <xref ref-type="fig" rid="F4">Figure&#x00A0;4</xref>. <sc>AttnFnet</sc> produced more accurate posture representations compared to U-Net, ViT-mlp, BPBnet, and BPWnet. <sc>AttnFnet</sc>&#x2019;s predictions were more closely aligned with the actual pressure distribution. U-Net often struggled with pressure distribution on the leg and head side, while ViT-mlp tended to predict higher pressure values around the edges of the human body. BPBnet produces blurry results due to its pixel loss reduction, while BPBnet doesn&#x2019;t produce blurry results but overestimates pressure values and couldn&#x2019;t outperform <sc>AttnFnet</sc>.</p>
<fig id="F4" position="float"><label>Figure 4</label>
<caption><p>Visual representation of the predicted pressure distributions using five different models and their comparison to the reference pressure image (Ref. PImg). Occlusion Free Depth Images (OFDI)s were used as input to the models. Each row represents a different depth input to the models. In the pressure distribution images, blue indicates low-pressure regions, and red indicates high-pressure regions. In the depth images, red indicates higher depth and blue indicates lower depth values.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fmedt-07-1621922-g004.tif"><alt-text content-type="machine-generated">A series of grid images showing depth analysis and neural network outputs. Each row includes an input depth image followed by output images from different models: Unet, AttnNet, ViT-mlp, BPBnet, and Ref. PImg. The images display human figures in various poses, with color variations indicating depth or feature emphasis. The input images are in red tones, while the outputs are in blue with brighter spots highlighting significant areas.</alt-text>
</graphic>
</fig>
<p>Notably, all models consistently overestimated pressure values compared to the actual distribution in the facial and pelvic regions.</p>
</sec>
<sec id="s4d"><label>4.4</label><title>Weight estimation</title>
<p>By using the predicted pressure distributions and the known area of each sensor, the normal force on the mattress was calculated (see <xref ref-type="sec" rid="s14">Supplementary Material, Section 1</xref>). This force provided an approximate estimate of the test subjects&#x2019; weights. <xref ref-type="fig" rid="F5">Figure&#x00A0;5</xref> shows scatter plots comparing the estimated weights of each participant based on actual and predicted pressure distributions from the proposed models.</p>
<fig id="F5" position="float"><label>Figure 5</label>
<caption><p>Scatter plots representing the errors in estimated weights (kg) of test subjects. Comparison between calculated weights (kg) from predicted pressure distributions, calculated weights from actual pressure distributions (kg) (Black points), and actual measured weights (kg) (red dashed line). <bold>(a)</bold> Estimated weights using raw depth images as input. <bold>(b)</bold> Estimated weights using cleaned depth images (OFDI) as input to the proposed models.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fmedt-07-1621922-g005.tif"><alt-text content-type="machine-generated">Two scatter plots compare predicted and measured weights in kilograms. Plot (a) shows four prediction models: Unet, AttnFnet, ViT-mlp, and calculated, with varying trends against measured weights. Plot (b) features similar predictions but shows different trends. A red dashed line represents measured weights in both plots.</alt-text>
</graphic>
</fig>
<p><xref ref-type="fig" rid="F5">Figure&#x00A0;5</xref> shows that the use of OFDIs improves the performance of <sc>AttnFnet</sc> and ViT-mlp, leading to more accurate pressure distributions and better weight estimations, as evidenced by the fitted line of <sc>AttnFnet</sc>&#x2019;s estimated weights.</p>
<p><xref ref-type="table" rid="T3">Table&#x00A0;3</xref> shows <sc>AttnFnet</sc> performs best in Posture MIOU while BPWnet gives better weight estimation among all models.</p>
<table-wrap id="T3" position="float"><label>Table 3</label>
<caption><p>Mean absolute weight difference between the calculated weight from the predicted pressure profile and the actual measured weight. Weight is computed using both raw and OFDI inputs with the U-Net, <sc>AttnFnet</sc>, and ViT-mlp models. The last column shows the Posture Mean Intersection Over Union (MIOU) from predictions using each method.</p></caption>
<table frame="hsides" rules="groups">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th valign="top" align="left" rowspan="2">Method</th>
<th valign="top" align="center" colspan="2">Mean absolute weight difference (kg)</th>
<th valign="top" align="center" rowspan="2">Posture MIOU</th>
</tr>
<tr>
<th valign="top" align="center">Raw input</th>
<th valign="top" align="center">OFDI input</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">U-Net</td>
<td valign="top" align="center"><bold>12.65</bold></td>
<td valign="top" align="center">12.30</td>
<td valign="top" align="center">0.7346</td>
</tr>
<tr>
<td valign="top" align="left"><sc>AttnFnet</sc></td>
<td valign="top" align="center">16.50</td>
<td valign="top" align="center">6.71</td>
<td valign="top" align="center"><bold>0.7515</bold></td>
</tr>
<tr>
<td valign="top" align="left">ViT-mlp</td>
<td valign="top" align="center">21.63</td>
<td valign="top" align="center">12.19</td>
<td valign="top" align="center">0.4910</td>
</tr>
<tr>
<td valign="top" align="left">BPBnet (<xref ref-type="bibr" rid="B10">10</xref>)</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">0.7329</td>
</tr>
<tr>
<td valign="top" align="left">BPWnet (<xref ref-type="bibr" rid="B10">10</xref>)</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center"><bold>5.64</bold></td>
<td valign="top" align="center">0.6566</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values denote the best score for each metric.</p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec id="s5" sec-type="discussion"><label>5</label><title>Discussion</title>
<p>The proposed <sc>AttnFnet</sc> model effectively infers body pressure distribution from a single depth image. The <sc>AttnFnet</sc> architecture leverages self-attention mechanisms to generate more refined features during image encoding in latent space, offering improved performance over U-Net. The results demonstrate that the proposed method outperforms state-of-the-art methods.</p>
<sec id="s5a"><label>5.1</label><title>Effectiveness of SSIML2 loss function</title>
<p>The use of the combined Structural Similarity Index Measure and L2 norm loss (SSIML2 loss) provided stable training and better performance. When the model was trained using only the L2 norm loss with adversarial loss, it exhibited signs of mode collapse, and the validation MSE loss started increasing after 40 epochs when the MSE could not be reduced further (see <xref ref-type="sec" rid="s14">Supplementary Material, Section 2</xref>). Training with the L2 norm loss resulted in a 130&#x0025; increase in MSE and a 31.17&#x0025; reduction in SSIM compared to the model trained with SSIML2 loss.</p>
</sec>
<sec id="s5b"><label>5.2</label><title>Robustness to noisy data</title>
<p>As shown in <xref ref-type="table" rid="T2">Table&#x00A0;2</xref>, the proposed model successfully generated pressure distributions even from noisy raw data, with significantly reduced error when using OFDI and <sc>PPress</sc>. The ability to handle raw depth images and generation of pressure distribution without introducing blurring demonstrates the robustness of the proposed method (more in <xref ref-type="sec" rid="s14">Supplementary Material, Section 2</xref>). This suggests that while the model is capable of learning from noisy input, preprocessing steps can enhance its predictive accuracy.</p>
</sec>
<sec id="s5c"><label>5.3</label><title>Plausibility of pressure distributions</title>
<p>The results from <xref ref-type="table" rid="T3">Table&#x00A0;3</xref> and <xref ref-type="fig" rid="F3">Figures&#x00A0;3</xref>&#x2013;<xref ref-type="fig" rid="F5">5</xref>, show that <sc>AttnFnet</sc>&#x2019;s attention over features helps the model produce more plausible feature distributions compared to other models. In <xref ref-type="fig" rid="F4">Figure&#x00A0;4</xref>, <sc>AttnFnet</sc> outperforms other methods in terms of posture representation and visual accuracy of the pressure distributions. Specifically, in <xref ref-type="fig" rid="F3">Figure&#x00A0;3</xref> it is evident that near the hip and head areas&#x2014;where all methods tend to overestimate pressure values&#x2014;<sc>AttnFnet</sc> tends to reduce overestimation.</p>
<p>Moreover, while <xref ref-type="table" rid="T3">Table&#x00A0;3</xref> and <xref ref-type="fig" rid="F5">Figure&#x00A0;5</xref> show that weight estimation from U-Net predictions does not improve significantly with preprocessed inputs, <sc>AttnFnet</sc>&#x2019;s performance increases notably. This indicates that <sc>AttnFnet</sc> learns the relationship between depth representation and pressure distribution more effectively through its attention mechanism. However, calculated weights from all methods exhibit some scatter and do not outperform the BPWnet from Clever et al. (<xref ref-type="bibr" rid="B10">10</xref>). This disparity is because Clever et al. (<xref ref-type="bibr" rid="B10">10</xref>) utilized a separate pre-trained network &#x201C;Betanet,&#x201D; to estimate the mass and height of the subject and incorporate this information into the loss function to improve results. In contrast, our method does not use any <xref ref-type="sec" rid="s14">Supplementary Material</xref> during training and relies solely on features from Occlusion Free Depth Images (OFDI).</p>
<p><xref ref-type="table" rid="T3">Table&#x00A0;3</xref> also shows the mean posture Intersection over Union (IOU), with the ViT-mlp method having the lowest score. The ViT-mlp variant tends to generate higher pressure values at the edges of the human posture, resulting in a visual representation that appears wider than the reference image. This is evident in <xref ref-type="fig" rid="F3">Figures&#x00A0;3</xref>, <xref ref-type="fig" rid="F4">4</xref>.</p>
<p>As shown in <xref ref-type="fig" rid="F4">Figure&#x00A0;4</xref>, BPBnet exhibits blurred predictions due to its training strategy based on pixel reduction losses (L1 and L2 losses). This approach tends to average pixel values, which can result in improved MSE performance but fails to yield better results across other evaluation metrics. In contrast, BPWnet does not exhibit blurring; however, it tends to overestimate pressure values compared to the actual distributions and fails to generate postures superior to those of the AttnFnet model, as evident in <xref ref-type="fig" rid="F4">Figure&#x00A0;4</xref>.</p>
</sec>
<sec id="s5d"><label>5.4</label><title>Model performance and capabilities</title>
<p>The proposed model achieved better performance across several evaluation metrics, including MFID, MSSIM, MSE, and MPSNR, compared to previous methods. Among the variants of <sc>AttnFnet</sc>, the ViT-mlp version showed the best MFID score. This improvement is partly due to how the FID score is calculated, which heavily depends on the specific version of the ImageNet dataset and the pre-trained Inception-V3 model employed for feature extraction. FID measures how closely the generated images resemble real ones by comparing high-level features, focusing on the mean and covariance of these features in both real and generated images. However, a lower FID score does not necessarily indicate identical pressure distributions; it also accounts for the diversity of generated data (<xref ref-type="bibr" rid="B35">35</xref>). Therefore, it is most reliable when evaluating realistic RGB images.</p>
<p>The self-attention mechanism in the <sc>AttnFnet</sc> model captures meaningful relationships between feature embeddings, producing features that encompass both local and global information. Skip connections in the architecture help the model retain high-resolution features and improve performance by facilitating gradient flow and feature reuse (see <xref ref-type="sec" rid="s14">Supplementary Material, Section 2</xref>). The proposed method initializes attention weights from Segment Anything Model (SAM) (<xref ref-type="bibr" rid="B3">3</xref>), which aids better weight initialization even though Segment Anything Model (SAM) was trained on a different objective. While we did not perform a comparative analysis of the model&#x2019;s performance without transfer learning, prior work by Raghu et al. (<xref ref-type="bibr" rid="B23">23</xref>) supports the argument by comparing <sc>ViT</sc>s to ResNet models with and without pretrained weights.</p>
<p>Despite the slower learning rate, <sc>AttnFnet</sc> achieved a lower validation loss faster than U-Net (see <xref ref-type="sec" rid="s14">Supplementary Material, Section 2</xref>). This suggests that the transformer/based architecture of <sc>AttnFnet</sc> is more efficient in capturing the complex relationships in the data, even with a reduced learning rate.</p>
<p>Overall, the experimental results validate that the <sc>AttnFnet</sc> model gives better performance in inferring pressure distributions from depth images. The incorporation of the SSIML2 loss function, robustness to noisy data, and effective use of self-attention mechanisms contribute to the model&#x2019;s improved accuracy and reliability. Additional performance measures can be found in the <xref ref-type="sec" rid="s14">Supplementary Material</xref>.</p>
</sec>
</sec>
<sec id="s6"><label>6</label><title>Future work and limitations</title>
<p>Although the proposed method outperforms other models still lacks clinical validation and can generate certain data dependency. To generalize the model and reduce data dependency, future work involves the collection of diverse datasets with patients and healthy controls.</p>
<p>Challenging errors, such as a person having a lipoma beneath the skin tissue or a very complex human posture, may cause the model to predict inaccurate pressure distributions. The authors expect future work towards incorporating physical plausibility constraints and informed learning approaches during training to reduce errors and ensure physically plausible pressure distributions.</p>
<p>The proposed model can be adapted for generalized image translation tasks. The authors expect future work toward the evaluation of the proposed method compared to state-of-the-art image translation methods.</p>
<p>Model employs <sc>cGAN</sc> to improve pressure prediction; however, GANs are sensitive towards hyperparameters and difficult to train. The authors will guide future work towards, conditional diffusion process to improve prediction even further.</p>
</sec>
<sec id="s7" sec-type="conclusions"><label>7</label><title>Conclusion</title>
<p>In conclusion, we have proposed a self-attention-based deep neural network, <sc>AttnFnet</sc>, to translate depth images into pressure images. We evaluated two variations of the proposed architecture&#x2014;ViT-mlp and <sc>AttnFnet</sc>&#x2014;against state-of-the-art methods. The proposed method outperforms the existing methods, achieving 91&#x0025; reduction in MSE and 30&#x0025; increment in MSSIM score compared to the state-of-the-art BPWnet. It also outperforms existing methods in qualitative analysis of the uncovered systematic lying postures of the real test subjects, demonstrating its potential for accurate pressure distribution prediction from depth images.</p>
<p>These findings can help detect and prevent early pressure ulcers by identifying risk areas of a patient lying on a bed. The current publicly available dataset is limited to supine and lateral postures; so future works involve extending it towards prone and sitting postures to cover diverse risk-affected areas.</p>
</sec>
</body>
<back>
<sec id="s8" sec-type="data-availability"><title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary Material/Github Repository (<xref ref-type="bibr" rid="B37">37</xref>), further inquiries can be directed to the corresponding author/s.</p>
</sec>
<sec id="s9" sec-type="ethics-statement"><title>Ethics statement</title>
<p>Ethical review and approval was not required for the study on human participants in accordance with the local legislation and institutional requirements. Written informed consent from the [patients/participants OR patients/participants legal guardian/next of kin] was not required to participate in this study in accordance with the national legislation and the institutional requirements.</p>
</sec>
<sec id="s10" sec-type="author-contributions"><title>Author contributions</title>
<p>NM: Writing &#x2013; original draft, Investigation, Formal analysis, Visualization, Software, Conceptualization, Validation, Writing &#x2013; review &#x0026; editing, Data curation, Methodology. HM: Writing &#x2013; review &#x0026; editing. JW: Writing &#x2013; review &#x0026; editing, Conceptualization. BH: Conceptualization, Supervision, Project administration, Writing &#x2013; review &#x0026; editing, Funding acquisition. AS: Funding acquisition, Project administration, Writing &#x2013; review &#x0026; editing, Conceptualization, Supervision, Investigation.</p>
</sec>
<sec id="s11" sec-type="funding-information"><title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This research was carried out within the framework of the project &#x201C;SAIL: SustAInable Lifecycle of Intelligent SocioTechnical Systems.&#x201D; SAIL is receiving funding from the program &#x201C;Netzwerke 2021,&#x201D; an initiative of the Ministry of Culture and Science of the State of North Rhine-Westphalia (Grant No.: NW21-059B).</p>
</sec>
<ack><title>Acknowledgments</title>
<p>The authors would like to acknowledge Dr. Matthias Fricke and David Pelkmann from the Center for Applied Data Science (CfADS) at Bielefeld University of Applied Sciences for providing access to the GPU compute cluster.</p>
</ack>
<sec id="s12" sec-type="COI-statement"><title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s13" sec-type="ai-statement"><title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec id="s15" sec-type="disclaimer"><title>Publisher&#x0027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s14" sec-type="supplementary-material"><title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmedt.2025.1621922/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmedt.2025.1621922/full&#x0023;supplementary-material</ext-link></p>
<supplementary-material id="SD1" content-type="local-data">
<media mimetype="application" mime-subtype="pdf" xlink:href="Datasheet1.pdf"/></supplementary-material>
</sec>
<ref-list><title>References</title>
<ref id="B1"><label>1.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Vaswani</surname><given-names>A</given-names></name><name><surname>Shazeer</surname><given-names>N</given-names></name><name><surname>Parmar</surname><given-names>N</given-names></name><name><surname>Uszkoreit</surname><given-names>J</given-names></name><name><surname>Jones</surname><given-names>L</given-names></name><name><surname>Gomez</surname><given-names>AN</given-names></name></person-group>, et al. <article-title>Attention is all you need. In: Guyon I, Von Luxburg U, Bengio S, Wallach H, Fergus R, Vishwanathan S, editors. <italic>Advances in Neural Information Processing Systems</italic>. Red Hook, NY: Curran Associates, Inc (2017). Vol. 30</article-title>.</citation></ref>
<ref id="B2"><label>2.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Dosovitskiy</surname><given-names>A</given-names></name><name><surname>Beyer</surname><given-names>L</given-names></name><name><surname>Kolesnikov</surname><given-names>A</given-names></name><name><surname>Weissenborn</surname><given-names>D</given-names></name><name><surname>Zhai</surname><given-names>X</given-names></name><name><surname>Unterthiner</surname><given-names>T</given-names></name></person-group>, et al. <article-title>An image is worth 16 &#x00D7; 16 words: transformers for image recognition at scale. In: <italic>International Conference on Learning Representation (ICLR)</italic>. (2020)</article-title>.</citation></ref>
<ref id="B3"><label>3.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Kirillov</surname><given-names>A</given-names></name><name><surname>Mintun</surname><given-names>E</given-names></name><name><surname>Ravi</surname><given-names>N</given-names></name><name><surname>Mao</surname><given-names>H</given-names></name><name><surname>Rolland</surname><given-names>C</given-names></name><name><surname>Gustafson</surname><given-names>L</given-names></name></person-group>, et al. <article-title>Segment anything. In: <italic>IEEE/CVF International Conference on Computer Vision (ICCV)</italic>. (2023). p. 3992&#x2013;4003</article-title>.</citation></ref>
<ref id="B4"><label>4.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Lu</surname><given-names>Z</given-names></name><name><surname>Xie</surname><given-names>H</given-names></name><name><surname>Liu</surname><given-names>C</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name></person-group>. <article-title>Bridging the gap between vision transformers and convolutional neural networks on small datasets. In: Koyejo S, Mohamed S, Agarwal A, Belgrave D, Cho K, Oh A, editors. <italic>Advances in Neural Information Processing Systems</italic>. Red Hook, NY: Curran Associates, Inc (2022). Vol. 35. p. 14663&#x2013;77</article-title>.</citation></ref>
<ref id="B5"><label>5.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tian</surname><given-names>C</given-names></name><name><surname>Xu</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>Z</given-names></name><name><surname>Zuo</surname><given-names>W</given-names></name><name><surname>Fei</surname><given-names>L</given-names></name><name><surname>Liu</surname><given-names>H</given-names></name></person-group>. <article-title>Attention-guided CNN for image denoising</article-title>. <source>Neural Netw</source>. (<year>2020</year>) <volume>124</volume>:<fpage>117</fpage>&#x2013;<lpage>29</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2019.12.024</pub-id><pub-id pub-id-type="pmid">31991307</pub-id></citation></ref>
<ref id="B6"><label>6.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Isola</surname><given-names>P</given-names></name><name><surname>Zhu</surname><given-names>JY</given-names></name><name><surname>Zhou</surname><given-names>T</given-names></name><name><surname>Efros</surname><given-names>AA</given-names></name></person-group>. <article-title>Image-to-image translation with conditional adversarial networks. In: <italic>IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</italic>. IEEE (2017)</article-title>.</citation></ref>
<ref id="B7"><label>7.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kottner</surname><given-names>J</given-names></name><name><surname>Cuddigan</surname><given-names>J</given-names></name><name><surname>Carville</surname><given-names>K</given-names></name><name><surname>Balzer</surname><given-names>K</given-names></name><name><surname>Berlowitz</surname><given-names>D</given-names></name><name><surname>Law</surname><given-names>S</given-names></name></person-group>, et al. <article-title>Prevention and treatment of pressure ulcers/injuries: the protocol for the second update of the international clinical practice guideline 2019</article-title>. <source>J Tissue Viability</source>. (<year>2019</year>) <volume>28</volume>:<fpage>51</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1016/j.jtv.2019.01.001</pub-id><pub-id pub-id-type="pmid">30658878</pub-id></citation></ref>
<ref id="B8"><label>8.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>S</given-names></name><name><surname>Huang</surname><given-names>X</given-names></name><name><surname>Fu</surname><given-names>N</given-names></name><name><surname>Li</surname><given-names>C</given-names></name><name><surname>Su</surname><given-names>Z</given-names></name><name><surname>Ostadabbas</surname><given-names>S</given-names></name></person-group>. <article-title>Simultaneously-collected multimodal lying pose dataset: enabling in-bed human pose monitoring</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. (<year>2023</year>) <volume>45</volume>:<fpage>1106</fpage>&#x2013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2022.3155712.</pub-id><pub-id pub-id-type="pmid">35239476</pub-id></citation></ref>
<ref id="B9"><label>9.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Ronneberger</surname><given-names>O</given-names></name><name><surname>Fischer</surname><given-names>P</given-names></name><name><surname>Brox</surname><given-names>T</given-names></name></person-group>. <article-title>U-net: convolutional networks for biomedical image segmentation. In: <italic>Medical Image Computing and Computer-Assisted Intervention &#x2013; MICCAI 2015</italic>. Springer International Publishing (2015). p. 234&#x2013;41</article-title>.</citation></ref>
<ref id="B10"><label>10.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Clever</surname><given-names>HM</given-names></name><name><surname>Grady</surname><given-names>PL</given-names></name><name><surname>Turk</surname><given-names>G</given-names></name><name><surname>Kemp</surname><given-names>CC</given-names></name></person-group>. <article-title>Bodypressure - inferring body pose and contact pressure from a depth image</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. (<year>2023</year>) <volume>45</volume>:<fpage>137</fpage>&#x2013;<lpage>53</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2022.3158902</pub-id><pub-id pub-id-type="pmid">35344483</pub-id></citation></ref>
<ref id="B11"><label>11.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Goodfellow</surname><given-names>I</given-names></name><name><surname>Pouget-Abadie</surname><given-names>J</given-names></name><name><surname>Mirza</surname><given-names>M</given-names></name><name><surname>Xu</surname><given-names>B</given-names></name><name><surname>Warde-Farley</surname><given-names>D</given-names></name><name><surname>Ozair</surname><given-names>S</given-names></name></person-group>, et al. <article-title>Generative adversarial nets. In: Ghahramani Z, Welling M, Cortes C, Lawrence N, Weinberger KQ, editors. <italic>Advances in Neural Information Processing Systems NIPS</italic>. Red Hook, NY: Curran Associates, Inc (2014). p. 2672&#x2013;80</article-title>.</citation></ref>
<ref id="B12"><label>12.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Zhu</surname><given-names>JY</given-names></name><name><surname>Park</surname><given-names>T</given-names></name><name><surname>Isola</surname><given-names>P</given-names></name><name><surname>Efros</surname><given-names>AA</given-names></name></person-group>. <article-title>Unpaired image-to-image translation using cycle-consistent adversarial networks. In: <italic>2017 IEEE International Conference on Computer Vision (ICCV)</italic>. (2017). p. 2242&#x2013;51</article-title>.</citation></ref>
<ref id="B13"><label>13.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Choi</surname><given-names>Y</given-names></name><name><surname>Choi</surname><given-names>M</given-names></name><name><surname>Kim</surname><given-names>M</given-names></name><name><surname>Ha</surname><given-names>JW</given-names></name><name><surname>Kim</surname><given-names>S</given-names></name><name><surname>Choo</surname><given-names>J</given-names></name></person-group>. <article-title>Stargan: unified generative adversarial networks for multi-domain image-to-image translation. In: <italic>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</italic>. (2018). p. 8789&#x2013;97</article-title>.</citation></ref>
<ref id="B14"><label>14.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Mao</surname><given-names>X</given-names></name><name><surname>Li</surname><given-names>Q</given-names></name><name><surname>Xie</surname><given-names>H</given-names></name><name><surname>Lau</surname><given-names>RY</given-names></name><name><surname>Wang</surname><given-names>Z</given-names></name><name><surname>Smolley</surname><given-names>SP</given-names></name></person-group>. <article-title>Least squares generative adversarial networks. In: <italic>2017 IEEE International Conference on Computer Vision (ICCV)</italic>. Los Alamitos, CA, USA: IEEE Computer Society (2017). p. 2813&#x2013;21</article-title>.</citation></ref>
<ref id="B15"><label>15.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Karras</surname><given-names>T</given-names></name><name><surname>Laine</surname><given-names>S</given-names></name><name><surname>Aila</surname><given-names>T</given-names></name></person-group>. <article-title>A style-based generator architecture for generative adversarial networks</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. (<year>2021</year>) <volume>43</volume>:<fpage>4217</fpage>&#x2013;<lpage>28</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2020.2970919</pub-id><pub-id pub-id-type="pmid">32012000</pub-id></citation></ref>
<ref id="B16"><label>16.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Radford</surname><given-names>A</given-names></name><name><surname>Metz</surname><given-names>L</given-names></name><name><surname>Chintala</surname><given-names>S</given-names></name></person-group>. <article-title>Unsupervised representation learning with deep convolutional generative adversarial networks. In: <italic>International Conference on Learning Representations</italic>. (2016)</article-title>.</citation></ref>
<ref id="B17"><label>17.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Mirza</surname><given-names>M</given-names></name><name><surname>Osindero</surname><given-names>S</given-names></name></person-group>. <article-title>Conditional generative adversarial nets. <italic>CoRR</italic> [Preprint]. <italic>abs/1411.1784</italic> (2014)</article-title>.</citation></ref>
<ref id="B18"><label>18.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lecun</surname><given-names>Y</given-names></name><name><surname>Bottou</surname><given-names>L</given-names></name><name><surname>Bengio</surname><given-names>Y</given-names></name><name><surname>Haffner</surname><given-names>P</given-names></name></person-group>. <article-title>Gradient-based learning applied to document recognition</article-title>. <source>Proc IEEE</source>. (<year>1998</year>) <volume>86</volume>:<fpage>2278</fpage>&#x2013;<lpage>324</lpage>. <pub-id pub-id-type="doi">10.1109/5.726791.</pub-id></citation></ref>
<ref id="B19"><label>19.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Deng</surname><given-names>J</given-names></name><name><surname>Dong</surname><given-names>W</given-names></name><name><surname>Socher</surname><given-names>R</given-names></name><name><surname>Li</surname><given-names>LJ</given-names></name><name><surname>Li</surname><given-names>K</given-names></name><name><surname>Fei-Fei</surname><given-names>L</given-names></name></person-group>. <article-title>Imagenet: a large-scale hierarchical image database. In: <italic>IEEE Conference on Computer Vision and Pattern Recognition</italic>. (2009). p. 248&#x2013;55</article-title>.</citation></ref>
<ref id="B20"><label>20.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>S</given-names></name><name><surname>Deng</surname><given-names>W</given-names></name></person-group>. <article-title>Very deep convolutional neural network based image classification using small training sample size. In: <italic>3rd IAPR Asian Conference on Pattern Recognition (ACPR)</italic>. (2015). p. 730&#x2013;4</article-title>.</citation></ref>
<ref id="B21"><label>21.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>He</surname><given-names>K</given-names></name><name><surname>Zhang</surname><given-names>X</given-names></name><name><surname>Ren</surname><given-names>S</given-names></name><name><surname>Sun</surname><given-names>J</given-names></name></person-group>. <article-title>Deep residual learning for image recognition. In: <italic>IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</italic>. (2016). p. 770&#x2013;8</article-title>.</citation></ref>
<ref id="B22"><label>22.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Howard</surname><given-names>AG</given-names></name><name><surname>Zhu</surname><given-names>M</given-names></name><name><surname>Chen</surname><given-names>B</given-names></name><name><surname>Kalenichenko</surname><given-names>D</given-names></name><name><surname>Wang</surname><given-names>W</given-names></name><name><surname>Weyand</surname><given-names>T</given-names></name></person-group>, et al. <article-title>Mobilenets: efficient convolutional neural networks for mobile vision applications. <italic>arXiv</italic> [Preprint]. <italic>arXiv:1704.04861</italic> (2017)</article-title>.</citation></ref>
<ref id="B23"><label>23.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Raghu</surname><given-names>M</given-names></name><name><surname>Unterthiner</surname><given-names>T</given-names></name><name><surname>Kornblith</surname><given-names>S</given-names></name><name><surname>Zhang</surname><given-names>C</given-names></name><name><surname>Dosovitskiy</surname><given-names>A</given-names></name></person-group>. <article-title>Do vision transformers see like convolutional neural networks? In: Ranzato M, Beygelzimer A, Dauphin Y, Liang PS, Vaughan JW, editors. <italic>Advances in Neural Information Processing Systems</italic>. Red Hook, NY: Curran Associates, Inc (2021). Vol. 34. p. 12116&#x2013;28</article-title>.</citation></ref>
<ref id="B24"><label>24.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Parmar</surname><given-names>N</given-names></name><name><surname>Vaswani</surname><given-names>A</given-names></name><name><surname>Uszkoreit</surname><given-names>J</given-names></name><name><surname>Kaiser</surname><given-names>L</given-names></name><name><surname>Shazeer</surname><given-names>N</given-names></name><name><surname>Ku</surname><given-names>A</given-names></name></person-group>, et al. <article-title>Image transformer. In: <italic>International Conference on Machine Learning</italic>. PMLR (2018). p. 4055&#x2013;64</article-title>.</citation></ref>
<ref id="B25"><label>25.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>Z</given-names></name><name><surname>Lin</surname><given-names>Y</given-names></name><name><surname>Cao</surname><given-names>Y</given-names></name><name><surname>Hu</surname><given-names>H</given-names></name><name><surname>Wei</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>Z</given-names></name></person-group>, et al. <article-title>Swin transformer: hierarchical vision transformer using shifted windows. In: <italic>IEEE/CVF International Conference on Computer Vision (ICCV)</italic>. (2021). p. 9992&#x2013;10002</article-title>.</citation></ref>
<ref id="B26"><label>26.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>W</given-names></name><name><surname>Xie</surname><given-names>E</given-names></name><name><surname>Li</surname><given-names>X</given-names></name><name><surname>Fan</surname><given-names>DP</given-names></name><name><surname>Song</surname><given-names>K</given-names></name><name><surname>Liang</surname><given-names>D</given-names></name></person-group>, et al. <article-title>Pyramid vision transformer: a versatile backbone for dense prediction without convolutions. In: <italic>IEEE/CVF International Conference on Computer Vision (ICCV)</italic>. (2021). p. 548&#x2013;58</article-title>.</citation></ref>
<ref id="B27"><label>27.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Zheng</surname><given-names>S</given-names></name><name><surname>Lu</surname><given-names>J</given-names></name><name><surname>Zhao</surname><given-names>H</given-names></name><name><surname>Zhu</surname><given-names>X</given-names></name><name><surname>Luo</surname><given-names>Z</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name></person-group>, et al. <article-title>Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers. In: <italic>IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</italic>. (2021). p. 6877&#x2013;86</article-title>.</citation></ref>
<ref id="B28"><label>28.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alderden</surname><given-names>J</given-names></name><name><surname>Pepper</surname><given-names>GA</given-names></name><name><surname>Wilson</surname><given-names>A</given-names></name><name><surname>Whitney</surname><given-names>JD</given-names></name><name><surname>Richardson</surname><given-names>S</given-names></name><name><surname>Butcher</surname><given-names>R</given-names></name></person-group>, et al. <article-title>Predicting pressure injury in critical care patients: a machine-learning model</article-title>. <source>Am J Crit Care</source>. (<year>2018</year>) <volume>27</volume>:<fpage>461</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.4037/ajcc2018525</pub-id><pub-id pub-id-type="pmid">30385537</pub-id></citation></ref>
<ref id="B29"><label>29.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ladios-Martin</surname><given-names>M</given-names></name><name><surname>Fern&#x00E1;ndez-de Maya</surname><given-names>J</given-names></name><name><surname>Ballesta-L&#x00F3;pez</surname><given-names>FJ</given-names></name><name><surname>Belso-Garzas</surname><given-names>A</given-names></name><name><surname>Mas-Asencio</surname><given-names>M</given-names></name><name><surname>Caba&#x00F1;ero-Mart&#x00ED;nez</surname><given-names>MJ</given-names></name></person-group>. <article-title>Predictive modeling of pressure injury risk in patients admitted to an intensive care unit</article-title>. <source>Am J Crit Care</source>. (<year>2020</year>) <volume>29</volume>:<fpage>e70</fpage>&#x2013;<lpage>e80</lpage>. <pub-id pub-id-type="doi">10.4037/ajcc2020237</pub-id><pub-id pub-id-type="pmid">32607572</pub-id></citation></ref>
<ref id="B30"><label>30.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aloweni</surname><given-names>F</given-names></name><name><surname>Ang</surname><given-names>SY</given-names></name><name><surname>Fook-Chong</surname><given-names>S</given-names></name><name><surname>Agus</surname><given-names>N</given-names></name><name><surname>Yong</surname><given-names>P</given-names></name><name><surname>Goh</surname><given-names>MM</given-names></name></person-group>, et al. <article-title>A prediction tool for hospital-acquired pressure ulcers among surgical patients: surgical pressure ulcer risk score</article-title>. <source>Int Wound J</source>. (<year>2019</year>) <volume>16</volume>:<fpage>164</fpage>&#x2013;<lpage>75</lpage>. <pub-id pub-id-type="doi">10.1111/iwj.13007</pub-id><pub-id pub-id-type="pmid">30289624</pub-id></citation></ref>
<ref id="B31"><label>31.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cramer</surname><given-names>EM</given-names></name><name><surname>Seneviratne</surname><given-names>MG</given-names></name><name><surname>Sharifi</surname><given-names>H</given-names></name><name><surname>Ozturk</surname><given-names>A</given-names></name><name><surname>Hernandez-Boussard</surname><given-names>T</given-names></name></person-group>. <article-title>Predicting the incidence of pressure ulcers in the intensive care unit using machine learning</article-title>. <source>EGEMS (Wash DC)</source>. (<year>2019</year>) <volume>7</volume>:<fpage>49</fpage>. <pub-id pub-id-type="doi">10.5334/egems.307</pub-id><pub-id pub-id-type="pmid">31534981</pub-id></citation></ref>
<ref id="B32"><label>32.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Clever</surname><given-names>H</given-names></name></person-group>. <article-title>Data from: SLP real cleaned up and reconstructed images (2021)</article-title>. <pub-id pub-id-type="doi">10.7910/DVN/ZS7TQS</pub-id></citation></ref>
<ref id="B33"><label>33.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Clever</surname><given-names>HM</given-names></name><name><surname>Erickson</surname><given-names>Z</given-names></name><name><surname>Kapusta</surname><given-names>A</given-names></name><name><surname>Turk</surname><given-names>G</given-names></name><name><surname>Liu</surname><given-names>CK</given-names></name><name><surname>Kemp</surname><given-names>CC</given-names></name></person-group>. <article-title>Bodies at rest: 3D human pose and shape estimation from a pressure image using synthetic data. In: <italic>IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</italic>. (2020). p. 6214&#x2013;23</article-title>.</citation></ref>
<ref id="B34"><label>34.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Kingma</surname><given-names>DP</given-names></name><name><surname>Ba</surname><given-names>J</given-names></name></person-group>. <article-title>Adam: a method for stochastic optimization. In: Bengio Y, LeCun Y, editors. <italic>3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7&#x2013;9, 2015, Conference Track Proceedings</italic>. (2015)</article-title>.</citation></ref>
<ref id="B35"><label>35.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Heusel</surname><given-names>M</given-names></name><name><surname>Ramsauer</surname><given-names>H</given-names></name><name><surname>Unterthiner</surname><given-names>T</given-names></name><name><surname>Nessler</surname><given-names>B</given-names></name><name><surname>Hochreiter</surname><given-names>S</given-names></name></person-group>. <article-title>Gans trained by a two time-scale update rule converge to a local nash equilibrium. In: <italic>Proceedings of the 31st International Conference on Neural Information Processing Systems</italic>. Curran Associates Inc. (2017). NIPS&#x2019;17. p. 6629&#x2013;40</article-title>.</citation></ref>
<ref id="B36"><label>36.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sethi</surname><given-names>D</given-names></name><name><surname>Bharti</surname><given-names>S</given-names></name><name><surname>Prakash</surname><given-names>C</given-names></name></person-group>. <article-title>A comprehensive survey on gait analysis: history, parameters, approaches, pose estimation, and future work</article-title>. <source>Artif Intell Med</source>. (<year>2022</year>) <volume>129</volume>:<fpage>102314</fpage>. <pub-id pub-id-type="doi">10.1016/j.artmed.2022.102314</pub-id><pub-id pub-id-type="pmid">35659390</pub-id></citation></ref>
<ref id="B37"><label>37.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Manavar</surname><given-names>N</given-names></name><name><surname>Meyer</surname><given-names>HG</given-names></name><name><surname>Schneider</surname><given-names>A</given-names></name></person-group>. <article-title>Data from: Attnfnet: model to translate depth to pressure images (2025)</article-title>. <pub-id pub-id-type="doi">10.5281/zenodo.15174067</pub-id></citation></ref></ref-list>
</back>
</article>