<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurorobot.</journal-id>
<journal-title>Frontiers in Neurorobotics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurorobot.</abbrev-journal-title>
<issn pub-type="epub">1662-5218</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnbot.2025.1630728</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A robust and effective framework for 3D scene reconstruction and high-quality rendering in nasal endoscopy surgery</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Ji</surname> <given-names>Xueqin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Zhao</surname> <given-names>Shuting</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2661836/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Liu</surname> <given-names>Di</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Wang</surname> <given-names>Feng</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2420093/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Chen</surname> <given-names>Xinrong</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/657738/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>The Third School of Clinical Medicine, Ningxia Medical University</institution>, <addr-line>Yinchuan</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Ultrasound, Peking University First Hospital Ningxia Women and Children&#x00027;s Hospital</institution>, <addr-line>Yinchuan</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Fudan University Academy for Engineering and Technology</institution>, <addr-line>Shanghai</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Shanghai Key Laboratory of Medical Imaging Computing and Computer Assisted Intervention</institution>, <addr-line>Shanghai</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>Department of Hepatobiliary Surgery, General Hospital of Ningxia Medical University</institution>, <addr-line>Yinchuan</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Hu Cao, Technical University of Munich, Germany</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Yinlong Liu, City University of Macau, Macao SAR, China</p>
<p>Kexue Fu, Shandong Academy of Sciences, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Feng Wang <email>wf4065335&#x00040;163.com</email></corresp>
<corresp id="c002">Xinrong Chen <email>chenxinrong&#x00040;fudan.edu.cn</email></corresp>
<fn fn-type="equal" id="fn001"><p>&#x02020;These authors have contributed equally to this work</p></fn></author-notes>
<pub-date pub-type="epub">
<day>27</day>
<month>06</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>19</volume>
<elocation-id>1630728</elocation-id>
<history>
<date date-type="received">
<day>18</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>29</day>
<month>05</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Ji, Zhao, Liu, Wang and Chen.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Ji, Zhao, Liu, Wang and Chen</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>In nasal endoscopic surgery, the narrow nasal cavity restricts the surgical field of view and the manipulation of surgical instruments. Therefore, precise real-time intraoperative navigation, which can provide precise 3D information, plays a crucial role in avoiding critical areas with dense blood vessels and nerves. Although significant progress has been made in endoscopic 3D reconstruction methods, their application in nasal scenarios still faces numerous challenges. On the one hand, there is a lack of high-quality, annotated nasal endoscopy datasets. On the other hand, issues such as motion blur and soft tissue deformations complicate the nasal endoscopy reconstruction process. To tackle these challenges, a series of nasal endoscopy examination videos are collected, and the pose information for each frame is recorded. Additionally, a novel model named Mip-EndoGS is proposed, which integrates 3D Gaussian Splatting for reconstruction and rendering and a diffusion module to reduce image blurring in endoscopic data. Meanwhile, by incorporating an adaptive low-pass filter into the rendering pipeline, the aliasing artifacts (jagged edges) are mitigated, which occur during the rendering process. Extensive quantitative and visual experiments show that the proposed model is capable of reconstructing 3D scenes within the nasal cavity in real-time, thereby offering surgeons more detailed and precise information about the surgical scene. Moreover, the proposed approach holds great potential for integration with AR-based surgical navigation systems to enhance intraoperative guidance.</p></abstract>
<kwd-group>
<kwd>nasal endoscopy</kwd>
<kwd>3D reconstruction</kwd>
<kwd>3D Gaussian Splatting</kwd>
<kwd>diffusion model</kwd>
<kwd>anti-aliasing</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="3"/>
<equation-count count="18"/>
<ref-count count="38"/>
<page-count count="11"/>
<word-count count="6508"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>The demand for endoscopes in transnasal surgery is growing, both for endoscopic examination and endoscopic surgery. For example, according to statistics, rhinosinusitis (RS), an inflammatory disease of the nasal cavity and paranasal sinuses, affects approximately one-six of adults in the United States, resulting in over 30 million diagnoses annually (Wyler and Mallon, <xref ref-type="bibr" rid="B34">2019</xref>; Rosenfeld et al., <xref ref-type="bibr" rid="B26">2015</xref>). Functional endoscopic sinus surgery (FESS), a common method for treating RS, involves inserting a slender endoscope into the nasal cavity to enter the sinus. The endoscope that enters the cavity provides the doctor with a clear field of view, which helps to accurately locate the lesion.</p>
<p>Endoscopy, compared to CT imaging, not only has a lower cost and no radiation, but also better real-time performance, which helps doctors accurately understand the relationship between target lesions and critical anatomical structures (M&#x000FC;nzer et al., <xref ref-type="bibr" rid="B21">2018</xref>; Pownell et al., <xref ref-type="bibr" rid="B24">1997</xref>). However, mainstream monocular endoscopy cannot obtain depth information about the internal structure of the nasal cavity, which limits its application in endoscopic examination and endoscopic surgery. Therefore, surface reconstruction from endoscopic sequences enables doctors to obtain 3D information of the internal structure of the nasal cavity, which will better facilitate examination decisions and guide surgical operations.</p>
<p>Structure from Motion (SfM) (Snavely et al., <xref ref-type="bibr" rid="B29">2006</xref>) and Simultaneous Localization and Mapping (SLAM) (Grasa et al., <xref ref-type="bibr" rid="B7">2013</xref>; Mur-Artal et al., <xref ref-type="bibr" rid="B22">2015</xref>) are widely used in depth estimation of endoscopic images, which recover 3D structures by tracking the position of feature points in different images. Widya et al. (<xref ref-type="bibr" rid="B33">2019</xref>) investigated how to utilize SfM to overcome the challenge of reconstructing gastric shapes from texture-limited endoscopic images. Leonard et al. (<xref ref-type="bibr" rid="B15">2016</xref>) studied an image-enhanced endoscopic navigation method based on the SfM algorithm to improve the accuracy and safety of functional endoscopic sinus surgery. Wang et al. (<xref ref-type="bibr" rid="B31">2020</xref>) proposed a bronchoscope enhancement scheme based on visual SLAM, which achieved the reconstruction of feature point models and improved navigation performance; (Mahmoud et al., <xref ref-type="bibr" rid="B17">2017</xref>) successfully stabilized the tracking of endoscope position by combining monocular endoscopy with ORB-SLAM, and successfully repositioned it after tracking loss.</p>
<p>In recent years, neural rendering (Kato et al., <xref ref-type="bibr" rid="B10">2018</xref>; Tewari et al., <xref ref-type="bibr" rid="B30">2020</xref>; Mildenhall et al., <xref ref-type="bibr" rid="B19">2021</xref>) used differentiable rendering and neural networks, surpassing the limited performance of traditional 3D reconstruction. For instance, Wang et al. (<xref ref-type="bibr" rid="B32">2022</xref>) utilized dynamic neural radiance fields to represent deformable surgical scenes and explored the potential of neural rendering in 3D reconstruction of surgical scenes. Batlle et al. (<xref ref-type="bibr" rid="B2">2023</xref>) introduced LightNeus, which combines neural implicit surface reconstruction technology with photometric models of light sources to achieve 3D reconstruction of the entire colon segment. Chen P. et al. (<xref ref-type="bibr" rid="B3">2024</xref>) first utilized Neural Radiance Fields (NeRF) (Mildenhall et al., <xref ref-type="bibr" rid="B19">2021</xref>) to achieve 3D reconstruction of dynamic cystoscopic examination scenes, which can recover scenes under limited perspectives and features, alleviating texture loss problem that traditional algorithms may encounter.</p>
<p>Furthermore, doctors can observe lesion areas from different perspectives through 3D reconstruction using videos obtained from endoscopic examinations, aiding in formulating more precise surgical plans and predicting surgical difficulty and risks. During surgery, the real-time rendering of the 3D scene inside the nasal cavity can be achieved through the posture of the endoscopic camera, providing additional perspective and depth information, which enables doctors to perform cutting, suturing, and other operations more accurately. However, the application of traditional methods in 3D reconstruction of nasal endoscopy has certain limitations. For example, geometry-based reconstruction techniques, such as SfM (Snavely et al., <xref ref-type="bibr" rid="B29">2006</xref>; Widya et al., <xref ref-type="bibr" rid="B33">2019</xref>; Leonard et al., <xref ref-type="bibr" rid="B15">2016</xref>; Schonberger and Frahm, <xref ref-type="bibr" rid="B27">2016</xref>) and SLAM (Grasa et al., <xref ref-type="bibr" rid="B7">2013</xref>; Wang et al., <xref ref-type="bibr" rid="B31">2020</xref>; Mahmoud et al., <xref ref-type="bibr" rid="B17">2017</xref>; Mur-Artal et al., <xref ref-type="bibr" rid="B22">2015</xref>), often struggle to accurately capture feature points in complex nasal scenes with rich vascular networks and lack of distinct textures, resulting in sparse reconstruction. Additionally, endoscopic images may be affected by lighting effects and lens jitter, leading to image blurring and making reconstruction more complex. The emerging technology based on NeRF (Wang et al., <xref ref-type="bibr" rid="B32">2022</xref>; Batlle et al., <xref ref-type="bibr" rid="B2">2023</xref>; Chen P. et al., <xref ref-type="bibr" rid="B3">2024</xref>) is to use implicit neural representation for volume parameterization of 3D space, which not only is the flexibility poor, but also has slow inference speed, greatly reducing the real-time performance of intraoperative surgery.</p>
<p>Therefore, in this paper, a nasal endoscope reconstruction model, Mip-EndoGS is proposed. Specifically, building upon the foundation of the 3D Gaussian Splatting model (3D-GS) (Kerbl et al., <xref ref-type="bibr" rid="B12">2023</xref>), we employ a diffusion model to alleviate the impact of dynamic blurring in endoscopic images on the reconstruction results. In addition, an adaptive low-pass filter is introduced to reduce aliasing artifacts during the rendering process. We collect a dataset of high-definition surgical videos of nasal examinations performed by professional physicians, recording the spatial position of each frame. Subsequently, we apply the proposed Mip-EndoGS model to this dataset, achieving high-quality and real-time rendering of 3D nasal endoscopic scenes. The main contributions of this paper are as follows.</p>
<list list-type="order">
<list-item><p>A nasal endoscopy reconstruction model, Mip-EndoGS, is proposed to achieve high-quality 3D reconstruction of nasal endoscopy, which integrates a diffusion module into the 3D Gaussian Splatting framework to remove blur from endoscopic images.</p></list-item>
<list-item><p>An adaptive low-pass filter is embedded into the Gaussian rendering pipeline to overcome aliasing artifacts, which achieves realistic 3D reconstruction of nasal endoscopy scenes.</p></list-item>
<list-item><p>Extensive quantitative and qualitative experiments are conducted to validate the proposed model&#x00027;s effectiveness in reconstructing and rendering nasal endoscopy scenes.</p></list-item>
</list></sec>
<sec sec-type="materials and methods" id="s2">
<title>2 Materials and methods</title>
<sec>
<title>2.1 Nasal endoscopy dataset</title>
<p>The nasal endoscopy dataset, NasED, is constructed by our own. There are 16 subjects with a total of 51 video segments. The data is collected using XION 4K endoscope and NDI optical surgical navigation system. The videos record the process from the inferior and middle nasal meatus to the pharyngeal orifice of the eustachian tube, capturing multi-angle shots of the internal nasal structures, such as the middle and inferior turbinates. Furthermore, the video data is preprocessed into nasal endoscopy examination images with a resolution of 1280 x 720, totaling over 30,000 frames.</p>
</sec>
<sec>
<title>2.2 Method architecture</title>
<p>The proposed high fidelity 3D reconstruction and rendering model framework is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, which comprises two stages, image enhancement based on the diffusion model and 3D-GS differentiable rendering using an adaptive low-pass filter. In the first stage, we uniformly sample several endoscopic views from the endoscopic video in chronological order and select relatively blurry views as input to the diffusion module (Chen Z. et al., <xref ref-type="bibr" rid="B4">2024</xref>) for deblurring processing. Subsequently, the deblurred views are merged with the original ones to obtain an image-enhanced sequence of endoscopic images. In the second stage, the optimized image sequence is processed through Structure-from-Motion (SfM) (Snavely et al., <xref ref-type="bibr" rid="B29">2006</xref>) algorithms to obtain sparse 3D point clouds and camera poses. These generated point clouds and camera poses are then inputted into the Gaussian splatting pipeline for fast differentiable rasterization rendering (Kerbl et al., <xref ref-type="bibr" rid="B12">2023</xref>). In the splatting rendering process, adaptive low-pass filtering is designed to overcome aliasing issues, thereby achieving high-quality 3D reconstruction of nasal endoscopic scenes.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>The overview of our Mip-EndoGS pipeline. Firstly, relatively blurry views are processed by the diffusion module for deblurring processing. Subsequently, the improved image sequence is processed through SfM to obtain sparse 3D point clouds and camera poses. Then, The generated point clouds and camera poses are fed into the Gaussian splatting pipeline for fast differentiable rendering. Adaptive low-pass filtering is applied during splatting to reduce aliasing and improve the quality of 3D nasal endoscopy reconstruction.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-19-1630728-g0001.tif"/>
</fig>
<sec>
<title>2.2.1 Image enhancement based on diffusion models</title>
<p>In the process of reconstructing 3D nasal cavity based on endoscopic video, the factors may potentially affect the quality of the images, such as the blurriness caused by the mutual compression of nasal tissues or induced by the dynamic movement of the endoscope. Meanwhile, the potential noise can affect feature extraction between consecutive frames. Therefore, the advanced HI-Diff (Chen Z. et al., <xref ref-type="bibr" rid="B4">2024</xref>) method is employed to denoise the captured nasal endoscopic images, which combines the Transformer based reconstruction module with the traditional diffusion model, and utilizes hierarchical concentration modules (Zamir et al., <xref ref-type="bibr" rid="B36">2022</xref>) to enhance the deblurring process.</p>
<p>The overall framework of HI-Diff deblurring is illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>. During the training process, given the input blurry image <italic>I</italic><sub>Blur</sub> and its corresponding ground truth image <italic>I</italic><sub>GT</sub>, there are two identical latent encoders (LE) (Rombach et al., <xref ref-type="bibr" rid="B25">2022</xref>) employed to process both images. Specifically, the concatenated form of the blurry image <italic>I</italic><sub>Blur</sub> and its corresponding ground truth image <italic>I</italic><sub>GT</sub> is first fed into one of the latent encoders to extract the prior features <italic>v</italic>. Simultaneously, the blurry image <italic>I</italic><sub>Blur</sub> is fed into another LE to be mapped to a conditional latent vector <italic>p</italic>. The specific procedure is as follows:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>v</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>E</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub><mml:mtext class="textrm" mathvariant="normal">&#x000A9;</mml:mtext><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi><mml:mi>l</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mtext class="textrm" mathvariant="normal">,</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>p</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>E</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>B</mml:mi><mml:mi>l</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mtext class="textrm" mathvariant="normal">,</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where <italic>f</italic><sub><italic>LE</italic>1</sub> and <italic>f</italic><sub><italic>LE</italic>2</sub> denote the mappings of the images into high-dimensional space, and &#x000A9; represents the concatenation of the two images.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>The framework of HI-Diff. Blurry views are fed into HI-Diff, which performs deblurring to recover clearer and more detailed structures.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-19-1630728-g0002.tif"/>
</fig>
<p>Subsequently, adhering to the procedures outlined in the diffusion model, the prior features are subjected to the addition of random Gaussian noise before being inputted into the denoising network, resulting in <italic>v</italic><sub><italic>T</italic></sub>. Concurrently, the conditional latent vector <italic>p</italic> is also fed into the denoising network. This denoising network, conditioned on both inputs, proceeds to predict the ultimate prior features <italic>v</italic><sub>1</sub>. The detailed process unfolds as follows:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mi>f</mml:mi><mml:mi>u</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mtext class="textrm" mathvariant="normal">,</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mtext class="textrm" mathvariant="normal">,</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where <italic>f</italic><sub><italic>diffusion</italic></sub> denotes the process of adding noise to the prior features <italic>v</italic>, and <italic>f</italic><sub><italic>denoising</italic></sub> represents the neural network. This network takes the vector <italic>v</italic><sub><italic>T</italic></sub>, which has been augmented with random noise, along with <italic>p</italic> as inputs, to predict the prior features <italic>v</italic><sub>1</sub>.</p>
<p>Moreover, due to the non-uniform blurriness induced by the dynamic motion of the endoscope, relying solely on a single scale of prior features may not adequately accommodate complex blurring scenarios. Hence, to acquire multi-scale prior features capable of adapting to various scales of intermediate features, <italic>v</italic><sub>1</sub> is downsampled twice. The specific procedure unfolds as follows:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>o</mml:mi><mml:mi>w</mml:mi><mml:mi>n</mml:mi><mml:mo>-</mml:mo><mml:mi>s</mml:mi><mml:mi>a</mml:mi><mml:mi>m</mml:mi><mml:mi>p</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mtext class="textrm" mathvariant="normal">,</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E6"><label>(6)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>o</mml:mi><mml:mi>w</mml:mi><mml:mi>n</mml:mi><mml:mo>-</mml:mo><mml:mi>s</mml:mi><mml:mi>a</mml:mi><mml:mi>m</mml:mi><mml:mi>p</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mtext class="textrm" mathvariant="normal">.</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>For the Transformer-based reconstruction module, given the input blurry image <italic>I</italic><sub><italic>Blur</italic></sub>, the reconstruction module undergoes multiple rounds of upsampling and downsampling before reconstructing the clear image <italic>I</italic><sub><italic>DB</italic></sub>. Furthermore, at each feature extraction stage, a hierarchical concentration module is positioned ahead of both the encoder and decoder, which serves to fuse the intermediate features <italic>X</italic><sub><italic>in</italic></sub> from the Transformer with the multi-scale prior features <italic>v</italic><sub>1</sub>,<italic>v</italic><sub>2</sub>,<italic>v</italic><sub>3</sub> from the diffusion model through cross-attention fusion. Its purpose is to enhance the deblurring process of the Transformer.</p>
<p>During the testing phase, we replace the ground truth image <italic>I</italic><sub><italic>GT</italic></sub> with randomly generated Gaussian noise. The blurry image <italic>I</italic><sub><italic>Blur</italic></sub> is then fed into the diffusion model to obtain the prior features <italic>v</italic><sub>1</sub>. Subsequently, these prior features are utilized to enhance the blurry image within the Transformer-based reconstruction module, resulting in the generation of high-quality, clear images.</p></sec>
<sec>
<title>2.2.2 Differentiable rendering through 3D Gaussians Splatting</title>
<p>To achieve fast differentiable rasterization rendering through 3D-GS splatting (Kerbl et al., <xref ref-type="bibr" rid="B12">2023</xref>), the sparse point clouds along with their corresponding camera poses are required, which can be estimated by tracking feature points across multiple images based on Structure-from-Motion (SfM)(Snavely et al., <xref ref-type="bibr" rid="B29">2006</xref>). Based on these point clouds, a set of Gaussian functions using the position mean &#x003BC; and covariance matrix <italic>A</italic> is defined. To enhance the representation of the scene, each Gaussian function is equipped with opacity &#x003C3; and a set of spherical harmonic functions. By introducing this anisotropic 3D Gaussian distribution as a high-quality and unstructured representation of the radiation field, not only can the model compactly represent 3D scenes, but flexible optimization processes are also supported. Specifically, the probability density function of the Gaussian model is as follows (Zwicker et al., <xref ref-type="bibr" rid="B38">2001b</xref>):</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>-</mml:mo><mml:mstyle mathvariant="bold"><mml:mi>&#x003BC;</mml:mi></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mi>A</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>-</mml:mo><mml:mstyle mathvariant="bold"><mml:mi>&#x003BC;</mml:mi></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mtext class="textrm" mathvariant="normal">,</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>A</italic> can be decomposed into two more specific components, the quaternion <italic>r</italic> and the 3D-vector <italic>s</italic>. Then, these components are transformed into the corresponding rotation and scaling matrices <italic>R</italic> and <italic>S</italic>. Therefore, the covariance matrix <italic>A</italic> can be represented as:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mo>=</mml:mo><mml:mi>R</mml:mi><mml:mi>S</mml:mi><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mtext class="textrm" mathvariant="normal">.</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>During the rendering stage, Gaussian elements need to be projected into the rendering space (Zwicker et al., <xref ref-type="bibr" rid="B37">2001a</xref>). Through view transformation, the new covariance matrix in the camera coordinate system can be calculated as follows:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mi>J</mml:mi><mml:mi>W</mml:mi><mml:mi>A</mml:mi><mml:msup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mi>J</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mtext class="textrm" mathvariant="normal">,</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>J</italic> is the Jacobian matrix approximating the affine transformation of the projection.</p>
<p>Additionally, these Gaussian elements are projected onto the imaging plane according to the observation matrix, and colors are blended based on opacity and depth (Kopanas et al., <xref ref-type="bibr" rid="B13">2022</xref>, <xref ref-type="bibr" rid="B14">2021</xref>). Therefore, the final color <italic>C</italic>(<italic>p</italic>) of the <italic>p</italic>-th pixel can be represented by blending M ordered points overlapping the pixel:</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>C</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>M</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mtext class="textrm" mathvariant="normal">,</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>with</p>
<disp-formula id="E11"><mml:math id="M11"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mstyle class="mbox"><mml:mtext>and</mml:mtext></mml:mstyle><mml:msub><mml:mrow><mml:mi>T</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x0220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:munderover></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mtext class="textrm" mathvariant="normal">.</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><italic>T</italic><sub><italic>i</italic></sub> is the transmittance, <italic>c</italic><sub><italic>i</italic></sub> represents the color of the Gaussian element along the direction of the ray, and &#x003BC;<sub><italic>i</italic></sub> denotes the projected 2D <italic>UV</italic> coordinates of the 3D Gaussians.</p>
<p>Efficient rendering and depth sorting are achieved through a fast tile-based differentiable raster izer. Additionally, the &#x003B1;-blending technique is introduced to adjust opacity &#x003C3; and scale parameter <italic>S</italic> through a sigmoid function, which ensures that the image synthesis maintains higher visual quality.</p>
<p>The rendered scene is compared with the corresponding image to calculate the loss for rapid backpropagation. The loss function consists of <inline-formula><mml:math id="M12"><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> loss and Structural Similarity Index Measure (SSIM), balanced by adjusting the weighting factor &#x003BB;. It is expressed as follows:</p>
<disp-formula id="E12"><label>(11)</label><mml:math id="M13"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>&#x003BB;</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="script">L</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>D-SSIM</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mtext class="textrm" mathvariant="normal">.</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Here, the Stochastic Gradient Descent (SGD) algorithm is utilized to optimize the model parameters iteratively to minimize the loss function (Fridovich-Keil et al., <xref ref-type="bibr" rid="B6">2022</xref>). To further optimize the model, adaptive density control is implemented to adjust the number and density of Gaussian elements for better scene representation. The introduction of transparency threshold &#x003F5;<sub>&#x003B1;</sub> and position gradient threshold &#x003C4;<sub><italic>pos</italic></sub> is used to control the addition and removal of Gaussian elements. The introduction of this adaptive control allows the method to better adapt to the geometric complexity of nasal endoscopy scenes.</p></sec>
<sec>
<title>2.2.3 Adaptive low-pass filter</title>
<p>In the rendering process, aliasing is a fundamental issue, as rendered images are usually sampled based on discrete raster grids, inevitably leading to visual artifacts such as jagged edges along object contours and Moir &#x000E9; fringes in textures. A similar phenomenon occurs when splashing elliptical Gaussian, when the scene is reconstructed and rendered at a lower sampling rate.</p>
<p>Previous research (Hu et al., <xref ref-type="bibr" rid="B9">2023</xref>; Yu et al., <xref ref-type="bibr" rid="B35">2023</xref>; Barron et al., <xref ref-type="bibr" rid="B1">2021</xref>) attempts to mitigate aliasing effects generally by prefiltering (Heckbert, <xref ref-type="bibr" rid="B8">1989</xref>; Mueller et al., <xref ref-type="bibr" rid="B20">1998</xref>) and super-sampling techniques (Cook, <xref ref-type="bibr" rid="B5">1986</xref>). For example, the EWA volume reconstruction (Zwicker et al., <xref ref-type="bibr" rid="B37">2001a</xref>) introduces the notion of resampling filters, combining the reconstruction algorithm with a low-pass kernel. Inspired by this method, we employ an anti-aliasing filter to alleviate aliasing artifacts during nasal endoscope rendering. Building upon <xref ref-type="disp-formula" rid="E10">Equation 10</xref>, we further elaborate the rasterization formula:</p>
<disp-formula id="E13"><label>(12)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>C</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x0220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:munderover></mml:mstyle><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mtext class="textrm" mathvariant="normal">,</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Where <inline-formula><mml:math id="M15"><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:math></inline-formula> represents the projection of the Gaussian distribution onto a two-dimensional plane, closely related to the two-dimensional covariance (Kopanas et al., <xref ref-type="bibr" rid="B14">2021</xref>). And the 2 &#x000D7; 2 variance matrix <italic>A</italic><sup>&#x02032;&#x02032;</sup> can be easily obtained from the 3 &#x000D7; 3 matrix <italic>A</italic>&#x02032; by skipping the third row and column:</p>
<disp-formula id="E14"><label>(13)</label><mml:math id="M16"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mi>a</mml:mi></mml:mtd><mml:mtd><mml:mi>b</mml:mi></mml:mtd><mml:mtd><mml:mi>c</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>b</mml:mi></mml:mtd><mml:mtd><mml:mi>d</mml:mi></mml:mtd><mml:mtd><mml:mi>e</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>c</mml:mi></mml:mtd><mml:mtd><mml:mi>e</mml:mi></mml:mtd><mml:mtd><mml:mi>f</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>&#x021D4;</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mi>a</mml:mi></mml:mtd><mml:mtd><mml:mi>b</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>b</mml:mi></mml:mtd><mml:mtd><mml:mi>d</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mtext class="textrm" mathvariant="normal">.</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Following that, to simulate the diffusion effect occurring during the propagation of light rays, the scale of the 2D covariance is adjusted (Kerbl et al., <xref ref-type="bibr" rid="B12">2023</xref>), for which a positive definite adjustment term is added to the original covariance matrix <italic>A</italic><sup>&#x02032;&#x02032;</sup>.</p>
<p>The adjustment term is a scalar multiplied by the unit matrix related to the hyperparameter, by which the scale of the covariance matrix is increased and the diffusion effect of light rays is simulated. Furthermore, in the actual imaging process, the light captured by each pixel accumulates within its surface area, meaning the final image is obtained by integrating the photon energy falling on each pixel (Shirley, <xref ref-type="bibr" rid="B28">2018</xref>). To achieve the actual imaging process more efficiently, the &#x0201C;Adaptive Low-Pass Filter&#x0201D; is proposed as shown in <xref ref-type="disp-formula" rid="E15">Equation 14</xref>, which adapts to different sampling rates and changes in perspective when processing endoscopic images, while maintaining the visual quality of the image.</p>
<disp-formula id="E15"><label>(14)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mrow><mml:mi mathvariant="script">N</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mi>D</mml:mi></mml:mrow></mml:msup><mml:msub><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mtext>x</mml:mtext></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle class="mbox"><mml:mtext>low-pass</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mfrac><mml:mrow><mml:mo>|</mml:mo><mml:msup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:msup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:mi>s</mml:mi><mml:mstyle mathvariant="bold"><mml:mtext>I</mml:mtext></mml:mstyle><mml:mo>|</mml:mo></mml:mrow></mml:mfrac></mml:mrow></mml:msqrt><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>-</mml:mo><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:mi>s</mml:mi><mml:mstyle mathvariant="bold"><mml:mtext>I</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>-</mml:mo><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mtext class="textrm" mathvariant="normal">.</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The scale parameter in the adaptive low-pass filter controls the extent of Gaussian smoothing. Intuitively, it simulates the physical diffusion of light across pixel areas due to limited resolution and sampling rates. A larger scale parameter induces stronger anti-aliasing but risks oversmoothing details, while a smaller scale parameter preserves sharpness but may cause jagged edges.</p>
</sec>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<p>In this section, the proposed method, Mip-EndoGS, is evaluated based on our nasal endoscopy dataset, NasED. Firstly, the implementation setting of Mip-EndoGS is presented. Then, we provide a detailed introduction of the metrics used in the experiment. Finally, the experimental results are showed, including both quantitative analysis and qualitative analysis.</p>
<sec>
<title>3.1 Experiments setting</title>
<p>The NasED dataset comprises several monocular nasal endoscopy video sequences, denoted as <inline-formula><mml:math id="M18"><mml:msubsup><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>. Here, <italic>T</italic> represents the number of sequences, and <italic>H</italic><sub><italic>i</italic></sub> denotes the <italic>i</italic>-th sequence. Each nasal sequence is divided into several frames, denoted as <inline-formula><mml:math id="M19"><mml:msubsup><mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>, where <italic>M</italic> is the total number of frames in the sequence, and <italic>j</italic> represents the index of the <italic>j</italic>-th frame. Hence, the <italic>i</italic>-th sequence and <italic>j</italic>-th frame&#x00027;s endoscopic view is represented as (<italic>H</italic><sub><italic>i</italic></sub>, <italic>A</italic><sub><italic>j</italic></sub>). From this dataset, we extracted four groups of video sequences: <italic>H</italic><sub>1</sub>, <italic>H</italic><sub>2</sub>, <italic>H</italic><sub>3</sub>, and <italic>H</italic><sub>4</sub>. Each group consists of randomly sampled consecutive 100-frame views, totaling 400 frames. Each sequence is split into 90% training data and 10% testing data. These video sequences are captured by a monocular camera, covering the internal structures of the nasal cavity and sinuses.</p>
<p>In the diffusion module, we adhere to experimental settings consistent with HI-Diff and load weights trained on the GoPro (Nah et al., <xref ref-type="bibr" rid="B23">2017</xref>) synthetic dataset for image denoising. Sparse point clouds and camera poses are obtained through COLMAP (Snavely et al., <xref ref-type="bibr" rid="B29">2006</xref>; Schonberger and Frahm, <xref ref-type="bibr" rid="B27">2016</xref>). The parameters of the Gaussian rendering pipeline (Kerbl et al., <xref ref-type="bibr" rid="B12">2023</xref>) follow the original method settings, except for the changes in the number of iterations. The scale parameter in the adaptive low-pass filter is set to 0.3, and the learning rate is set to 1e-4. The network is trained on an NVIDIA RTX A6000 device.</p>
</sec>
<sec>
<title>3.2 Metric</title>
<p>To conduct a thorough assessment of our experimental results, various methods are employed to evaluate the reconstruction outcomes, primarily comprising quantitative analysis and qualitative assessment through visualization. For quantitative analysis, we utilized several commonly used evaluation metrics, including the Structural Similarity Index Measure (SSIM), Peak Signal-to-Noise Ratio (PSNR), and Learned Perceptual Image Patch Similarity (LPIPS).</p>
<p>The computation of SSIM is as follows, which measures the similarity between two images in terms of brightness, contrast, and structure:</p>
<disp-formula id="E16"><label>(15)</label><mml:math id="M20"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo class="qopname">SSIM</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:msub><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:msub><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mtext class="textrm" mathvariant="normal">,</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>x</italic> and <italic>y</italic> represent the two images to be compared, &#x003BC;<sub><italic>x</italic></sub> and &#x003BC;<sub><italic>y</italic></sub> denote their mean intensities, <inline-formula><mml:math id="M21"><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="M22"><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula> represent their variances, &#x003C3;<sub><italic>xy</italic></sub> indicates their covariance, and <italic>c</italic><sub>1</sub> and <italic>c</italic><sub>2</sub> are variables used to stabilize the denominator.</p>
<p>The definition of PSNR is as follows:</p>
<disp-formula id="E17"><label>(16)</label><mml:math id="M23"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo class="qopname">PSNR</mml:mo><mml:mo>=</mml:mo><mml:mn>20</mml:mn><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mo class="qopname">log</mml:mo></mml:mrow><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mtext>MA</mml:mtext><mml:msub><mml:mrow><mml:mtext>X</mml:mtext></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:mtext>MSE</mml:mtext></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mtext class="textrm" mathvariant="normal">,</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where MAX<sub><italic>I</italic></sub> represents the maximum possible pixel value of the image, and MSE is the mean squared error between the reconstructed image and the reference image.</p>
<p>LPIPS employs deep learning models to evaluate the perceptual similarity between images, capturing texture and structural differences crucial for human visual perception:</p>
<disp-formula id="E18"><label>(17)</label><mml:math id="M24"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">LPIPS</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mtext class="textrm" mathvariant="normal">,</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003D5;<sub><italic>l</italic></sub>(<italic>x</italic>) and &#x003D5;<sub><italic>l</italic></sub>(<italic>y</italic>) represent the feature maps of images <italic>x</italic> and <italic>y</italic> extracted by a pre-trained deep neural network at layer <italic>l</italic>, and <italic>w</italic><sub><italic>l</italic></sub> is a learned weight used to emphasize the importance of each layer&#x00027;s contribution to perceptual similarity.</p>
<p>By applying these metrics, we can quantitatively analyze the quality of our image reconstructions.</p>
</sec>
<sec>
<title>3.3 Results analysis</title>
<sec>
<title>3.3.1 Evaluation on full resolution</title>
<p>To validate the model&#x00027;s strong generalization capability, we selected sequences from different subjects. The results are presented in <xref ref-type="table" rid="T1">Table 1</xref> and <xref ref-type="fig" rid="F3">Figure 3</xref>. <xref ref-type="fig" rid="F3">Figure 3</xref> illustrates the rendering effects of sequences <italic>H</italic><sub>1</sub>, <italic>H</italic><sub>2</sub> and <italic>H</italic><sub>3</sub> after 40k iterations of model training.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Quantitative comparison of rendering quality on different video sequences.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#8f9496;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center" colspan="3"><bold>H1</bold></th>
<th valign="top" align="center" colspan="3"><bold>H2</bold></th>
<th valign="top" align="center" colspan="3"><bold>H3</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#8f9496;color:#ffffff">
<td/>
<td valign="top" align="center"><bold>PSNR</bold></td>
<td valign="top" align="center"><bold>SSIM</bold></td>
<td valign="top" align="center"><bold>LPIPS</bold></td>
<td valign="top" align="center"><bold>PSNR</bold></td>
<td valign="top" align="center"><bold>SSIM</bold></td>
<td valign="top" align="center"><bold>LPIPS</bold></td>
<td valign="top" align="center"><bold>PSNR</bold></td>
<td valign="top" align="center"><bold>SSIM</bold></td>
<td valign="top" align="center"><bold>LPIPS</bold></td>
</tr> <tr>
<td valign="top" align="left">3D-GS</td>
<td valign="top" align="center">26.52</td>
<td valign="top" align="center"><bold>0.942</bold></td>
<td valign="top" align="center"><bold>0.114</bold></td>
<td valign="top" align="center">32.67</td>
<td valign="top" align="center">0.953</td>
<td valign="top" align="center">0.137</td>
<td valign="top" align="center">26.34</td>
<td valign="top" align="center">0.925</td>
<td valign="top" align="center">0.156</td>
</tr> <tr>
<td valign="top" align="left">Mip-EndoGS</td>
<td valign="top" align="center"><bold>27.50</bold></td>
<td valign="top" align="center">0.936</td>
<td valign="top" align="center"><bold>0.114</bold></td>
<td valign="top" align="center"><bold>35.95</bold></td>
<td valign="top" align="center"><bold>0.971</bold></td>
<td valign="top" align="center"><bold>0.022</bold></td>
<td valign="top" align="center"><bold>30.16</bold></td>
<td valign="top" align="center"><bold>0.934</bold></td>
<td valign="top" align="center"><bold>0.145</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The best results are in bold.</p>
</table-wrap-foot>
</table-wrap>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Qualitative results presentation on different video sequences.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-19-1630728-g0003.tif"/>
</fig>
<p>A comparison with the Ground Truth reveals that despite the narrow field of view and lack of texture in nasal endoscopic views, our method renders nasal structures distinctly with clear textures. Compared to the original 3D-GS method, the proposed approach demonstrates higher stability, effectively reducing issues such as significant aliasing, artifacts, and distortions in certain areas observed in the output of 3D-GS. Quantitative evaluation through <xref ref-type="table" rid="T1">Table 1</xref> shows notable improvements in the PSNR metrics across all four datasets. Additionally, except for <italic>H</italic><sub>1</sub>, the SSIM and LPIPS metrics for the other three datasets also achieve superior results.</p></sec>
<sec>
<title>3.3.2 Compared with COLMAP</title>
<p>The proposed method, Mip-EndoGS, is compared with the current mainstream reconstruction methods, Depth Map Fusion (Merrell et al., <xref ref-type="bibr" rid="B18">2007</xref>) and the Poisson method (Kazhdan and Hoppe, <xref ref-type="bibr" rid="B11">2013</xref>) in COLMAP. and the visual results are shown in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Comparison of surface reconstruction from ours and COLMAP.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-19-1630728-g0004.tif"/>
</fig>
<p>Data <italic>H</italic><sub>1</sub> is utilized in this evaluation. Evidently, the nasal endoscopic scenes reconstruct with Mip-EndoGS exhibit more realistic and smoother features, demonstrating excellent visual outcomes. Apart from comparison with COLMAP, we also attempt reconstruction using methods based on neural radiance fields such as NeRF (Mildenhall et al., <xref ref-type="bibr" rid="B19">2021</xref>) and Neuraludf (Long et al., <xref ref-type="bibr" rid="B16">2023</xref>). However, due to the unique characteristics of nasal structures, these methods all fail.</p></sec>
<sec>
<title>3.3.3 Evaluation on different iterations</title>
<p><xref ref-type="table" rid="T2">Table 2</xref> and <xref ref-type="fig" rid="F5">Figure 5</xref> respectively present the quantitative results and visual effects of 3D-GS and Mip-EndoGS at 6k and 40k iterations (evaluated using <italic>H</italic><sub>3</sub> data), which shows that our model is capable of capturing the structures within the nasal cavity clearly after 6k iterations, with the PSNR metric significantly outperforming the rendering results of 3D-GS at the same iteration count.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Quantitative comparison of different outcomes after 6k and 40k iterations.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#8f9496;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center" colspan="3"><bold>6k</bold></th>
<th valign="top" align="center" colspan="3"><bold>40k</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#8f9496;color:#ffffff">
<td/>
<td valign="top" align="center"><bold>PSNR</bold></td>
<td valign="top" align="center"><bold>Train time</bold></td>
<td valign="top" align="center"><bold>FPS</bold></td>
<td valign="top" align="center"><bold>PSNR</bold></td>
<td valign="top" align="center"><bold>Train time</bold></td>
<td valign="top" align="center"><bold>FPS</bold></td>
</tr> <tr>
<td valign="top" align="left">3D-GS</td>
<td valign="top" align="center">20.63</td>
<td valign="top" align="center">1 m 53 s</td>
<td valign="top" align="center">112</td>
<td valign="top" align="center">26.34</td>
<td valign="top" align="center">12 m 23 s</td>
<td valign="top" align="center">91</td>
</tr> <tr>
<td valign="top" align="left">Mip-EndoGS</td>
<td valign="top" align="center"><bold>27.49</bold></td>
<td valign="top" align="center">3 m 14 s</td>
<td valign="top" align="center">105</td>
<td valign="top" align="center"><bold>30.16</bold></td>
<td valign="top" align="center">14m</td>
<td valign="top" align="center">86</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The best results are in bold.</p>
</table-wrap-foot>
</table-wrap>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Qualitative comparison of different outcomes after 6k and 40k iterations.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-19-1630728-g0005.tif"/>
</fig>
<p>In terms of time, the proposed model only takes around 3 minutes for 6k iterations, less than 1/4 of the time required for 40k iterations. This shorter training time, coupled with clear structural representation, is crucial for real-time surgical navigation. Moreover, although the addition of the diffusion module slightly affects the training time and rendering speed of our model, it still achieves real-time rendering capability.</p></sec>
<sec>
<title>3.3.4 Evaluation on various resolution</title>
<p>To simulate the reconstruction effects of scenes at low sampling rates, the original data are downsampled to obtain datasets with resolutions reduced to 1/2, 1/4, and 1/8 of the original resolution. We train the model on the original resolution data and render on the downsampled datasets accordingly. The quantitative evaluation is conducted using <italic>H</italic><sub>1</sub> and <italic>H</italic><sub>4</sub> data (as shown in <xref ref-type="table" rid="T3">Table 3</xref>), where the proposed method outperforms 3D-GS in rendering quality at lower resolutions. The visual results for <italic>H</italic><sub>4</sub> are shown in <xref ref-type="fig" rid="F6">Figure 6</xref>, where the proposed method produces the higher fidelity imagery without apparent artifacts and aliasing.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Quantitative comparison of single-scale training and multi-scale testing.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#8f9496;color:#ffffff">
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Method</bold></th>
<th valign="top" align="center" colspan="5"><bold>PSNR</bold></th>
<th valign="top" align="center" colspan="5"><bold>SSIM</bold></th>
<th valign="top" align="center" colspan="5"><bold>LPIPS</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#8f9496;color:#ffffff">
<td/>
<td/>
<td valign="top" align="center"><bold>Full Res</bold>.</td>
<td valign="top" align="center"><bold>1/2 Res</bold>.</td>
<td valign="top" align="center"><bold>1/4 Res</bold>.</td>
<td valign="top" align="center"><bold>1/8 Res</bold>.</td>
<td valign="top" align="center"><bold>Avg</bold>.</td>
<td valign="top" align="center"><bold>Full Res</bold>.</td>
<td valign="top" align="center"><bold>1/2 Res</bold>.</td>
<td valign="top" align="center"><bold>1/4 Res</bold>.</td>
<td valign="top" align="center"><bold>1/8 Res</bold>.</td>
<td valign="top" align="center"><bold>Avg</bold>.</td>
<td valign="top" align="center"><bold>Full Res</bold>.</td>
<td valign="top" align="center"><bold>1/2 Res</bold>.</td>
<td valign="top" align="center"><bold>1/4 Res</bold>.</td>
<td valign="top" align="center"><bold>1/8 Res</bold>.</td>
<td valign="top" align="center"><bold>Avg</bold>.</td>
</tr> <tr>
<td valign="top" align="left">H1</td>
<td valign="top" align="center">3D-GS</td>
<td valign="top" align="center">26.52</td>
<td valign="top" align="center">26.50</td>
<td valign="top" align="center">26.82</td>
<td valign="top" align="center">29.44</td>
<td valign="top" align="center">27.32</td>
<td valign="top" align="center"><bold>0.942</bold></td>
<td valign="top" align="center"><bold>0.936</bold></td>
<td valign="top" align="center">0.910</td>
<td valign="top" align="center"><bold>0.962</bold></td>
<td valign="top" align="center">0.938</td>
<td valign="top" align="center"><bold>0.114</bold></td>
<td valign="top" align="center">0.085</td>
<td valign="top" align="center">0.089</td>
<td valign="top" align="center">0.052</td>
<td valign="top" align="center">0.085</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Mip-EndoGS</td>
<td valign="top" align="center"><bold>27.50</bold></td>
<td valign="top" align="center"><bold>27.79</bold></td>
<td valign="top" align="center"><bold>28.66</bold></td>
<td valign="top" align="center"><bold>31.10</bold></td>
<td valign="top" align="center"><bold>28.74</bold></td>
<td valign="top" align="center">0.936</td>
<td valign="top" align="center">0.927</td>
<td valign="top" align="center"><bold>0.933</bold></td>
<td valign="top" align="center">0.954</td>
<td valign="top" align="center"><bold>0.938</bold></td>
<td valign="top" align="center"><bold>0.114</bold></td>
<td valign="top" align="center"><bold>0.084</bold></td>
<td valign="top" align="center"><bold>0.066</bold></td>
<td valign="top" align="center"><bold>0.046</bold></td>
<td valign="top" align="center"><bold>0.076</bold></td>
</tr> <tr>
<td valign="top" align="left">H4</td>
<td valign="top" align="center">3D-GS</td>
<td valign="top" align="center">20.81</td>
<td valign="top" align="center">20.69</td>
<td valign="top" align="center">19.87</td>
<td valign="top" align="center">20.14</td>
<td valign="top" align="center">20.38</td>
<td valign="top" align="center"><bold>0.883</bold></td>
<td valign="top" align="center"><bold>0.854</bold></td>
<td valign="top" align="center">0.835</td>
<td valign="top" align="center">0.812</td>
<td valign="top" align="center">0.846</td>
<td valign="top" align="center">0.197</td>
<td valign="top" align="center">0.203</td>
<td valign="top" align="center">0.202</td>
<td valign="top" align="center">0.230</td>
<td valign="top" align="center">0.208</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Mip-EndoGS</td>
<td valign="top" align="center"><bold>21.33</bold></td>
<td valign="top" align="center"><bold>21.24</bold></td>
<td valign="top" align="center"><bold>21.14</bold></td>
<td valign="top" align="center"><bold>23.73</bold></td>
<td valign="top" align="center"><bold>21.86</bold></td>
<td valign="top" align="center">0.872</td>
<td valign="top" align="center">0.832</td>
<td valign="top" align="center"><bold>0.837</bold></td>
<td valign="top" align="center"><bold>0.876</bold></td>
<td valign="top" align="center"><bold>0.854</bold></td>
<td valign="top" align="center"><bold>0.191</bold></td>
<td valign="top" align="center"><bold>0.191</bold></td>
<td valign="top" align="center"><bold>0.180</bold></td>
<td valign="top" align="center"><bold>0.125</bold></td>
<td valign="top" align="center"><bold>0.172</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The best results are in bold.</p>
</table-wrap-foot>
</table-wrap>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Qualitative comparison of single-scale training and multi-scale testing.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-19-1630728-g0006.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>Nasal endoscopic scene reconstruction contributes to a comprehensive understanding of the surgical environment, precise surgical localization, and critical information provision for minimally invasive procedures. However, nasal cavity structures are not only narrow and intricate, but also lack distinctive texture features. Additionally, the influence of endoscopic lighting often makes it challenging to capture nasal cavity structural characteristics. Moreover, the quality of views collected by endoscopy is difficult to guarantee, often resulting in blurriness and contamination.</p>
<p>To address these issues, this paper introduces an advanced nasal endoscopic reconstruction model, Mip-EndoGS, which enables real-time rendering of scenes and synthesis of new viewpoints during surgery by pre-training before surgery. The proposed method consists of two parts, an image enhancement module based on diffusion models and a 3D-GS differentiable rendering pipeline using adaptive low-pass filters. The image enhancement module used in this paper integrates a Transformer-based reconstruction module with traditional diffusion models and employs a hierarchical attention mechanism to enhance the deblurring process of the Transformer, achieving denoising effects on collected nasal endoscopic images. For the differentiable rendering pipeline based on 3D-GS, we embed an adaptive low-pass filter to overcome aliasing artifacts, which simulates the diffusion effect during light propagation and integrates the photon energy falling on each pixel to adapt to changes in sampling rates and viewpoints.</p>
<p>The proposed method can reconstruct highly realistic nasal endoscopic scenes on the NasED dataset. As shown in the experimental results, the reconstructed nasal structures are distinct with clear textures. Compared to the original 3D-GS, the proposed method demonstrates higher stability, effectively alleviating issues such as aliasing artifacts and distortions during rendering. The high-quality reconstruction results can provide more accurate 3D information, assisting surgeons in diagnosis and reducing surgical risks.</p>
<p>In practice, this task will be combined with motion tracking technology to create a more convenient and intelligent surgical navigation workspace. Additionally, with the development of augmented reality and virtual display technologies, doctors can perform detailed surgical simulations preoperatively and provide real-time three-dimensional views intraoperatively. Such capabilities are particularly valuable in complex or minimally invasive procedures, where accurate spatial perception is critical. These technological advancements can provide doctors with more intuitive and easier-to-use surgical assistance and offer patients higher-quality medical services.</p>
<p>However, certain limitations still exist, such as the occlusions caused by medical instruments and the hands of the surgeon during surgery, as well as deformations of nasal tissues from various angles. These failure cases highlight the need for further optimization in complex surgical environments. To address these challenges, more intelligent surgical planning and navigation technologies are urgently needed.</p></sec>
<sec sec-type="conclusions" id="s5">
<title>5 Conclusion</title>
<p>In this work, a novel method, Mip-EndoGS, is proposed to reconstruct the scene of nasal endoscopy. The method combines the diffusion model and 3D Gaussian model, initially employing the diffusion model for deblurring and then achieving high-quality real-time rendering using 3D Gaussian. Additionally, we collect high-definition surgical video datasets from nasal examinations performed by professional doctors and validate the proposed method on this dataset. In the experiment, the proposed method demonstrates superior performance in both quantitative assessment and visual analysis. In the future, we plan not only to expand this dataset but also to further refine the related algorithms.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors upon request.</p>
</sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>The studies involving humans were approved by approved by Eye and ENT Hospital of Fudan University (protocol code 2023188-1 of approval). The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>XJ: Data curation, Formal analysis, Methodology, Conceptualization, Writing &#x02013; original draft. SZ: Software, Writing &#x02013; original draft, Methodology, Visualization. DL: Investigation, Writing &#x02013; review &#x00026; editing, Data curation, Resources, Formal analysis. FW: Writing &#x02013; review &#x00026; editing, Data curation, Methodology, Supervision, Project administration. XC: Project administration, Methodology, Supervision, Investigation, Writing &#x02013; review &#x00026; editing, Conceptualization.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported by the Ningxia Hui Autonomous Region Key Research and Development Program (Grant No. 2024BEG02018) and the 2022 Medical-Engineering Special Funded Project of the General Hospital of Ningxia Medical University (Grant No. NYZYYG-007).</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p></sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Barron</surname> <given-names>J. T.</given-names></name> <name><surname>Mildenhall</surname> <given-names>B.</given-names></name> <name><surname>Tancik</surname> <given-names>M.</given-names></name> <name><surname>Hedman</surname> <given-names>P.</given-names></name> <name><surname>Martin-Brualla</surname> <given-names>R.</given-names></name> <name><surname>Srinivasan</surname> <given-names>P. P.</given-names></name></person-group> (<year>2021</year>). &#x0201C;Mip-NeRF: A multiscale representation for anti-aliasing neural radiance fields,&#x0201D;9D in <italic>Proceedings of the IEEE/CVF International Conference on Computer Vision</italic> (<publisher-loc>Montreal, QC</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>5855</fpage>&#x02013;<lpage>5864</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Batlle</surname> <given-names>V. M.</given-names></name> <name><surname>Montiel</surname> <given-names>J. M.</given-names></name> <name><surname>Fua</surname> <given-names>P.</given-names></name> <name><surname>Tard&#x000F3;s</surname> <given-names>J. D.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Lightneus: Neural surface reconstruction in endoscopy using illumination decline,&#x0201D;</article-title> in <source>International Conference on Medical Image Computing and Computer-Assisted Intervention</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>502</fpage>&#x02013;<lpage>512</lpage>.</citation>
</ref>
<ref id="B3">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>P.</given-names></name> <name><surname>Gunderson</surname> <given-names>N. M.</given-names></name> <name><surname>Lewis</surname> <given-names>A.</given-names></name> <name><surname>Speich</surname> <given-names>J. R.</given-names></name> <name><surname>Porter</surname> <given-names>M. P.</given-names></name> <name><surname>Seibel</surname> <given-names>E. J.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;Enabling rapid and high-quality 3D scene reconstruction in cystoscopy through neural radiance fields,&#x0201D;</article-title> in <source>Medical Imaging 2024: Image-Guided Procedures, Robotic Interventions, and Modeling</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>SPIE</publisher-name>), <fpage>350</fpage>&#x02013;<lpage>359</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>D.</given-names></name> <name><surname>Gu</surname> <given-names>J.</given-names></name> <name><surname>Kong</surname> <given-names>L.</given-names></name> <name><surname>Yuan</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>&#x0201C;Hierarchical integration diffusion model for realistic image deblurring,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 36</source> (<publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>MIT Press</publisher-name>).</citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cook</surname> <given-names>R. L.</given-names></name></person-group> (<year>1986</year>). <article-title>Stochastic sampling in computer graphics</article-title>. <source>ACM Trans. Graph</source>. <volume>5</volume>, <fpage>51</fpage>&#x02013;<lpage>72</lpage>. <pub-id pub-id-type="doi">10.1145/7529.8927</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Fridovich-Keil</surname> <given-names>S.</given-names></name> <name><surname>Yu</surname> <given-names>A.</given-names></name> <name><surname>Tancik</surname> <given-names>M.</given-names></name> <name><surname>Chen</surname> <given-names>Q.</given-names></name> <name><surname>Recht</surname> <given-names>B.</given-names></name> <name><surname>Kanazawa</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Plenoxels: Radiance fields without neural networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>New Orleans, LA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>5501</fpage>&#x02013;<lpage>5510</lpage>.<pub-id pub-id-type="pmid">40212877</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grasa</surname> <given-names>O. G.</given-names></name> <name><surname>Bernal</surname> <given-names>E.</given-names></name> <name><surname>Casado</surname> <given-names>S.</given-names></name> <name><surname>Gil</surname> <given-names>I.</given-names></name> <name><surname>Montiel</surname> <given-names>J.</given-names></name></person-group> (<year>2013</year>). <article-title>Visual slam for handheld monocular endoscope</article-title>. <source>IEEE Trans. Med. Imag</source>. <volume>33</volume>, <fpage>135</fpage>&#x02013;<lpage>146</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2013.2282997</pub-id><pub-id pub-id-type="pmid">24107925</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Heckbert</surname> <given-names>P. S.</given-names></name></person-group> (<year>1989</year>). <source>Fundamentals of Texture Mapping and Image Warping</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://www2.eecs.berkeley.edu/Pubs/TechRpts/1989/5504.html">http://www2.eecs.berkeley.edu/Pubs/TechRpts/1989/5504.html</ext-link><pub-id pub-id-type="pmid">39110560</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>W.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Ma</surname> <given-names>L.</given-names></name> <name><surname>Yang</surname> <given-names>B.</given-names></name> <name><surname>Gao</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Ma</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Tri-MipRF: Tri-Mip representation for efficient anti-aliasing neural radiance fields,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF International Conference on Computer Vision</source> (<publisher-loc>Paris</publisher-loc>: <publisher-name>IEEE</publisher-name>) <fpage>19774</fpage>&#x02013;<lpage>19783</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kato</surname> <given-names>H.</given-names></name> <name><surname>Ushiku</surname> <given-names>Y.</given-names></name> <name><surname>Harada</surname> <given-names>T.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Neural 3D mesh renderer,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Salt Lake City, UT</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>3907</fpage>&#x02013;<lpage>3916</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kazhdan</surname> <given-names>M.</given-names></name> <name><surname>Hoppe</surname> <given-names>H.</given-names></name></person-group> (<year>2013</year>). <article-title>Screened poisson surface reconstruction</article-title>. <source>ACM Trans. Graph</source>. <volume>32</volume>, <fpage>1</fpage>&#x02013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1145/2487228.2487237</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kerbl</surname> <given-names>B.</given-names></name> <name><surname>Kopanas</surname> <given-names>G.</given-names></name> <name><surname>Leimk&#x000FC;hler</surname> <given-names>T.</given-names></name> <name><surname>Drettakis</surname> <given-names>G.</given-names></name></person-group> (<year>2023</year>). <article-title>3d gaussian splatting for real-time radiance field rendering</article-title>. <source>ACM Trans. Graph</source>. <volume>42</volume>, <fpage>1</fpage>&#x02013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1145/3592433</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kopanas</surname> <given-names>G.</given-names></name> <name><surname>Leimk&#x000FC;hler</surname> <given-names>T.</given-names></name> <name><surname>Rainer</surname> <given-names>G.</given-names></name> <name><surname>Jambon</surname> <given-names>C.</given-names></name> <name><surname>Drettakis</surname> <given-names>G.</given-names></name></person-group> (<year>2022</year>). <article-title>Neural point catacaustics for novel-view synthesis of reflections</article-title>. <source>ACM Trans. Graph</source>. <volume>41</volume>, <fpage>1</fpage>&#x02013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1145/3550454.3555497</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kopanas</surname> <given-names>G.</given-names></name> <name><surname>Philip</surname> <given-names>J.</given-names></name> <name><surname>Leimk&#x000FC;hler</surname> <given-names>T.</given-names></name> <name><surname>Drettakis</surname> <given-names>G.</given-names></name></person-group> (<year>2021</year>). <article-title>Point-based neural rendering with per-view optimization</article-title>. <source>Comp. Graphics Forum</source> <volume>40</volume>, <fpage>29</fpage>&#x02013;<lpage>43</lpage>. <pub-id pub-id-type="doi">10.1111/cgf.14339</pub-id></citation>
</ref>
<ref id="B15">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Leonard</surname> <given-names>S.</given-names></name> <name><surname>Reiter</surname> <given-names>A.</given-names></name> <name><surname>Sinha</surname> <given-names>A.</given-names></name> <name><surname>Ishii</surname> <given-names>M.</given-names></name> <name><surname>Taylor</surname> <given-names>R. H.</given-names></name> <name><surname>Hager</surname> <given-names>G. D.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Image-based navigation for functional endoscopic sinus surgery using structure from motion,&#x0201D;</article-title> in <source>Medical Imaging 2016: Image Processing</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>SPIE</publisher-name>), <fpage>235</fpage>&#x02013;<lpage>241</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Long</surname> <given-names>X.</given-names></name> <name><surname>Lin</surname> <given-names>C.</given-names></name> <name><surname>Liu</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>P.</given-names></name> <name><surname>Theobalt</surname> <given-names>C.</given-names></name> <name><surname>Komura</surname> <given-names>T.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Neuraludf: Learning unsigned distance fields for multi-view reconstruction of surfaces with arbitrary topologies,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Vancouver, BC</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>20834</fpage>&#x02013;<lpage>20843</lpage>.<pub-id pub-id-type="pmid">38015705</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Mahmoud</surname> <given-names>N.</given-names></name> <name><surname>Cirauqui</surname> <given-names>I.</given-names></name> <name><surname>Hostettler</surname> <given-names>A.</given-names></name> <name><surname>Doignon</surname> <given-names>C.</given-names></name> <name><surname>Soler</surname> <given-names>L.</given-names></name> <name><surname>Marescaux</surname> <given-names>J.</given-names></name> <name><surname>Montiel</surname> <given-names>J. M. M.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Orbslam-based endoscope tracking and 3D reconstruction,&#x0201D;</article-title> in <source>Computer-Assisted and Robotic Endoscopy: Third International Workshop, CARE 2016, Held in Conjunction with MICCAI 2016</source> (<publisher-loc>Athens</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>72</fpage>&#x02013;<lpage>83</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Merrell</surname> <given-names>P.</given-names></name> <name><surname>Akbarzadeh</surname> <given-names>A.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Mordohai</surname> <given-names>P.</given-names></name> <name><surname>Frahm</surname> <given-names>J.-M.</given-names></name> <name><surname>Yang</surname> <given-names>R.</given-names></name> <name><surname>Nist&#x000E9;r</surname> <given-names>D.</given-names></name> <name><surname>Pollefeys</surname> <given-names>M.</given-names></name></person-group> (<year>2007</year>). <article-title>&#x0201C;Real-time visibility-based fusion of depth maps,&#x0201D;</article-title> in <source>2007 IEEE 11th International Conference on Computer Vision</source> (<publisher-loc>Rio de Janeiro</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>8</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mildenhall</surname> <given-names>B.</given-names></name> <name><surname>Srinivasan</surname> <given-names>P. P.</given-names></name> <name><surname>Tancik</surname> <given-names>M.</given-names></name> <name><surname>Barron</surname> <given-names>J. T.</given-names></name> <name><surname>Ramamoorthi</surname> <given-names>R.</given-names></name> <name><surname>Ng</surname> <given-names>R.</given-names></name></person-group> (<year>2021</year>). <article-title>Nerf: Representing scenes as neural radiance fields for view synthesis</article-title>. <source>Commun. ACM</source> <volume>65</volume>, <fpage>99</fpage>&#x02013;<lpage>106</lpage>. <pub-id pub-id-type="doi">10.1145/3503250</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mueller</surname> <given-names>K.</given-names></name> <name><surname>Moller</surname> <given-names>T.</given-names></name> <name><surname>Swan</surname> <given-names>J. E.</given-names></name> <name><surname>Crawfis</surname> <given-names>R.</given-names></name> <name><surname>Shareef</surname> <given-names>N.</given-names></name> <name><surname>Yagel</surname> <given-names>R.</given-names></name></person-group> (<year>1998</year>). <article-title>Splatting errors and antialiasing</article-title>. <source>IEEE Trans. Visualizat. Comp. Graph</source>. <volume>4</volume>, <fpage>178</fpage>&#x02013;<lpage>191</lpage>. <pub-id pub-id-type="doi">10.1109/2945.694987</pub-id><pub-id pub-id-type="pmid">34615319</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>M&#x000FC;nzer</surname> <given-names>B.</given-names></name> <name><surname>Schoeffmann</surname> <given-names>K.</given-names></name> <name><surname>B&#x000F6;sz&#x000F6;rmenyi</surname> <given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>Content-based processing and analysis of endoscopic images and videos: a survey</article-title>. <source>Multimedia Tools Appl</source>. <volume>77</volume>, <fpage>1323</fpage>&#x02013;<lpage>1362</lpage>. <pub-id pub-id-type="doi">10.1007/s11042-016-4219-z</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mur-Artal</surname> <given-names>R.</given-names></name> <name><surname>Montiel</surname> <given-names>J. M. M.</given-names></name> <name><surname>Tardos</surname> <given-names>J. D.</given-names></name></person-group> (<year>2015</year>). <article-title>Orb-slam: a versatile and accurate monocular slam system</article-title>. <source>IEEE Trans. Robot</source>. <volume>31</volume>, <fpage>1147</fpage>&#x02013;<lpage>1163</lpage>. <pub-id pub-id-type="doi">10.1109/TRO.2015.2463671</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Nah</surname> <given-names>S.</given-names></name> <name><surname>Hyun Kim</surname> <given-names>T.</given-names></name> <name><surname>Mu Lee</surname> <given-names>K.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Deep multi-scale convolutional neural network for dynamic scene deblurring,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Honolulu, HI</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>3883</fpage>&#x02013;<lpage>3891</lpage>.<pub-id pub-id-type="pmid">35044913</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pownell</surname> <given-names>P. H.</given-names></name> <name><surname>Minoli</surname> <given-names>J. J.</given-names></name> <name><surname>Rohrich</surname> <given-names>R. J.</given-names></name></person-group> (<year>1997</year>). <article-title>Diagnostic nasal endoscopy</article-title>. <source>Plastic Reconstruct. Surg</source>. <volume>99</volume>, <fpage>1451</fpage>&#x02013;<lpage>1458</lpage>. <pub-id pub-id-type="doi">10.1097/00006534-199704001-00042</pub-id><pub-id pub-id-type="pmid">9105379</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Rombach</surname> <given-names>R.</given-names></name> <name><surname>Blattmann</surname> <given-names>A.</given-names></name> <name><surname>Lorenz</surname> <given-names>D.</given-names></name> <name><surname>Esser</surname> <given-names>P.</given-names></name> <name><surname>Ommer</surname> <given-names>B.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;High-resolution image synthesis with latent diffusion models,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>New Orleans, LA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>10684</fpage>&#x02013;<lpage>10695</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rosenfeld</surname> <given-names>R. M.</given-names></name> <name><surname>Piccirillo</surname> <given-names>J. F.</given-names></name> <name><surname>Chandrasekhar</surname> <given-names>S. S.</given-names></name> <name><surname>Brook</surname> <given-names>I.</given-names></name> <name><surname>Ashok Kumar</surname> <given-names>K.</given-names></name> <name><surname>Kramper</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Clinical practice guideline (update): adult sinusitis</article-title>. <source>Otolaryngol.-Head Neck Surg</source>. <volume>152</volume>, <fpage>S1</fpage>&#x02013;<lpage>S39</lpage>. <pub-id pub-id-type="doi">10.1177/0194599815572097</pub-id><pub-id pub-id-type="pmid">25832968</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Schonberger</surname> <given-names>J. L.</given-names></name> <name><surname>Frahm</surname> <given-names>J.-M.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Structure-from-motion revisited,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>4104</fpage>&#x02013;<lpage>4113</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Shirley</surname> <given-names>P.</given-names></name></person-group> (<year>2018</year>). <source>Ray Tracing in One Weekend</source>. <publisher-loc>Seattle, WA</publisher-loc>: <publisher-name>Amazon Digital Services LLC, 4</publisher-name>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Snavely</surname> <given-names>N.</given-names></name> <name><surname>Seitz</surname> <given-names>S. M.</given-names></name> <name><surname>Szeliski</surname> <given-names>R.</given-names></name></person-group> (<year>2006</year>). <article-title>&#x0201C;Photo tourism: exploring photo collections in 3D,&#x0201D;</article-title> in <source>ACM Siggraph 2006 Papers</source>, <fpage>835</fpage>&#x02013;<lpage>846</lpage>.</citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tewari</surname> <given-names>A.</given-names></name> <name><surname>Fried</surname> <given-names>O.</given-names></name> <name><surname>Thies</surname> <given-names>J.</given-names></name> <name><surname>Sitzmann</surname> <given-names>V.</given-names></name> <name><surname>Lombardi</surname> <given-names>S.</given-names></name> <name><surname>Sunkavalli</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>State of the art on neural rendering</article-title>. <source>Comp. Graphics Forum</source> <volume>39</volume>, <fpage>701</fpage>&#x02013;<lpage>727</lpage>. <pub-id pub-id-type="doi">10.1111/cgf.14022</pub-id></citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Oda</surname> <given-names>M.</given-names></name> <name><surname>Hayashi</surname> <given-names>Y.</given-names></name> <name><surname>Villard</surname> <given-names>B.</given-names></name> <name><surname>Kitasaka</surname> <given-names>T.</given-names></name> <name><surname>Takabatake</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>A visual slam-based bronchoscope tracking scheme for bronchoscopic navigation</article-title>. <source>Int. J. Comp. Assisted Radiol. Surg</source>. <volume>15</volume>, <fpage>1619</fpage>&#x02013;<lpage>1630</lpage>. <pub-id pub-id-type="doi">10.1007/s11548-020-02241-9</pub-id><pub-id pub-id-type="pmid">32770324</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Long</surname> <given-names>Y.</given-names></name> <name><surname>Fan</surname> <given-names>S. H.</given-names></name> <name><surname>Dou</surname> <given-names>Q.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Neural rendering for stereo 3D reconstruction of deformable tissues in robotic surgery,&#x0201D;</article-title> in <source>International Conference on Medical Image Computing and Computer-Assisted Intervention</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer Nature Switzerland</publisher-name>), <fpage>431</fpage>&#x02013;<lpage>441</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Widya</surname> <given-names>A. R.</given-names></name> <name><surname>Monno</surname> <given-names>Y.</given-names></name> <name><surname>Okutomi</surname> <given-names>M.</given-names></name> <name><surname>Suzuki</surname> <given-names>S.</given-names></name> <name><surname>Gotoda</surname> <given-names>T.</given-names></name> <name><surname>Miki</surname> <given-names>K.</given-names></name></person-group> (<year>2019</year>). <article-title>Whole stomach 3D reconstruction and frame localization from monocular endoscope video</article-title>. <source>IEEE J. Transl. Eng. Health Med</source>. <volume>7</volume>, <fpage>1</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1109/JTEHM.2019.2946802</pub-id><pub-id pub-id-type="pmid">32309059</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wyler</surname> <given-names>B.</given-names></name> <name><surname>Mallon</surname> <given-names>W. K.</given-names></name></person-group> (<year>2019</year>). <article-title>Sinusitis update</article-title>. <source>Emerg. Med. Clini</source>. <volume>37</volume>, <fpage>41</fpage>&#x02013;<lpage>54</lpage>. <pub-id pub-id-type="doi">10.1016/j.emc.2018.09.007</pub-id><pub-id pub-id-type="pmid">30454779</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>A.</given-names></name> <name><surname>Huang</surname> <given-names>B.</given-names></name> <name><surname>Sattler</surname> <given-names>T.</given-names></name> <name><surname>Geiger</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>Mip-splatting: Alias-free 3d gaussian splatting</article-title>. <source>arXiv</source> [preprint] arXiv:2311.16493. <pub-id pub-id-type="doi">10.1109/CVPR52733.2024.01839</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zamir</surname> <given-names>S. W.</given-names></name> <name><surname>Arora</surname> <given-names>A.</given-names></name> <name><surname>Khan</surname> <given-names>S.</given-names></name> <name><surname>Hayat</surname> <given-names>M.</given-names></name> <name><surname>Khan</surname> <given-names>F. S.</given-names></name> <name><surname>Yang</surname> <given-names>M.-H.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Restormer: Efficient transformer for high-resolution image restoration,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>New Orleans, LA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>5728</fpage>&#x02013;<lpage>5739</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zwicker</surname> <given-names>M.</given-names></name> <name><surname>Pfister</surname> <given-names>H.</given-names></name> <name><surname>Van Baar</surname> <given-names>J.</given-names></name> <name><surname>Gross</surname> <given-names>M.</given-names></name></person-group> (<year>2001a</year>). <article-title>&#x0201C;Ewa volume splatting,&#x0201D;</article-title> in <source>Proceedings Visualization, 2001</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>29</fpage>&#x02013;<lpage>538</lpage>.</citation>
</ref>
<ref id="B38">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zwicker</surname> <given-names>M.</given-names></name> <name><surname>Pfister</surname> <given-names>H.</given-names></name> <name><surname>Van Baar</surname> <given-names>J.</given-names></name> <name><surname>Gross</surname> <given-names>M.</given-names></name></person-group> (<year>2001b</year>). <article-title>&#x0201C;Surface splatting,&#x0201D;</article-title> in <source>Proceedings of the 28th Annual Conference on Computer Graphics and Interactive Techniques</source> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>371</fpage>&#x02013;<lpage>378</lpage>.</citation>
</ref>
</ref-list>
</back>
</article> 