<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Oncol.</journal-id>
<journal-title>Frontiers in Oncology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Oncol.</abbrev-journal-title>
<issn pub-type="epub">2234-943X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fonc.2023.1120392</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Oncology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Segmentation stability of human head and neck cancer medical images for radiotherapy applications under de-identification conditions: Benchmarking data sharing and artificial intelligence use-cases</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Sahlsten</surname>
<given-names>Jaakko</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wahid</surname>
<given-names>Kareem A.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1784331"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Glerean</surname>
<given-names>Enrico</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/30117"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jaskari</surname>
<given-names>Joel</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Naser</surname>
<given-names>Mohamed A.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1890945"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>He</surname>
<given-names>Renjie</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1951042"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kann</surname>
<given-names>Benjamin H.</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>M&#xe4;kitie</surname>
<given-names>Antti</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Fuller</surname>
<given-names>Clifton D.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/379818"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kaski</surname>
<given-names>Kimmo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1489045"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Computer Science, Aalto University School of Science</institution>, <addr-line>Espoo</addr-line>, <country>Finland</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Radiation Oncology, The University of Texas MD Anderson Cancer Center</institution>, <addr-line>Houston, TX</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Neuroscience and Biomedical Engineering, Aalto University</institution>, <addr-line>Espoo</addr-line>, <country>Finland</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Artificial Intelligence in Medicine Program, Brigham and Women&#x2019;s Hospital, Dana-Farber Cancer Institute, Harvard Medical School</institution>, <addr-line>Boston, MA</addr-line>, <country>United States</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Department of Otorhinolaryngology, Head and Neck Surgery, University of Helsinki and Helsinki University Hospital</institution>, <addr-line>Helsinki</addr-line>, <country>Finland</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Jasper Nijkamp, Aarhus University, Denmark</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Mark J. Gooding, Inpictura Ltd, United Kingdom; Elisa D&#x2019;Angelo, University Hospital of Modena, Italy</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Clifton D. Fuller, <email xlink:href="mailto:cdfuller@mdanderson.org">cdfuller@mdanderson.org</email>; Kimmo Kaski, <email xlink:href="mailto:kimmo.kaski@aalto.fi">kimmo.kaski@aalto.fi</email>
</p>
</fn>
<fn fn-type="other" id="fn002">
<p>This article was submitted to Cancer Imaging and Image-directed Interventions, a section of the journal Frontiers in Oncology</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>28</day>
<month>02</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>13</volume>
<elocation-id>1120392</elocation-id>
<history>
<date date-type="received">
<day>09</day>
<month>12</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>13</day>
<month>02</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Sahlsten, Wahid, Glerean, Jaskari, Naser, He, Kann, M&#xe4;kitie, Fuller and Kaski</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Sahlsten, Wahid, Glerean, Jaskari, Naser, He, Kann, M&#xe4;kitie, Fuller and Kaski</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Demand for head and neck cancer (HNC) radiotherapy data in algorithmic development has prompted increased image dataset sharing. Medical images must comply with data protection requirements so that re-use is enabled without disclosing patient identifiers. Defacing, i.e., the removal of facial features from images, is often considered a reasonable compromise between data protection and re-usability for neuroimaging data. While defacing tools have been developed by the neuroimaging community, their acceptability for radiotherapy applications have not been explored. Therefore, this study systematically investigated the impact of available defacing algorithms on HNC organs at risk (OARs).</p>
</sec>
<sec>
<title>Methods</title>
<p>A publicly available dataset of magnetic resonance imaging scans for 55 HNC patients with eight segmented OARs (bilateral submandibular glands, parotid glands, level II neck lymph nodes, level III neck lymph nodes) was utilized. Eight publicly available defacing algorithms were investigated: afni_refacer, DeepDefacer, defacer, fsl_deface, mask_face, mri_deface, pydeface, and quickshear. Using a subset of scans where defacing succeeded (N=29), a 5-fold cross-validation 3D U-net based OAR auto-segmentation model was utilized to perform two main experiments: 1.) comparing original and defaced data for training when evaluated on original data; 2.) using original data for training and comparing the model evaluation on original and defaced data. Models were primarily assessed using the Dice similarity coefficient (DSC).</p>
</sec>
<sec>
<title>Results</title>
<p>Most defacing methods were unable to produce any usable images for evaluation, while mask_face, fsl_deface, and pydeface were unable to remove the face for 29%, 18%, and 24% of subjects, respectively. When using the original data for evaluation, the composite OAR DSC was statistically higher (p &#x2264; 0.05) for the model trained with the original data with a DSC of 0.760 compared to the mask_face, fsl_deface, and pydeface models with DSCs of 0.742, 0.736, and 0.449, respectively. Moreover, the model trained with original data had decreased performance (p &#x2264; 0.05) when evaluated on the defaced data with DSCs of 0.673, 0.693, and 0.406 for mask_face, fsl_deface, and pydeface, respectively.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>Defacing algorithms may have a significant impact on HNC OAR auto-segmentation model training and testing. This work highlights the need for further development of HNC-specific image anonymization methods.</p>
</sec>
</abstract>
<kwd-group>
<kwd>anonymization</kwd>
<kwd>radiotherapy</kwd>
<kwd>head and neck cancer</kwd>
<kwd>MRI</kwd>
<kwd>medical imaging</kwd>
<kwd>artificial intelligence (AI)</kwd>
<kwd>autosegmentation</kwd>
<kwd>defacing</kwd>
</kwd-group>
<counts>
<fig-count count="4"/>
<table-count count="2"/>
<equation-count count="2"/>
<ref-count count="48"/>
<page-count count="10"/>
<word-count count="4882"/>
</counts>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<title>Introduction</title>
<p>The landscape of data democratization is rapidly changing. The rise of open science practices, inspired by coalitions such as the Center for Open Science (<xref ref-type="bibr" rid="B1">1</xref>), and the FAIR (Findable, Accessible, Interoperable, and Reusable) guiding principles (<xref ref-type="bibr" rid="B2">2</xref>), has spurred interest in public data sharing. Subsequently, the medical imaging community has increasingly adopted these practices through initiatives such as The Cancer Imaging Archive (<xref ref-type="bibr" rid="B3">3</xref>). Given the appropriate removal of protected health information through anonymization techniques, public repositories have democratized the access to medical imaging data such that the world at large can now help develop algorithmic approaches to improve clinical decision-making. Among the medical professions seeking to leverage these large datasets, radiation oncology has the potential to vastly benefit from these open science practices (<xref ref-type="bibr" rid="B4">4</xref>). Imaging is crucial to radiotherapy workflows, particularly for organ at risk (OAR) and tumor segmentation (<xref ref-type="bibr" rid="B5">5</xref>, <xref ref-type="bibr" rid="B6">6</xref>). Moreover, in recent years public data competitions, such as the Head and Neck Tumor Segmentation and Outcome Prediction in positron emission tomography/computed tomography (PET/CT) Images (HECKTOR) challenge (<xref ref-type="bibr" rid="B7">7</xref>&#x2013;<xref ref-type="bibr" rid="B9">9</xref>), have been targeted to improve the radiotherapy workflow. However, there is a particular facet of medical image dissemination for radiotherapy applications that has spurred controversy, namely the anonymization of head and neck cancer (HNC) related images.</p>
<p>While the public dissemination of HNC image data is invaluable to improve the radiotherapy workflow, concerns have been raised regarding readily identifiable facial features on medical imaging. Importantly, the U.S. Health Insurance Portability and Accountability Act references &#x201c;full-face photographs and any comparable images&#x201d; as a part of protected health information (<xref ref-type="bibr" rid="B10">10</xref>). This policy introduces some uncertainty in the dissemination of high-resolution images, where the intricacies of facial features can be reconstructed to generate similar or &#x201c;comparable&#x201d; visualizations with relative ease. Several studies have shown the potential danger in releasing unaltered medical images containing facial features, as they can often be easily recognized by humans and/or machines (<xref ref-type="bibr" rid="B11">11</xref>&#x2013;<xref ref-type="bibr" rid="B15">15</xref>). For example, using facial recognition software paired with image-derived facial reconstructions, one study found up to 83% of research participants could be identified from their magnetic resonance imaging (MRI) scans (<xref ref-type="bibr" rid="B13">13</xref>). Similar alarming results have been demonstrated for CT images (<xref ref-type="bibr" rid="B14">14</xref>). While brain images are often processed such that obvious facial features are removed (i.e., skull stripping), these crude techniques remove large anatomic regions necessary for building predictive models with HNC imaging data. &#x201c;Defacing&#x201d; tools, where voxels that correspond to the areas of the patient&#x2019;s facial features are either removed or altered, offer one solution. However, they may still engender the potential loss of voxel-level information needed for predictive modeling or treatment planning, thereby prohibiting their use in data resharing strategies for radiotherapy applications. While several studies have investigated the effects of defacing for neuroimaging (<xref ref-type="bibr" rid="B16">16</xref>&#x2013;<xref ref-type="bibr" rid="B21">21</xref>), there have not yet been any systematic studies on the effects of defacing tools for radiotherapy applications.</p>
<p>Inspired by the increasing demand for public HNC imaging datasets and the importance of protecting the privacy of patients, a systematic analysis of a number of existing methods for facial anonymization on HNC MRI images was performed. Through qualitative and quantitative analysis using open-source datasets and tools, the efficacies of defacing approaches on whole images and structures relevant to radiation treatment planning were determined. Moreover, the effects of these approaches on auto-segmentation, a specific domain application that is increasingly relevant for HNC public datasets, were also examined. This study is an important first step towards the development of robust approaches for the safe and trusted democratization of HNC imaging data.</p>
</sec>
<sec id="s2">
<title>Methods</title>
<sec id="s2_1">
<title>Dataset</title>
<p>For this analysis, a publicly available dataset hosted on the TCIA, the American Association of Physicists in Medicine RT-MAC Grand Challenge 2019 (AAPM) dataset (<xref ref-type="bibr" rid="B22">22</xref>), was utilized. The AAPM dataset consists of T2-weighted MRI scans of 55 HNC patients that are labeled for OAR segmentations of bilateral: i) submandibular glands, ii) level II neck lymph nodes, iii) level III neck lymph nodes, and iv) parotid glands. Structures were annotated as being on the right or left side of the patient anatomy. The spatial resolution of the scans is 0.5&#xa0;mm &#xd7; 0.5&#xa0;mm with 2.0&#xa0;mm spacing. Additional technical details on the AAPM images and segmentations can be found in the corresponding data descriptor (<xref ref-type="bibr" rid="B22">22</xref>). Defacing experiments were also attempted using the HECKTOR 2021 training dataset (<xref ref-type="bibr" rid="B8">8</xref>) containing 224 HNC patients with CT scans. Additional technical details on the HECKTOR dataset can be found in the corresponding overview papers (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B9">9</xref>).</p>
</sec>
<sec id="s2_2">
<title>Defacing methods</title>
<p>For defacing the images, the same methods as taken into consideration by Schwartz et&#xa0;al. (<xref ref-type="bibr" rid="B16">16</xref>), as well as novel tools that benefit from recent advances in deep learning were used. The most popular tools use a co-registration to a template in order to identify face and ears and then identify those structures in the original image, which should be removed or blurred. The following 6 co-registration based methods: <italic>afni_refacer</italic>, <italic>fsl_deface</italic> (<xref ref-type="bibr" rid="B23">23</xref>), <italic>mask_face</italic> (<xref ref-type="bibr" rid="B24">24</xref>), <italic>mri_deface</italic> (<xref ref-type="bibr" rid="B18">18</xref>), <italic>pydeface</italic> (<xref ref-type="bibr" rid="B25">25</xref>), and <italic>quickshear</italic> were implemented. Two more recent methods using deep learning technology were also included: <italic>defacer</italic> (<xref ref-type="bibr" rid="B26">26</xref>) and <italic>DeepDefacer</italic> (<xref ref-type="bibr" rid="B27">27</xref>). These methods utilize pre-trained deep learning models using data from public neuroimaging datasets to identify facial features to be removed. An automated pipeline for applying all these defacing methods is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/eglerean/faceai_testingdefacing">https://github.com/eglerean/faceai_testingdefacing</ext-link>. Each defacing method was tested with all subjects such that, for each subject, a defaced volume was produced as well as a volumetric mask of which voxels were affected by defacing. All methods were run with the default parameters and standard reference images.</p>
</sec>
<sec id="s2_3">
<title>Defacing performance</title>
<p>After applying the defacing methods, the success or failure of a defacing method was determined by visually inspecting all the defaced volumes (i.e., performing scanwise quality control). Specifically, a binary categorization of each scan was implemented: &#x201c;1&#x201d; if the eyes, nose, and mouth were removed (i.e., defacing succeeded), &#x201c;0&#x201d; if the eyes, nose, or mouth were not removed (i.e., defacing failed). Subsequently, the amount of voxels present in the structures after application of the defacing algorithm were quantitatively measured.</p>
</sec>
<sec id="s2_4">
<title>Deep learning model for OAR segmentation reliability</title>
<p>To evaluate the OAR segmentation performance under different defacing schemes from volumetric MRI data, a convolutional neural network architecture, 3D U-net, which has found wide success in HNC-related segmentation tasks (<xref ref-type="bibr" rid="B28">28</xref>&#x2013;<xref ref-type="bibr" rid="B33">33</xref>), was utilized. Both contractive and expansive pathways include four blocks, where each block consists of two convolutional layers with a kernel size of 3, and each convolution is followed by an instance normalization layer and a LeakyReLU activation with 0.1 negative slope. The max-pooling and transpose convolutional layers have a kernel size and stride of 2. The last convolutional layer has a kernel size and stride of 1 with 9 output channels and a softmax activation. The model architecture is shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>. Experiments were developed in Python v. 3.6.10 (<xref ref-type="bibr" rid="B34">34</xref>) using Pytorch 1.8.1 (<xref ref-type="bibr" rid="B35">35</xref>) with a U-net model from Project MONAI 0.7.0 (<xref ref-type="bibr" rid="B36">36</xref>) and data preprocessing and augmentation with TorchIO 0.18.61 (<xref ref-type="bibr" rid="B37">37</xref>).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>U-net network architecture with blocks on the contractive path colored in red and blocks on the expanding path colored in green. Each block includes two convolutions, each followed by instance normalization and Leaky ReLU activation, subsequently followed by a max-pool layer (red arrow) or transpose convolution layer (green arrow) on contractive and expanding paths, respectively. The number shown in each block indicates the number of channels of the feature map. Arrows with the letter C indicate concatenation.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-13-1120392-g001.tif"/>
</fig>
<p>A subset of patients for which defacing was deemed successful were used for building the segmentation models. The subset was randomly split with 5-fold cross validation: for each cross-validation iteration one fold was used for model testing, one fold was used for model validation, and the remaining three folds were used for model training. The reported segmentation performance was based on the test fold that was not used for model development. The same random splits were used for training and evaluating the models trained on original or defaced data.</p>
<p>Data preprocessing after the defacing included linear resampling to 2&#xa0;mm isotropic resolution with the intensity scaled into a range of [-1,1]. The training data was augmented with random transforms that were applied with a probability (p), independently of each other. The used transforms were random elastic deformations (p=10%) for all axes, random flips for inferior-superior and anterior-posterior axes (p=50%), random rotation (-10&#xb0; to 10&#xb0;) of all axes (p=50%), random bias field (p=50%), and random gamma (p=50%). The model was trained using the cross-entropy loss for the 8 OAR classes and background with parameter updates computed using the Adam optimizer with (0.001 learning rate, 0.9 &#x3b2;<sub>1</sub>, 0.999 &#x3b2;<sub>2</sub>, and AMSGrad). The model training was stopped early after 60 epochs for non-improvement of the validation loss.</p>
</sec>
<sec id="s2_5">
<title>Segmentation evaluation</title>
<p>Two experiments to evaluate the impact of defacing on the resulting segmentations were performed. In order to determine the impact of defacing on algorithmic development, models were trained on original or defaced data using the original target data for evaluation. Subsequently, in order to determine the impact of defacing on algorithms not originally developed for defaced data, a model was trained using the original data and its performance was evaluated by using the original data or the defaced data.</p>
<p>For both experiments, the performance of the models were quantified primarily with the Dice similarity coefficient (DSC) and the mean surface distance (MSD), defined as follows:</p>
<disp-formula>
<label>,</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula>
<label>,</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mo>+</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mstyle>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>TP</italic> denotes true positives, <italic>FP</italic> false positives, <italic>FN</italic> false negatives, <italic>P</italic> the set of segmentation surface voxels of the model output, and <italic>T</italic> the set of segmentation surface voxels of the annotation. The distance from the surface metric is defined as:</p>
<p>
<italic>d</italic>(<italic>a</italic>,&#xa0;<italic>B</italic>)=<italic>min</italic>
<sub>
<italic>b</italic>&#x2208;<italic>B</italic>
</sub>{&#x2016;<italic>a</italic>&#x2212;<italic>b</italic>&#x2016;<sub>2</sub>} . These metrics were selected because of their ubiquity in literature and ability to capture both volumetric overlap and boundary distances (<xref ref-type="bibr" rid="B38">38</xref>, <xref ref-type="bibr" rid="B39">39</xref>). The model output was resampled into the original resolution with the nearest-neighbor sampling and evaluated against the original resolution segmentations. MSD was measured in millimeters. When comparing the performance measures between the segmentation models, Wilcoxon signed rank tests (<xref ref-type="bibr" rid="B40">40</xref>) were implemented, with p-values less than or equal to 0.05 considered as significant. To correct for multiple hypotheses, a Benjamini-Hochberg false discovery rate procedure (<xref ref-type="bibr" rid="B41">41</xref>) was implemented by taking into account all the OARs and models compared. Statistical comparisons were performed using the statannotations 0.4.4 Python package (<ext-link ext-link-type="uri" xlink:href="https://github.com/trevismd/statannotations">https://github.com/trevismd/statannotations</ext-link>). Notably, any ROI metrics that yielded empty outputs were omitted from the comparisons. Additional surface metric values (mean Hausdorff distance at 95% and Hausdorff distance at 95%) were also calculated as part of the supplementary analysis (details in <xref ref-type="supplementary-material" rid="SM1">
<bold>Appendix A</bold>
</xref>).</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>Results</title>
<sec id="s3_1">
<title>Defacing performance</title>
<p>Five of the methods tested (afni_refacer, quickshear, mri_deface, DeepDefacer, and defacer) failed for all subjects in the AAPM dataset. Therefore, for all subsequent analyses only the mask_face, fsl_deface, and pydeface methods were considered. There was scanwise quality control to remove the defaced scans with poor quality from the analyses, which resulted in 16 (29%), 10 (18%), and 13 (24%) scans removed from mask_face, fsl_deface, and pydeface, respectively, with all these methods working on 29 patient scans. A barplot comparison of the ratio of remaining OAR voxels after defacing and quality control is depicted in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>. In addition, the defacing methods removed some OARs completely, which were also omitted from the segmentation evaluation. After filtering unusable data, the total number of OARs available for use in segmentation experiments was 232 for the original data and mask_face, 231 for fsl_deface, and 169 for pydeface. A full comparison of omitted OARs is shown in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Ratio of preserved voxels in comparison to the original segmentation mask after defacing (mask_face, fsl_deface, and pydeface) for each of the organs at risk, where defacing was successful for N=39, N=42, and N=45, respectively. The mean and standard deviation are represented as the center and extremes of the error bars, respectively.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-13-1120392-g002.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Quantitative details on the number of organs at risk available after the defacing was applied for all 55 patient scans.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="middle" colspan="3" align="center">Completely removed after successful defacing</th>
<th valign="middle" colspan="3" align="center">Unavailable for segmentation analysis*</th>
</tr>
<tr>
<th valign="top" align="left">Organ at risk/Defacing method</th>
<th valign="top" align="center">mask_face</th>
<th valign="top" align="center">fsl_deface</th>
<th valign="top" align="center">pydeface</th>
<th valign="top" align="center">mask_face</th>
<th valign="top" align="center">fsl_deface</th>
<th valign="top" align="center">pydeface</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Left Submandibular Gland</td>
<td valign="top" align="center">0 (0%)</td>
<td valign="top" align="center">2 (4%)</td>
<td valign="top" align="center">6 (11%)</td>
<td valign="top" align="center">16 (29%)</td>
<td valign="top" align="center">14 (25%)</td>
<td valign="top" align="center">13 (24%)</td>
</tr>
<tr>
<td valign="top" align="left">Right Submandibular Gland</td>
<td valign="top" align="center">0 (0%)</td>
<td valign="top" align="center">1 (2%)</td>
<td valign="top" align="center">7 (13%)</td>
<td valign="top" align="center">16 (29%)</td>
<td valign="top" align="center">14 (25%)</td>
<td valign="top" align="center">14 (25%)</td>
</tr>
<tr>
<td valign="top" align="left">Left Neck Lymph Node Level II</td>
<td valign="top" align="center">0 (0%)</td>
<td valign="top" align="center">0 (0%)</td>
<td valign="top" align="center">4 (7%)</td>
<td valign="top" align="center">16 (29%)</td>
<td valign="top" align="center">13 (24%)</td>
<td valign="top" align="center">11 (20%)</td>
</tr>
<tr>
<td valign="top" align="left">Right Neck Lymph Node Level II</td>
<td valign="top" align="center">0 (0%)</td>
<td valign="top" align="center">0 (0%)</td>
<td valign="top" align="center">4 (7%)</td>
<td valign="top" align="center">16 (29%)</td>
<td valign="top" align="center">13 (24%)</td>
<td valign="top" align="center">11 (20%)</td>
</tr>
<tr>
<td valign="top" align="left">Left Neck Lymph Node Level III</td>
<td valign="top" align="center">0 (0%)</td>
<td valign="top" align="center">0 (0%)</td>
<td valign="top" align="center">45 (82%)</td>
<td valign="top" align="center">16 (29%)</td>
<td valign="top" align="center">13 (24%)</td>
<td valign="top" align="center">51 (93%)</td>
</tr>
<tr>
<td valign="top" align="left">Right Neck Lymph Node Level III</td>
<td valign="top" align="center">0 (0%)</td>
<td valign="top" align="center">1 (2%)</td>
<td valign="top" align="center">44 (80%)</td>
<td valign="top" align="center">16 (29%)</td>
<td valign="top" align="center">13 (24%)</td>
<td valign="top" align="center">50 (91%)</td>
</tr>
<tr>
<td valign="top" align="left">Left Parotid</td>
<td valign="top" align="center">0 (0%)</td>
<td valign="top" align="center">0 (0%)</td>
<td valign="top" align="center">6 (11%)</td>
<td valign="top" align="center">16 (29%)</td>
<td valign="top" align="center">13 (24%)</td>
<td valign="top" align="center">11 (20%)</td>
</tr>
<tr>
<td valign="top" align="left">Right Parotid</td>
<td valign="top" align="center">0 (0%)</td>
<td valign="top" align="center">0 (0%)</td>
<td valign="top" align="center">6 (11%)</td>
<td valign="top" align="center">16 (29%)</td>
<td valign="top" align="center">13 (24%)</td>
<td valign="top" align="center">11 (20%)</td>
</tr>
<tr>
<td valign="top" align="left">Total omitted</td>
<td valign="top" align="center">0 (0%)</td>
<td valign="top" align="center">4 (1%)</td>
<td valign="top" align="center">122 (28%)</td>
<td valign="top" align="center">128 (29%)</td>
<td valign="top" align="center">106 (24%)</td>
<td valign="top" align="center">172 (39%)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Only the mask_face, fsl_deface, and pydeface methods yielded usable data. The first group of columns correspond to the organs at risk that were completely removed from the cases with successful defacing. The second group of columns correspond to all items in the first group of columns plus incorporating any of the cases where defacing failed. Defacing success or failure was counted from scanwise quality control. *Organs at risk in these columns were omitted for all the subsequent segmentation-related experiments.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>All of the tested defacing methods were unable to provide sufficient data for segmentation analysis in the HECKTOR CT dataset. Specifically, fsl_deface and pydeface methods successfully defaced 18 (8%) and 102 (46%) scans, respectively. All other methods (afni_refacer, quickshear, mri_deface, DeepDefacer, defacer, and mask_face) failed to correctly deface any of the scans. Although pydeface had the highest success rate on defacing, it only preserved the brain. Thus, no further analysis was performed for this dataset.</p>
</sec>
<sec id="s3_2">
<title>Segmentation performance</title>
<p>The 29 patient scans for which the defacing was deemed successful were used to construct and evaluate segmentation models for the mask_face, fsl_deface, and pydeface methods. The model DSC performances pooled across all structures based on training input and valid evaluation target combinations are shown in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. The models trained using the original, mask_face, and fsl_deface input data had the highest composite mean DSC when evaluated on the original target data with values of 0.760, 0.742, and 0.736, respectively, while the model trained on pydeface input data had the highest composite mean DSC of 0.653 when evaluated on pydeface target data. In contrast, the models trained using original mask_face, and fsl_deface input data had the lowest composite mean DSC when evaluated on pydeface target data with values of 0.406, 0.413, 0.465, respectively, while the model trained using pydeface input data had the lowest composite mean DSC of 0.395 when evaluated on fsl_deface target data. All comparisons within the same evaluation data are statistically different from each other (p &#x2264; 0.05) with the exception of mask_face and fsl_deface trained models evaluated on original data, and original as well as mask_face trained models evaluated on pydeface data.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Composite DSC performance - mean (standard deviation) - of all structures for all combinations of training data (rows) and evaluation data (columns).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="top" align="center">Evaluated on original (N =2 32)</th>
<th valign="top" align="center">Evaluated on mask_face (N = 232)</th>
<th valign="top" align="center">Evaluated on fsl_deface (N = 231)</th>
<th valign="top" align="center">Evaluated on pydeface (N = 169)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Trained on original</td>
<td valign="top" align="center">0.760 (0.112)</td>
<td valign="top" align="center">0.673 (0.181)</td>
<td valign="top" align="center">0.693 (0.140)</td>
<td valign="top" align="center">0.406 (0.304)</td>
</tr>
<tr>
<td valign="top" align="left">Trained on mask_face</td>
<td valign="top" align="center">0.742 (0.115)</td>
<td valign="top" align="center">0.733 (0.120)</td>
<td valign="top" align="center">0.668 (0.143)</td>
<td valign="top" align="center">0.413 (0.312)</td>
</tr>
<tr>
<td valign="top" align="left">Trained on fsl_deface</td>
<td valign="top" align="center">0.736 (0.108)</td>
<td valign="top" align="center">0.643 (0.185)</td>
<td valign="top" align="center">0.733 (0.122)</td>
<td valign="top" align="center">0.465 (0.293)</td>
</tr>
<tr>
<td valign="top" align="left">Trained on pydeface</td>
<td valign="top" align="center">0.449 (0.333)</td>
<td valign="top" align="center">0.417 (0.325)</td>
<td valign="top" align="center">0.395 (0.301)</td>
<td valign="top" align="center">0.653 (0.258)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The number of total segmentation maps evaluated is shown in brackets on the header. All comparisons within the same evaluation data are statistically different from each other (p &#x2264; 0.05) with the exception of mask_face and fsl_deface trained models evaluated on original data, and original and mask_face trained models evaluated on pydeface data. Statistical significance was measured with Wilcoxon signed-rank tests corrected with Benjamini-Hochberg procedure comparisons within evaluation data.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<sec id="s3_2_1">
<title>Defacing impact on model training</title>
<p>The analysis was based on eight OAR structure segmentations from 29 patients totaling 232 evaluations. The MSD of left and right level III neck lymph nodes for pydeface trained models were omitted from the analysis as all the model outputs were empty. Full comparisons of the model performance for each OAR are depicted in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>. Additional surface distance metrics are shown in <xref ref-type="supplementary-material" rid="SM1">
<bold>Appendix A (Figure A1)</bold>
</xref>. Overall, the model trained with the original data performed better than the models trained with the defaced data for the majority of structures and evaluation metrics. Both metrics were significantly better for the model trained with the original data compared to the model trained with mask_face data for the left submandibular gland and right level II neck lymph node, while only the DSC was significantly better for the right submandibular gland and right level III neck lymph node. Similarly, both metrics were significantly better for the model trained with the original data compared to the model trained with fsl_deface data for the right level II neck lymph node, left parotid, and right parotid, while only the DSC was significantly better for the right level III neck lymph node. Moreover, both metrics were significantly better for the model trained with the original data compared to the model trained with pydeface data for all the structures.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Performance of the models trained on original or defaced data and evaluated on the original data. The mean and standard deviation for each metric are represented as the center and extremes of the error bars, respectively. Statistical significance was determined using Wilcoxon signed-rank tests corrected with Benjamini-Hochberg procedure for all OARs and models. Comparison symbols: ns (p &gt; 0.05), * (p &#x2264; 0.05), ** (p &#x2264; 0.01), *** (p &#x2264; 1e-4), **** (p &#x2264; 1e-5).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-13-1120392-g003.tif"/>
</fig>
</sec>
<sec id="s3_2_2">
<title>Defacing impact on model testing</title>    <p>In these results, only valid target data with successful defacing on all three methods using non-empty segmentation structures were included. This was obtained using results from 26 left submandibular glands, 27 right submandibular glands, 1 left neck level III lymph nodes, 2 right neck level III lymph nodes, and 28 of each of the remaining structures. Due to the low number of cases for the right and left level III lymph nodes, they were omitted from the comparison. In addition, for the MSD metric, empty model output segmentations were discarded resulting in evaluation of 1 left submandibular gland for fsl_deface and mask_face and 14 for pydeface, 1 and 6 right submandibular glands on fsl_deface and pydeface, respectively, 1 left level II lymph node for pydeface, and 2 left parotids for pydeface. The model evaluated on the original data performed significantly better than the models evaluated on the defaced data for all of the structures and both evaluation metrics except in the case of left submandibular gland DSC for fsl_deface which exhibited a non-significant difference. The full comparison of the model performance for each of the OARs is shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>. Additional surface distance metrics are shown in <xref ref-type="supplementary-material" rid="SM1">
<bold>Appendix A (Figure A2)</bold>
</xref>.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>The performance of models trained on the original data when evaluated on the original, mask_face, fsl_deface, or pydeface data for the six organs at risk included in the analysis. Only cases that were available for all the methods were included: 28 segmentations were used for all structures except in the case of the left and right submandibular glands where 26 and 27 segmentations were used, respectively. In addition, for the MSD metric, empty model output segmentations were discarded, which resulted in a smaller number of evaluated structures. The number of evaluated structures is shown on top of the barplot. The mean and standard deviation for each metric are represented as the center and extremes of the error bars, respectively. Statistical significance was measured with Wilcoxon signed-rank tests corrected with Benjamini-Hochberg procedure for all OARs and models. Comparison symbols: ns (p &gt; 0.05), * (p &#x2264; 0.05), ** (p &#x2264; 0.01), *** (p &#x2264; 1e-4), **** (p &#x2264; 1e-5).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-13-1120392-g004.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<title>Discussion</title>
<p>This study has systematically investigated the impact of a variety of defacing algorithms on structures of interest used for radiotherapy treatment planning. This study demonstrated that the overall usability of segmentations is heavily dependent on the choice of the defacing algorithm. Moreover, the results indicate that several OARs have the potential to be negatively impacted by the defacing algorithms, which is shown by the decreased performance of auto-segmentation algorithms trained and evaluated on defaced data in comparison to algorithms trained and evaluated on non-defaced data.</p>
<p>Defacing for HNC applications should be deemed optimal if the method simultaneously removes all recognizable facial features from the image and no voxels from structures of interest are affected. In this study, eight commonly available defacing algorithms developed by the neuroimaging community were applied: afni_refacer, mri_deface, defacer, DeepDefacer, mask_face, fsl_deface, pydeface, and quickshear. Unfortunately, for the investigated CT data, no defacing method was able to yield successful removal of facial features while preserving the OARs. This is not necessarily surprising given that the methods investigated were developed primarily with MRI in mind; these results echo previous similar work using CT data (<xref ref-type="bibr" rid="B42">42</xref>). Importantly, even when applied to MRI data of HNC patients, many of these defacing methods outright failed for most if not all patients. Therefore, despite extant studies demonstrating the acceptability of these methods to remove facial features from neuroimaging scans (<xref ref-type="bibr" rid="B16">16</xref>&#x2013;<xref ref-type="bibr" rid="B21">21</xref>), these tools may not necessarily be robust to HNC-related imaging. Moreover, for those defacing algorithms that were able to successfully remove facial information in the MRI data, i.e. mask_face, fsl_deface, and pydeface, it was shown that regardless of the choice of the method, there was a loss of voxel-level information for all the OAR structures investigated. Importantly, pydeface leads to a greater number of lost voxels than mask_face and fsl_deface for all the OAR structures, with the exception of the parotid glands. While mask_face and fsl_deface lead to relatively minimal reduction of available voxels in many cases, the loss of topographic information in a radiotherapy workflow cannot be underscored enough. It is well known that even minor variations in the delineation of tumors and OARs can drastically alter the resulting radiotherapy dose delivered to a patient, which can impact important clinical outcomes such as toxicity and overall survival (<xref ref-type="bibr" rid="B43">43</xref>&#x2013;<xref ref-type="bibr" rid="B46">46</xref>). Therefore, the loss of voxel-level information of OARs caused by the defacing algorithms, while potentially visibly imperceptible, can still affect downstream clinical workflows.</p>
<p>Relatively few studies have been conducted that determined the downstream analysis effects of defacing algorithms. For example, recent studies by Schwartz et&#xa0;al. (<xref ref-type="bibr" rid="B16">16</xref>) and Mikulan et&#xa0;al. (<xref ref-type="bibr" rid="B21">21</xref>) demonstrated that several defacing methods showed differences in specific neuroimaging applications, namely brain volume measurements and electroencephalography-related calculations. In this study, as a proxy for a clinically relevant task, an OAR auto-segmentation workflow was developed to investigate the impact of defacing-induced voxel-level information loss on downstream radiotherapy applications. As evident through both pooled analysis and investigation of individual OARs for auto-segmentation model training and evaluation, performance is often modestly decreased for fsl_deface and mask_face but greatly decreased for pydeface; these results were consistent with the overall voxel-level information loss. While pydeface has been shown to have favorable results for use with neuroimaging data (<xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B21">21</xref>), its negative impact on HNC imaging is apparent. Therefore, in cases where defacing is unavoidable, mask_face or fsl_deface should likely be preferred for HNC image anonymization. Regardless, this study demonstrates existing approaches to anonymize facial data may not be sufficient for implementation on HNC-related datasets, particularly for deep learning model training and testing.</p>
<p>This study has several limitations. Firstly, to examine defacing methods as they are currently distributed (&#x201c;out-of-the-box&#x201d;), modifications to the templates or models utilized in any methods were not performed. Further preprocessing either of the CT and MRI data as well as subject specific settings could have helped some of the methods to better identify the face. In addition, more suitable templates for the HNC images (for both CT and MRI) would likely improve the defacing performance; for the registration-based methods, algorithms likely expected scans to cover the whole brain, while the field-of-view of the images for HNC mostly covered the neck and mouth, leaving the top of the brain excluded. Notably, additional deep learning model training schemes (i.e., transfer learning) may potentially allow for eventual implementation of existing deep learning methods on domain-specific datasets (i.e., HNC radiotherapy), but this negates the immediate interoperability of these tools. Furthermore, no additional image processing other than what was integrated into the defacing methods was implemented; it may be possible alternative processing could change these results. Secondly, while a robust analysis utilizing multiple relevant metrics established in existing literature (<xref ref-type="bibr" rid="B38">38</xref>) was performed to evaluate OAR auto-segmentation, there is not always a perfect correlation between spatial similarity metrics and radiotherapy plan acceptability (<xref ref-type="bibr" rid="B39">39</xref>). This study has not tested the downstream effects of defacing on radiotherapy plan generation, which may lead to different results from what was observed for the OAR segmentation. Thirdly, this study was limited to public data with no modifications. Only structures that were already available in existing datasets were analyzed. Moreover, as an initial exploration of defacing methods for radiotherapy applications, only a single imaging modality on a relatively limited sample size, namely T2-weighted MRI, was investigated for auto-segmentation experiments, despite the HNC radiotherapy workflow commonly incorporating additional modalities (<xref ref-type="bibr" rid="B47">47</xref>). Thus, experiments on additional imaging modalities and larger diverse HNC patient populations should be the subject of future investigations. Fourthly, the current analysis does not thoroughly explore possible performance confounding related to phenotypical and individual variables such as sex, ethnicity, and age of the measured individuals. Finally, this study has focused on defacing methods as an avenue for public data sharing for training and evaluating machine learning models, but privacy-preserving modeling approaches, e.g., through federated learning (<xref ref-type="bibr" rid="B48">48</xref>), may also act as a potential alternative solution.</p>
</sec>
<sec id="s5" sec-type="conclusion">
<title>Conclusion</title>
<p>In summary, by using publicly available data, the effects of eight established defacing algorithms, afni_refacer, mask_face, mri_deface, defacer, DeepDefacer, quickshear, fsl_deface, and pydeface, have been systematically investigated for radiotherapy applications. Specifically, the impact of defacing directly on ground-truth HNC OARs was determined and a deep learning based OAR auto-segmentation workflow to investigate the use of defaced data for algorithmic training and evaluation was developed. All methods failed to properly remove facial features on the CT dataset investigated. Moreover, it was observed that only fsl_deface, mask_face, and pydeface yielded usable images from the MRI dataset, but still decreased the total number of voxels in OARs and negatively impacted the performance of OAR auto-segmentation, with pydeface having more severe negative effects than mask_face or fsl_deface. This study is an important step towards ensuring widespread privacy-preserving dissemination of HNC imaging data without endangering data usability. Given that current defacing methods remove critical data, future larger studies should investigate alternative approaches for anonymizing facial data that preserve radiotherapy-related structures. Moreover, studies on the impact of these methods on radiotherapy plan generation, the inclusion of a greater number of OARs and target structures, and the incorporation of additional imaging modalities are also warranted.</p>
</sec>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://wiki.cancerimagingarchive.net/display/Public/AAPM+RT-MAC+Grand+Challenge+2019">https://wiki.cancerimagingarchive.net/display/Public/AAPM+RT-MAC+Grand+Challenge+2019</ext-link>.</p>
</sec>
<sec id="s7" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>Ethical review and approval was not required for the study on human participants in accordance with the local legislation and institutional requirements. Written informed consent for participation was not required for this study in accordance with the national legislation and the institutional requirements.</p>
</sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>Study concepts: all authors; Study design: JS, EG, and JJ; Data acquisition: KW, MN, and RH; Quality control of data and algorithms: JS and EG; Data analysis and interpretation: JS, KW, EG, JJ, BK, AM, and KK; Manuscript editing: JS, KW, EG, JJ, BK, AM, KK, and CF. All authors contributed to the article and approved the submitted version.</p>
</sec>
</body>
<back>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>This work was supported by the National Institutes of Health (NIH)/National Cancer Institute (NCI) through a Cancer Center Support Grant (CCSG; P30CA016672-44). MN is supported by an NIH grant (R01DE028290-01). KW is supported by a training fellowship from The University of Texas Health Science Center at Houston Center for Clinical and Translational Sciences TL1 Program (TL1TR003169), the American Legion Auxiliary Fellowship in Cancer Research, and an NIH/National Institute for Dental and Craniofacial Research (NIDCR) F31 fellowship (1 F31DE031502-01). CF received funding from the NIH/NIDCR (1R01DE025248-01/R56DE025248); an NIH/NIDCR Academic-Industrial Partnership Award (R01DE028290); the National Science Foundation (NSF), Division of Mathematical Sciences, Joint NIH/NSF Initiative on Quantitative Approaches to Biomedical Big Data (QuBBD) Grant (NSF 1557679); the NIH Big Data to Knowledge (BD2K) Program of the NCI Early Stage Development of Technologies in Biomedical Computing, Informatics, and Big Data Science Award (1R01CA214825); the NCI Early Phase Clinical Trials in Imaging and Image-Guided Interventions Program (1R01CA218148); an NIH/NCI Pilot Research Program Award from the UT MD Anderson CCSG Radiation Oncology and Cancer Imaging Program (P30CA016672); an NIH/NCI Head and Neck Specialized Programs of Research Excellence (SPORE) Developmental Research Program Award (P50CA097007); and the National Institute of Biomedical Imaging and Bioengineering (NIBIB) Research Education Program (R25EB025787).</p>
</sec>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fonc.2023.1120392/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fonc.2023.1120392/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet_1.pdf" id="SM1" mimetype="application/pdf"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Foster</surname> <given-names>ED</given-names>
</name>
<name>
<surname>Deardorff</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Open science framework (OSF)</article-title>. <source>J Med Libr Assoc JMLA</source> (<year>2017</year>) <volume>105</volume>:<fpage>203</fpage>. doi: <pub-id pub-id-type="doi">10.5195/jmla.2017.88</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wilkinson</surname> <given-names>MD</given-names>
</name>
<name>
<surname>Dumontier</surname> <given-names>M</given-names>
</name>
<name>
<surname>IjJ</surname> <given-names>A</given-names>
</name>
<name>
<surname>Appleton</surname> <given-names>G</given-names>
</name>
<name>
<surname>Axton</surname> <given-names>M</given-names>
</name>
<name>
<surname>Baak</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>The FAIR guiding principles for scientific data management and stewardship</article-title>. <source>Sci Data</source> (<year>2016</year>) <volume>3</volume>:<fpage>1</fpage>&#x2013;<lpage>9</lpage>. doi: <pub-id pub-id-type="doi">10.1038/sdata.2016.18</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clark</surname> <given-names>K</given-names>
</name>
<name>
<surname>Vendt</surname> <given-names>B</given-names>
</name>
<name>
<surname>Smith</surname> <given-names>K</given-names>
</name>
<name>
<surname>Freymann</surname> <given-names>J</given-names>
</name>
<name>
<surname>Kirby</surname> <given-names>J</given-names>
</name>
<name>
<surname>Koppel</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>The cancer imaging archive (TCIA): Maintaining and operating a public information repository</article-title>. <source>J Digit Imaging</source> (<year>2013</year>) <volume>26</volume>:<page-range>1045&#x2013;57</page-range>. doi: <pub-id pub-id-type="doi">10.1007/s10278-013-9622-7</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wahid</surname> <given-names>KA</given-names>
</name>
<name>
<surname>Glerean</surname> <given-names>E</given-names>
</name>
<name>
<surname>Sahlsten</surname> <given-names>J</given-names>
</name>
<name>
<surname>Jaskari</surname> <given-names>J</given-names>
</name>
<name>
<surname>Kaski</surname> <given-names>K</given-names>
</name>
<name>
<surname>Naser</surname> <given-names>MA</given-names>
</name>
<etal/>
</person-group>. <article-title>Artificial intelligence for radiation oncology applications using public datasets</article-title>. In: <source>Seminars in radiation oncology</source>. <publisher-name>Elsevier</publisher-name> (<year>2022</year>). p. <page-range>400&#x2013;14</page-range>.</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Press</surname> <given-names>RH</given-names>
</name>
<name>
<surname>Shu</surname> <given-names>H-KG</given-names>
</name>
<name>
<surname>Shim</surname> <given-names>H</given-names>
</name>
<name>
<surname>Mountz</surname> <given-names>JM</given-names>
</name>
<name>
<surname>Kurland</surname> <given-names>BF</given-names>
</name>
<name>
<surname>Wahl</surname> <given-names>RL</given-names>
</name>
<etal/>
</person-group>. <article-title>The use of quantitative imaging in radiation oncology: A quantitative imaging network (QIN) perspective</article-title>. <source>Int J Radiat Oncol Biol Phys</source> (<year>2018</year>) <volume>102</volume>:<page-range>1219&#x2013;35</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.ijrobp.2018.06.023</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beaton</surname> <given-names>L</given-names>
</name>
<name>
<surname>Bandula</surname> <given-names>S</given-names>
</name>
<name>
<surname>Gaze</surname> <given-names>MN</given-names>
</name>
<name>
<surname>Sharma</surname> <given-names>RA</given-names>
</name>
</person-group>. <article-title>How rapid advances in imaging are defining the future of precision radiation oncology</article-title>. <source>Br J Cancer</source> (<year>2019</year>) <volume>120</volume>:<page-range>779&#x2013;90</page-range>. doi: <pub-id pub-id-type="doi">10.1038/s41416-019-0412-y</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Andrearczyk</surname> <given-names>V</given-names>
</name>
<name>
<surname>Oreiller</surname> <given-names>V</given-names>
</name>
<name>
<surname>Jreige</surname> <given-names>M</given-names>
</name>
<name>
<surname>Valli&#xe8;res</surname> <given-names>M</given-names>
</name>
<name>
<surname>Castelli</surname> <given-names>J</given-names>
</name>
<name>
<surname>Elhalawani</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>Overview of the HECKTOR challenge at MICCAI 2020: automatic head and neck tumor segmentation in PET/CT</article-title>. In: <source>3D head and neck tumor segmentation in PET/CT challenge</source>. <publisher-name>Springer</publisher-name> (<year>2020</year>). p. <fpage>1</fpage>&#x2013;<lpage>21</lpage>.</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Andrearczyk</surname> <given-names>V</given-names>
</name>
<name>
<surname>Oreiller</surname> <given-names>V</given-names>
</name>
<name>
<surname>Boughdad</surname> <given-names>S</given-names>
</name>
<name>
<surname>Rest</surname> <given-names>CCL</given-names>
</name>
<name>
<surname>Elhalawani</surname> <given-names>H</given-names>
</name>
<name>
<surname>Jreige</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Overview of the HECKTOR challenge at MICCAI 2021: Automatic head and neck tumor segmentation and outcome prediction in PET/CT images</article-title>. In: <source>3D head and neck tumor segmentation in PET/CT challenge</source>. <publisher-name>(Strasbourg, France: Springer)</publisher-name> (<year>2021</year>). p. <fpage>1</fpage>&#x2013;<lpage>37</lpage>.</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oreiller</surname> <given-names>V</given-names>
</name>
<name>
<surname>Andrearczyk</surname> <given-names>V</given-names>
</name>
<name>
<surname>Jreige</surname> <given-names>M</given-names>
</name>
<name>
<surname>Boughdad</surname> <given-names>S</given-names>
</name>
<name>
<surname>Elhalawani</surname> <given-names>H</given-names>
</name>
<name>
<surname>Castelli</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Head and neck tumor segmentation in PET/CT: The HECKTOR challenge</article-title>. <source>Med Image Anal</source> (<year>2022</year>) <volume>77</volume>:<elocation-id>102336</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.media.2021.102336</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meystre</surname> <given-names>SM</given-names>
</name>
<name>
<surname>Friedlin</surname> <given-names>FJ</given-names>
</name>
<name>
<surname>South</surname> <given-names>BR</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>S</given-names>
</name>
<name>
<surname>Samore</surname> <given-names>MH</given-names>
</name>
</person-group>. <article-title>Automatic de-identification of textual documents in the electronic health record: A review of recent research</article-title>. <source>BMC Med Res Methodol</source> (<year>2010</year>) <volume>10</volume>:<fpage>1</fpage>&#x2013;<lpage>16</lpage>. doi: <pub-id pub-id-type="doi">10.1186/1471-2288-10-70</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prior</surname> <given-names>FW</given-names>
</name>
<name>
<surname>Brunsden</surname> <given-names>B</given-names>
</name>
<name>
<surname>Hildebolt</surname> <given-names>C</given-names>
</name>
<name>
<surname>Nolan</surname> <given-names>TS</given-names>
</name>
<name>
<surname>Pringle</surname> <given-names>M</given-names>
</name>
<name>
<surname>Vaishnavi</surname> <given-names>SN</given-names>
</name>
<etal/>
</person-group>. <article-title>Facial recognition from volume-rendered magnetic resonance imaging data</article-title>. <source>IEEE Trans Inf Technol BioMed</source> (<year>2008</year>) <volume>13</volume>:<fpage>5</fpage>&#x2013;<lpage>9</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TITB.2008.2003335</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mazura</surname> <given-names>JC</given-names>
</name>
<name>
<surname>Juluru</surname> <given-names>K</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>JJ</given-names>
</name>
<name>
<surname>Morgan</surname> <given-names>TA</given-names>
</name>
<name>
<surname>John</surname> <given-names>M</given-names>
</name>
<name>
<surname>Siegel</surname> <given-names>EL</given-names>
</name>
</person-group>. <article-title>Facial recognition software success rates for the identification of 3D surface reconstructed facial images: Implications for patient privacy and security</article-title>. <source>J Digit Imaging</source> (<year>2012</year>) <volume>25</volume>:<page-range>347&#x2013;51</page-range>. doi: <pub-id pub-id-type="doi">10.1007/s10278-011-9429-3</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schwarz</surname> <given-names>CG</given-names>
</name>
<name>
<surname>Kremers</surname> <given-names>WK</given-names>
</name>
<name>
<surname>Therneau</surname> <given-names>TM</given-names>
</name>
<name>
<surname>Sharp</surname> <given-names>RR</given-names>
</name>
<name>
<surname>Gunter</surname> <given-names>JL</given-names>
</name>
<name>
<surname>Vemuri</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>Identification of anonymous MRI research participants with face-recognition software</article-title>. <source>N Engl J Med</source> (<year>2019</year>) <volume>381</volume>:<page-range>1684&#x2013;6</page-range>. doi: <pub-id pub-id-type="doi">10.1056/NEJMc1908881</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Parks</surname> <given-names>CL</given-names>
</name>
<name>
<surname>Monson</surname> <given-names>KL</given-names>
</name>
</person-group>. <article-title>Automated facial recognition of computed tomography-derived facial images: patient privacy implications</article-title>. <source>J Digit Imaging</source> (<year>2017</year>) <volume>30</volume>:<page-range>204&#x2013;14</page-range>. doi: <pub-id pub-id-type="doi">10.1007/s10278-016-9932-7</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Delbarre</surname> <given-names>DJ</given-names>
</name>
<name>
<surname>Santos</surname> <given-names>L</given-names>
</name>
<name>
<surname>Ganjgahi</surname> <given-names>H</given-names>
</name>
<name>
<surname>Horner</surname> <given-names>N</given-names>
</name>
<name>
<surname>McCoy</surname> <given-names>A</given-names>
</name>
<name>
<surname>Westerberg</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>Application of a convolutional neural network to the quality control of MRI defacing</article-title>. <source>Comput Biol Med</source> (<year>2022</year>) <volume>151</volume>:<elocation-id>106211</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.106211</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schwarz</surname> <given-names>CG</given-names>
</name>
<name>
<surname>Kremers</surname> <given-names>WK</given-names>
</name>
<name>
<surname>Wiste</surname> <given-names>HJ</given-names>
</name>
<name>
<surname>Gunter</surname> <given-names>JL</given-names>
</name>
<name>
<surname>Vemuri</surname> <given-names>P</given-names>
</name>
<name>
<surname>Spychalla</surname> <given-names>AJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Changing the face of neuroimaging research: Comparing a new MRI de-facing technique with popular alternatives</article-title>. <source>NeuroImage</source> (<year>2021</year>) <volume>231</volume>:<fpage>117845</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuroimage.2021.117845</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schimke</surname> <given-names>N</given-names>
</name>
<name>
<surname>Kuehler</surname> <given-names>M</given-names>
</name>
<name>
<surname>Hale</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Preserving privacy in structural neuroimages</article-title>. In: <source>IFIP annual conference on data and applications security and privacy</source>. <publisher-name>(Richmond, VA, USA: Springer)</publisher-name> (<year>2011</year>). p. <page-range>301&#x2013;8</page-range>.</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bischoff-Grethe</surname> <given-names>A</given-names>
</name>
<name>
<surname>Ozyurt</surname> <given-names>IB</given-names>
</name>
<name>
<surname>Busa</surname> <given-names>E</given-names>
</name>
<name>
<surname>Quinn</surname> <given-names>BT</given-names>
</name>
<name>
<surname>Fennema-Notestine</surname> <given-names>C</given-names>
</name>
<name>
<surname>Clark</surname> <given-names>CP</given-names>
</name>
<etal/>
</person-group>. <article-title>A technique for the deidentification of structural brain MR images</article-title>. <source>Hum Brain Mapp</source> (<year>2007</year>) <volume>28</volume>:<fpage>892</fpage>&#x2013;<lpage>903</lpage>. doi: <pub-id pub-id-type="doi">10.1002/hbm.20312</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Theyers</surname> <given-names>AE</given-names>
</name>
<name>
<surname>Zamyadi</surname> <given-names>M</given-names>
</name>
<name>
<surname>O&#x2019;Reilly</surname> <given-names>M</given-names>
</name>
<name>
<surname>Bartha</surname> <given-names>R</given-names>
</name>
<name>
<surname>Symons</surname> <given-names>S</given-names>
</name>
<name>
<surname>MacQueen</surname> <given-names>GM</given-names>
</name>
<etal/>
</person-group>. <article-title>Multisite comparison of MRI defacing software across multiple cohorts</article-title>. <source>Front Psychiatry</source> (<year>2021</year>) <volume>12</volume>:<elocation-id>189</elocation-id>. doi: <pub-id pub-id-type="doi">10.3389/fpsyt.2021.617997</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Sitter</surname> <given-names>A</given-names>
</name>
<name>
<surname>Visser</surname> <given-names>M</given-names>
</name>
<name>
<surname>Brouwer</surname> <given-names>I</given-names>
</name>
<name>
<surname>Cover</surname> <given-names>K</given-names>
</name>
<name>
<surname>van Schijndel</surname> <given-names>R</given-names>
</name>
<name>
<surname>Eijgelaar</surname> <given-names>R</given-names>
</name>
<etal/>
</person-group>. <article-title>Facing privacy in neuroimaging: removing facial features degrades performance of image analysis methods</article-title>. <source>Eur Radiol</source> (<year>2020</year>) <volume>30</volume>:<page-range>1062&#x2013;74</page-range>. doi: <pub-id pub-id-type="doi">10.1007/s00330-019-06459-3</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mikulan</surname> <given-names>E</given-names>
</name>
<name>
<surname>Russo</surname> <given-names>S</given-names>
</name>
<name>
<surname>Zauli</surname> <given-names>FM</given-names>
</name>
<name>
<surname>d&#x2019;Orio</surname> <given-names>P</given-names>
</name>
<name>
<surname>Parmigiani</surname> <given-names>S</given-names>
</name>
<name>
<surname>Favaro</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>A comparative study between state-of-the-art MRI deidentification and AnonyMI, a new method combining re-identification risk reduction and geometrical preservation</article-title>. (Hoboken, USA: John Wiley &amp; Sons, Inc.) (<year>2021</year>) <volume>42</volume>
<issue>(17)</issue>
<fpage>:5523&#x2013;34</fpage>. doi: <pub-id pub-id-type="doi">10.1101/2021.07.30.454335</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cardenas</surname> <given-names>CE</given-names>
</name>
<name>
<surname>Mohamed</surname> <given-names>ASR</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Gooding</surname> <given-names>M</given-names>
</name>
<name>
<surname>Veeraraghavan</surname> <given-names>H</given-names>
</name>
<name>
<surname>Kalpathy-Cramer</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Head and neck cancer patient images for determining auto-segmentation accuracy in T2-weighted magnetic resonance imaging through expert manual segmentations</article-title>. <source>Med Phys</source> (<year>2020</year>) <volume>47</volume>:<page-range>2317&#x2013;22</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/mp.13942</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alfaro-Almagro</surname> <given-names>F</given-names>
</name>
<name>
<surname>Jenkinson</surname> <given-names>M</given-names>
</name>
<name>
<surname>Bangerter</surname> <given-names>NK</given-names>
</name>
<name>
<surname>Andersson</surname> <given-names>JL</given-names>
</name>
<name>
<surname>Griffanti</surname> <given-names>L</given-names>
</name>
<name>
<surname>Douaud</surname> <given-names>G</given-names>
</name>
<etal/>
</person-group>. <article-title>Image processing and quality control for the first 10,000 brain imaging datasets from UK biobank</article-title>. <source>Neuroimage</source> (<year>2018</year>) <volume>166</volume>:<page-range>400&#x2013;24</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.neuroimage.2017.10.034</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Milchenko</surname> <given-names>M</given-names>
</name>
<name>
<surname>Marcus</surname> <given-names>D</given-names>
</name>
</person-group>. <article-title>Obscuring surface anatomy in volumetric imaging data</article-title>. <source>Neuroinformatics</source> (<year>2013</year>) <volume>11</volume>:<fpage>65</fpage>&#x2013;<lpage>75</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s12021-012-9160-3</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Gulban</surname> <given-names>O</given-names>
</name>
<name>
<surname>Nielson</surname> <given-names>D</given-names>
</name>
<name>
<surname>Poldrack</surname> <given-names>R</given-names>
</name>
<name>
<surname>Gorgolewski</surname> <given-names>C</given-names>
</name>
</person-group>. <source>Poldracklab/pydeface: v2.0.0</source> . Available at: <uri xlink:href="https://github.com/poldracklab/pydeface">https://github.com/poldracklab/pydeface</uri>.</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jeong</surname> <given-names>YU</given-names>
</name>
<name>
<surname>Yoo</surname> <given-names>S</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>Y-H</given-names>
</name>
<name>
<surname>Shim</surname> <given-names>WH</given-names>
</name>
</person-group>. <article-title>De-identification of facial features in magnetic resonance images: Software development using deep learning technology</article-title>. <source>J Med Internet Res</source> (<year>2020</year>) <volume>22</volume>:<elocation-id>e22739</elocation-id>. doi: <pub-id pub-id-type="doi">10.2196/22739</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khazane</surname> <given-names>A</given-names>
</name>
<name>
<surname>Hoachuck</surname> <given-names>J</given-names>
</name>
<name>
<surname>Gorgolewski</surname> <given-names>KJ</given-names>
</name>
<name>
<surname>Poldrack</surname> <given-names>RA</given-names>
</name>
</person-group>. <article-title>DeepDefacer: Automatic removal of facial features <italic>via</italic> U-net image segmentation</article-title>. <source>arXiv</source> (<year>2022</year>) arXiv:2205.15536. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2205.15536</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wahid</surname> <given-names>KA</given-names>
</name>
<name>
<surname>Ahmed</surname> <given-names>S</given-names>
</name>
<name>
<surname>He</surname> <given-names>R</given-names>
</name>
<name>
<surname>van Dijk</surname> <given-names>LV</given-names>
</name>
<name>
<surname>Teuwen</surname> <given-names>J</given-names>
</name>
<name>
<surname>McDonald</surname> <given-names>BA</given-names>
</name>
<etal/>
</person-group>. <article-title>Evaluation of deep learning-based multiparametric MRI oropharyngeal primary tumor auto-segmentation and investigation of input channel effects: Results from a prospective imaging registry</article-title>. <source>Clin Transl Radiat Oncol</source> (<year>2022</year>) <volume>32</volume>:<fpage>6</fpage>&#x2013;<lpage>14</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ctro.2021.10.003</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McDonald</surname> <given-names>BA</given-names>
</name>
<name>
<surname>Cardenas</surname> <given-names>C</given-names>
</name>
<name>
<surname>O&#x2019;Connell</surname> <given-names>N</given-names>
</name>
<name>
<surname>Ahmed</surname> <given-names>S</given-names>
</name>
<name>
<surname>Naser</surname> <given-names>MA</given-names>
</name>
<name>
<surname>Wahid</surname> <given-names>KA</given-names>
</name>
<etal/>
</person-group>. <article-title>Investigation of autosegmentation techniques on T2-weighted MRI for off-line dose reconstruction in MR-linac adapt to position workflow for head and neck cancers</article-title>. <source>medRxiv</source> (<year>2021</year>). doi: <pub-id pub-id-type="doi">10.1101/2021.09.30.21264327</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Taku</surname> <given-names>N</given-names>
</name>
<name>
<surname>Wahid</surname> <given-names>KA</given-names>
</name>
<name>
<surname>van Dijk</surname> <given-names>LV</given-names>
</name>
<name>
<surname>Sahlsten</surname> <given-names>J</given-names>
</name>
<name>
<surname>Jaskari</surname> <given-names>J</given-names>
</name>
<name>
<surname>Kaski</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>Auto-detection and segmentation of involved lymph nodes in HPV-associated oropharyngeal cancer using a convolutional deep learning neural network</article-title>. <source>Clin Transl Radiat Oncol</source> (<year>2022</year>) <volume>36</volume>:<fpage>47</fpage>&#x2013;<lpage>55</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ctro.2022.06.007</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Naser</surname> <given-names>MA</given-names>
</name>
<name>
<surname>van Dijk</surname> <given-names>LV</given-names>
</name>
<name>
<surname>He</surname> <given-names>R</given-names>
</name>
<name>
<surname>Wahid</surname> <given-names>KA</given-names>
</name>
<name>
<surname>Fuller</surname> <given-names>CD</given-names>
</name>
</person-group>. <source>Tumor segmentation in patients with head and neck cancers using deep learning based-on multi-modality PET/CT images</source>. <publisher-name>Springer</publisher-name> (<year>2020</year>) p. <fpage>85</fpage>&#x2013;<lpage>98</lpage>.</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Naser</surname> <given-names>MA</given-names>
</name>
<name>
<surname>van Dijk</surname> <given-names>LV</given-names>
</name>
<name>
<surname>He</surname> <given-names>R</given-names>
</name>
<name>
<surname>Wahid</surname> <given-names>KA</given-names>
</name>
<name>
<surname>Fuller</surname> <given-names>CD.</given-names>
</name>
</person-group> <article-title>Tumor segmentation in patients with head and neck cancers using deep learning based-on multi-modality PET/CT images</article-title>. <source>In Head and Neck Tumor Segmentation: First Challenge, HECKTOR 2020, Held in Conjunction with MICCAI 2020, Proceedings 1 2021</source>; <publisher-loc>Lima, Peru</publisher-loc>: <publisher-name>Springer International Publishing (2020)</publisher-name> p. <page-range>85&#x2013;98</page-range>.</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Naser</surname> <given-names>MA</given-names>
</name>
<name>
<surname>Wahid</surname> <given-names>KA</given-names>
</name>
<name>
<surname>Grossberg</surname> <given-names>AJ</given-names>
</name>
<name>
<surname>Olson</surname> <given-names>B</given-names>
</name>
<name>
<surname>Jain</surname> <given-names>R</given-names>
</name>
<name>
<surname>El-Habashy</surname> <given-names>D</given-names>
</name>
<etal/>
</person-group>. <article-title>Deep learning auto-segmentation of cervical skeletal muscle for sarcopenia analysis in patients with head and neck cancer</article-title>. <source>Front Oncol</source> (<year>2022</year>) <volume>12</volume>. doi: <pub-id pub-id-type="doi">10.3389/fonc.2022.930432</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Rossum</surname> <given-names>G</given-names>
</name>
<name>
<surname>Drake</surname> <given-names>FL</given-names>
<suffix>Jr.</suffix>
</name>
</person-group> <article-title>Python Reference manual</article-title>. <source>Centrum voor Wiskunde en Informatica Amsterdam</source> (<year>1995</year>).</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paszke</surname> <given-names>A</given-names>
</name>
<name>
<surname>Gross</surname> <given-names>S</given-names>
</name>
<name>
<surname>Massa</surname> <given-names>F</given-names>
</name>
<name>
<surname>Lerer</surname> <given-names>A</given-names>
</name>
<name>
<surname>Bradbury</surname> <given-names>J</given-names>
</name>
<name>
<surname>Chanan</surname> <given-names>G</given-names>
</name>
<etal/>
</person-group>. <article-title>Pytorch: An imperative style, high-performance deep learning library</article-title>. <source>Adv Neural Inf Process Syst</source> (<year>2019</year>) <volume>32</volume>:<page-range>8026&#x2013;37</page-range>.</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="book">
<source>The MONAI consortium</source>. <publisher-name>Project MONAI</publisher-name> (<year>2020</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.5281/zenodo.4323059</pub-id>
</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>P&#xe9;rez-Garc&#xed;a</surname> <given-names>F</given-names>
</name>
<name>
<surname>Sparks</surname> <given-names>R</given-names>
</name>
<name>
<surname>Ourselin</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>TorchIO: a Python library for efficient loading, preprocessing, augmentation and patch-based sampling of medical images in deep learning</article-title>. <source>Comput Methods Programs BioMed</source> (<year>2021</year>) <volume>208</volume>:<fpage>106236</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cmpb.2021.106236</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Taha</surname> <given-names>AA</given-names>
</name>
<name>
<surname>Hanbury</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Metrics for evaluating 3D medical image segmentation: Analysis, selection, and tool</article-title>. <source>BMC Med Imaging</source> (<year>2015</year>) <volume>15</volume>:<fpage>1</fpage>&#x2013;<lpage>28</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s12880-015-0068-x</pub-id>
</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sherer</surname> <given-names>MV</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>D</given-names>
</name>
<name>
<surname>Elguindi</surname> <given-names>S</given-names>
</name>
<name>
<surname>Duke</surname> <given-names>S</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>L-T</given-names>
</name>
<name>
<surname>Cacicedo</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Metrics to evaluate the performance of auto-segmentation for radiation treatment planning: A critical review</article-title>. <source>Radiother Oncol</source> (<year>2021</year>) <volume>160</volume>:<fpage>185&#x2013;91</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.radonc.2021.05.003</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wilcoxon</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>Individual comparisons by ranking methods</article-title>. In: <source>Breakthroughs in statistics</source>. <publisher-name>(New York, USA: Springer)</publisher-name> (<year>1992</year>). p. <fpage>196</fpage>&#x2013;<lpage>202</lpage>.</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Benjamini</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Hochberg</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>Controlling the false discovery rate: a practical and powerful approach to multiple testing</article-title>. <source>J R Stat Soc Ser B Methodol</source> (<year>1995</year>) <volume>57</volume>:<fpage>289</fpage>&#x2013;<lpage>300</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.2517-6161.1995.tb02031.x</pub-id>
</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Muschelli</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Recommendations for processing head CT data</article-title>. <source>Front Neuroinformatics</source> (<year>2019</year>) <volume>13</volume>:<elocation-id>61</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fninf.2019.00061</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>D</given-names>
</name>
<name>
<surname>Lapen</surname> <given-names>K</given-names>
</name>
<name>
<surname>Sherer</surname> <given-names>MV</given-names>
</name>
<name>
<surname>Kantor</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Boyce</surname> <given-names>LM</given-names>
</name>
<etal/>
</person-group>. <article-title>A systematic review of contouring guidelines in radiation oncology: Analysis of frequency, methodology, and delivery of consensus recommendations</article-title>. <source>Int J Radiat Oncol Biol Phys</source> (<year>2020</year>) <volume>107</volume>:<page-range>827&#x2013;35</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.ijrobp.2020.04.011</pub-id>
</citation>
</ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abrams</surname> <given-names>RA</given-names>
</name>
<name>
<surname>Winter</surname> <given-names>KA</given-names>
</name>
<name>
<surname>Regine</surname> <given-names>WF</given-names>
</name>
<name>
<surname>Safran</surname> <given-names>H</given-names>
</name>
<name>
<surname>Hoffman</surname> <given-names>JP</given-names>
</name>
<name>
<surname>Lustig</surname> <given-names>R</given-names>
</name>
<etal/>
</person-group>. <article-title>Failure to adhere to protocol specified radiation therapy guidelines was associated with decreased survival in RTOG 9704&#x2013;a phase III trial of adjuvant chemotherapy and chemoradiotherapy for patients with resected adenocarcinoma of the pancreas</article-title>. <source>Int J Radiat Oncol Biol Phys</source> (<year>2012</year>) <volume>82</volume>:<page-range>809&#x2013;16</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.ijrobp.2010.11.039</pub-id>
</citation>
</ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peters</surname> <given-names>LJ</given-names>
</name>
<name>
<surname>O&#x2019;Sullivan</surname> <given-names>B</given-names>
</name>
<name>
<surname>Giralt</surname> <given-names>J</given-names>
</name>
<name>
<surname>Fitzgerald</surname> <given-names>TJ</given-names>
</name>
<name>
<surname>Trotti</surname> <given-names>A</given-names>
</name>
<name>
<surname>Bernier</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Critical impact of radiotherapy protocol compliance and quality in the treatment of advanced head and neck cancer: Results from TROG 02.02</article-title>. <source>J Clin Oncol</source> (<year>2010</year>) <volume>28</volume>:<fpage>2996</fpage>&#x2013;<lpage>3001</lpage>. doi: <pub-id pub-id-type="doi">10.1200/JCO.2009.27.4498</pub-id>
</citation>
</ref>
<ref id="B46">
<label>46</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ohri</surname> <given-names>N</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>X</given-names>
</name>
<name>
<surname>Dicker</surname> <given-names>AP</given-names>
</name>
<name>
<surname>Doyle</surname> <given-names>LA</given-names>
</name>
<name>
<surname>Harrison</surname> <given-names>AS</given-names>
</name>
<name>
<surname>Showalter</surname> <given-names>TN</given-names>
</name>
</person-group>. <article-title>Radiotherapy protocol deviations and clinical outcomes: a meta-analysis of cooperative group clinical trials</article-title>. <source>J Natl Cancer Inst</source> (<year>2013</year>) <volume>105</volume>:<page-range>387&#x2013;93</page-range>. doi: <pub-id pub-id-type="doi">10.1093/jnci/djt001</pub-id>
</citation>
</ref>
<ref id="B47">
<label>47</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Salzillo</surname> <given-names>TC</given-names>
</name>
<name>
<surname>Taku</surname> <given-names>N</given-names>
</name>
<name>
<surname>Wahid</surname> <given-names>KA</given-names>
</name>
<name>
<surname>McDonald</surname> <given-names>BA</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
<name>
<surname>van Dijk</surname> <given-names>LV</given-names>
</name>
<etal/>
</person-group>. <article-title>Advances in imaging for HPV-related oropharyngeal cancer: Applications to radiation oncology</article-title>. In: <source>Seminars in radiation oncology</source>. <publisher-name>Elsevier</publisher-name> (<year>2021</year>). p. <page-range>371&#x2013;88</page-range>.</citation>
</ref>
<ref id="B48">
<label>48</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaissis</surname> <given-names>G</given-names>
</name>
<name>
<surname>Ziller</surname> <given-names>A</given-names>
</name>
<name>
<surname>Passerat-Palmbach</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ryffel</surname> <given-names>T</given-names>
</name>
<name>
<surname>Usynin</surname> <given-names>D</given-names>
</name>
<name>
<surname>Trask</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>End-to-end privacy preserving deep learning on multi-institutional medical imaging</article-title>. <source>Nat Mach Intell</source> (<year>2021</year>) <volume>3</volume>:<page-range>473&#x2013;84</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s42256-021-00337-8</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>