<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="article-commentary">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Public Health</journal-id>
<journal-title>Frontiers in Public Health</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Public Health</abbrev-journal-title>
<issn pub-type="epub">2296-2565</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpubh.2021.755837</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Public Health</subject>
<subj-group>
<subject>General Commentary</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Commentary: Data Processing Thresholds for Abundance and Sparsity and Missed Biological Insights in an Untargeted Chemical Analysis of Blood Specimens for Exposomics</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Keski-Rahkonen</surname> <given-names>Pekka</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1436162/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Robinson</surname> <given-names>Oliver</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/649435/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Alfano</surname> <given-names>Rossella</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/685790/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Plusquin</surname> <given-names>Michelle</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/609924/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Scalbert</surname> <given-names>Augustin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Nutrition and Metabolism Branch, International Agency for Research on Cancer (IARC/WHO)</institution>, <addr-line>Lyon</addr-line>, <country>France</country></aff>
<aff id="aff2"><sup>2</sup><institution>Medical Research Council Centre for Environment and Health, School of Public Health, Imperial College London</institution>, <addr-line>London</addr-line>, <country>United Kingdom</country></aff>
<aff id="aff3"><sup>3</sup><institution>Centre for Environmental Sciences, Hasselt University</institution>, <addr-line>Hasselt</addr-line>, <country>Belgium</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Alexandros Siskos, Imperial College London, United Kingdom</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Julian Avila, Broad Institute, United States</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Pekka Keski-Rahkonen <email>keskip&#x00040;iarc.fr</email></corresp>
<fn fn-type="other" id="fn001"><p>This article was submitted to Environmental health and Exposome, a section of the journal Frontiers in Public Health</p></fn></author-notes>
<pub-date pub-type="epub">
<day>17</day>
<month>01</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>9</volume>
<elocation-id>755837</elocation-id>
<history>
<date date-type="received">
<day>09</day>
<month>08</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>06</day>
<month>12</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2022 Keski-Rahkonen, Robinson, Alfano, Plusquin and Scalbert.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Keski-Rahkonen, Robinson, Alfano, Plusquin and Scalbert</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license> </permissions>
<related-article id="RA1" related-article-type="commentary-article" journal-id="Front. Public Health" journal-id-type="nlm-ta" vol="9" page="653599" xlink:href="34178917" ext-link-type="pubmed">A Commentary on <article-title>Data Processing Thresholds for Abundance and Sparsity and Missed Biological Insights in an Untargeted Chemical Analysis of Blood Specimens for Exposomics</article-title> by Barupal, D. K., Baygi, S. F., Wright, R. O., and Arora, M. (2021). Front. Public Health 9:653599. doi: <object-id>10.3389/fpubh.2021.653599</object-id></related-article>
<kwd-group>
<kwd>metabolomics</kwd>
<kwd>pre-processing</kwd>
<kwd>data analysis</kwd>
<kwd>exposome</kwd>
<kwd>exposomics</kwd>
</kwd-group>
<counts>
<fig-count count="1"/>
<table-count count="0"/>
<equation-count count="0"/>
<ref-count count="10"/>
<page-count count="3"/>
<word-count count="2084"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>We read with interest the paper by Barupal et al. on the effect of untargeted metabolomics data filtering thresholds that was recently published in (<xref ref-type="bibr" rid="B1">1</xref>). The authors used publicly available liquid chromatography-mass spectrometry data of 499 newborn cord blood samples. This data was generated by us in December 2015, and later published as part of our studies on the association of cord blood metabolome and birth weight (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B3">3</xref>) and postnatal growth trajectories (<xref ref-type="bibr" rid="B4">4</xref>). Barupal et al. were critical of our decision to exclude sporadic, low-abundance information from the dataset before statistical analysis, suspecting we might have lost biologically relevant information. To study this, they pre-processed the data using their own methodology, imputed missing values and computed correlations between chromatographic peak height and birth weight for the features detected. They then assessed the effect of the filtering thresholds we had used, finding this to result in the loss of many features they found associated with birth weight, some of which they propose were linked to C19-steroid and acylcarnitine metabolism. Their conclusion was that we had missed these metabolites and thus insights into these pathways, supporting their view of using data processing thresholds for peak height and detection frequencies at minimal possible levels or entirely avoiding them.</p>
<p>While welcoming the idea of lowering filtering thresholds to allow deeper mining of the metabolomics data for exposome research, we found errors in the paper&#x00027;s interpretation of our work that we wish to correct. We would also like to further discuss the benefits and challenges associated with untargeted metabolomics data filtering.</p>
</sec>
<sec sec-type="discussion" id="s2">
<title>Discussion</title>
<p>Untargeted metabolomics relies on automatic algorithms to find chromatographic features in the mass spectrometric data. Several software tools exist, and while they share the same overall aim, there are marked differences in their output (<xref ref-type="bibr" rid="B5">5</xref>, <xref ref-type="bibr" rid="B6">6</xref>). Methods for abundance measurement vary, and there are differences in detection frequency and in the amount of noise produced, especially for features at low abundance levels (<xref ref-type="bibr" rid="B7">7</xref>), so that filtering thresholds for these qualities are not directly transferrable. However, there were considerable methodological differences between our original work (<xref ref-type="bibr" rid="B2">2</xref>) and the study of Barupal et al. that we believe have led to errors in their interpretation of our results. Firstly, the pre-processing software was not the same, and different parameters for feature finding and intensity measurement were used. Secondly, methods for missing value imputation, statistical models used, and the number of features included in the analysis were different.</p>
<p>Barupal et al. highlighted two features they claim we missed due to the filtering applied: &#x0201C;<italic>m/z</italic> 412.3035 at 5.75 min&#x0201D; (speculative hydroxy-acyl carnitine) and &#x0201C;<italic>m/z</italic> 289.2162 at 4.83 min&#x0201D; (speculative testosterone). These features were shown to not reach the chosen threshold (chromatographic peak height &#x0003E;10,000 in at least 2% of the samples). However, in contrast to what the Barupal et al. paper claims, both features passed the filtering in our original study, and can be found in the published dataset (<xref ref-type="bibr" rid="B3">3</xref>) (available from MetaboLights). The disagreement seems to be related to differences in data pre-processing. Barupal et al. used MS-DIAL, and the highest peak in the dataset was reportedly 12,392,001, whereas in our study, based on Agilent MassHunter, the highest peak was 15,115,12. A similar relative difference was seen for the maximum peak heights of &#x0201C;<italic>m/z</italic> 289.2162 at 4.83 min&#x0201D; and &#x0201C;<italic>m/z</italic> 412.3035 at 5.75 min,&#x0201D; which in the Barupal et al. paper were 11,937 and 11,160, respectively, but 15,661 and 15,801, respectively, in our dataset.</p>
<p>Thus, we did not miss &#x0201C;<italic>m/z</italic> 412.3035 at 5.75 min,&#x0201D; which we also found associated with birthweight and identified as 3-hydroxyhexadecadienoylcarnitine (acylcarnitine C16:1) (<xref ref-type="bibr" rid="B2">2</xref>). We also detected &#x0201C;<italic>m/z</italic> 289.2162 at 4.83 min,&#x0201D; but in contrast to the unadjusted analysis of Barupal et al. it was not associated with birthweight in our model adjusted for gestational age, cohort, sex of the child, maternal height, maternal weight, and paternal height after multiple testing correction, so it was not discussed in our original paper (<xref ref-type="bibr" rid="B2">2</xref>). Barupal et al. suggested this feature is &#x0201C;probably testosterone,&#x0201D; but this is not correct based on the large difference in retention times when compared against testosterone reference standard (4.8 vs. 5.9 min, respectively).</p>
<p>The main conclusion of the Barupal et al. paper was that minimal or no thresholds for intensity and detection frequency should be used for metabolomics data filtering. We agree that this will minimize the loss of information. However, it will also result in a very large number of features with mostly missing values, as shown in <xref ref-type="fig" rid="F1">Figure 1</xref> that presents the discussed dataset prior to any detection frequency-based filtering. A missing value can be due to undetectably low or non-existent signal, but also related to the algorithm&#x00027;s inability to recognize a feature. This makes it difficult to find a universally applicable imputation strategy (<xref ref-type="bibr" rid="B8">8</xref>). Moreover, sensitivity of the feature finding methods leads to the presence of noise in the data, especially at low intensity levels (<xref ref-type="bibr" rid="B7">7</xref>). Noise and infrequent features are commonly filtered out in studies such as our original work for two main reasons: (1) analysis of extensively imputed data may lead to compromised inferences, and (2) high number variables increases the penalization of <italic>p</italic>-values, and therefore reduce statistical power. In our study, we intentionally filtered our data to a level we considered provided the optimal balance between metabolite detection and quality of measurements for our quantitative analysis.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Distribution of the 18,766 features with a peak height above 10,000 in our original study, before filtering on detection frequency (<xref ref-type="bibr" rid="B2">2</xref>). Detection frequency refers to the percentage of samples where the feature was detected. The histogram shows the number and relative frequency of features per each 10% class.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpubh-09-755837-g0001.tif"/>
</fig>
<p>Data filtering, especially on feature intensity, requires familiarity with the analytical instruments and methods used. In the Barupal et al. paper, much emphasis was based on the assumed dynamic range of the mass spectrometer and the signal-to-noise ratio (S/N) of the peaks. However, a method of extrapolating minimum usable abundance from a dynamic range estimate, or by using S/N, does not ensure analytical performance at the lowest levels (<xref ref-type="bibr" rid="B9">9</xref>). For example, the US EPA specifies a statistical approach to detection limits for environmental pollutants, including repeatability of the measurement rather than abundance or S/N alone (<xref ref-type="bibr" rid="B10">10</xref>).</p>
<p>For these reasons, we cannot agree with the Barupal et al. paper&#x00027;s suggestion that our data was &#x0201C;poorly explored&#x0201D; and that we &#x0201C;may have missed many metabolic hypotheses in relation to birth weight.&#x0201D; There are different ways to analyze the same untargeted metabolomics data and we made informed decisions on the filtering thresholds that we believe best served our statistical analyses. For other purposes and statistical models, different strategies may be better suited, and we agree that in studies where the data analysis tolerates infrequently detected features or extensively imputed data, an entirely unfiltered dataset would be valuable. For instance, these methods may lend themselves to (sufficiently powered) exploratory studies, with the metabolic feature categorized as detectable or non-detectable.</p>
<p>In conclusion, while we welcome the development and application of new pre-processing and filtering methods in the metabolomics field, the application of less stringent filtering thresholds by Barupal et al. did not demonstrate additional metabolic insights over our original study. The choice of pre-processing and filtering methods should consider the study design and implications on the final statistical analysis.</p>
</sec>
<sec id="s3">
<title>Author Contributions</title>
<p>PK-R, OR, RA, and AS contributed to the conception of the commentary. PK-R wrote the first draft of the manuscript and produced the figure. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec sec-type="funding-information" id="s4">
<title>Funding</title>
<p>OR was supported by a UKRI Future Leaders Fellowship (MR/S03532X/1). RA received funding from the Bijzonder Onderzoeksfonds (BOF) Hasselt University through a Ph.D. fellowship.</p>
</sec>
<sec id="s5">
<title>Author Disclaimer</title>
<p>Where authors are identified as personnel of the International Agency for Research on Cancer/World Health Organization, the authors alone are responsible for the views expressed in this article and they do not necessarily represent the decisions, policy or views of the International Agency for Research on Cancer/World Health Organization.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest. The handling editor declared a past collaboration with one of the authors OR.</p>
</sec>
<sec sec-type="disclaimer" id="s6">
<title>Publisher&#x00027;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barupal</surname> <given-names>DK</given-names></name> <name><surname>Baygi</surname> <given-names>SF</given-names></name> <name><surname>Wright</surname> <given-names>RO</given-names></name> <name><surname>Arora</surname> <given-names>M</given-names></name></person-group>. <article-title>Data processing thresholds for abundance and sparsity and missed biological insights in an untargeted chemical analysis of blood specimens for exposomics</article-title>. <source>Front Public Health.</source> (<year>2021</year>) <volume>9</volume>:<fpage>653599</fpage>. <pub-id pub-id-type="doi">10.3389/fpubh.2021.653599</pub-id><pub-id pub-id-type="pmid">34178917</pub-id></citation></ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Robinson</surname> <given-names>O</given-names></name> <name><surname>Keski-Rahkonen</surname> <given-names>P</given-names></name> <name><surname>Chatzi</surname> <given-names>L</given-names></name> <name><surname>Kogevinas</surname> <given-names>M</given-names></name> <name><surname>Nawrot</surname> <given-names>T</given-names></name> <name><surname>Pizzi</surname> <given-names>C</given-names></name> <etal/></person-group>. <article-title>Cord blood metabolic signatures of birth weight: a population-based study</article-title>. <source>J Proteome Res.</source> (<year>2018</year>) <volume>17</volume>:<fpage>1235</fpage>&#x02013;<lpage>47</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jproteome.7b00846</pub-id><pub-id pub-id-type="pmid">29401400</pub-id></citation></ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alfano</surname> <given-names>R</given-names></name> <name><surname>Chadeau-Hyam</surname> <given-names>M</given-names></name> <name><surname>Ghantous</surname> <given-names>A</given-names></name> <name><surname>Keski-Rahkonen</surname> <given-names>P</given-names></name> <name><surname>Chatzi</surname> <given-names>L</given-names></name> <name><surname>Perez</surname> <given-names>AE</given-names></name> <etal/></person-group>. <article-title>A multi-omic analysis of birthweight in newborn cord blood reveals new underlying mechanisms related to cholesterol metabolism</article-title>. <source>Metabolism.</source> (<year>2020</year>) <volume>110</volume>:<fpage>154292</fpage>. <pub-id pub-id-type="doi">10.1016/j.metabol.2020.154292</pub-id><pub-id pub-id-type="pmid">32553738</pub-id></citation></ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Handakas</surname> <given-names>E</given-names></name> <name><surname>Keski-Rahkonen</surname> <given-names>P</given-names></name> <name><surname>Chatzi</surname> <given-names>L</given-names></name> <name><surname>Alfano</surname> <given-names>R</given-names></name> <name><surname>Roumeliotaki</surname> <given-names>T</given-names></name> <name><surname>Plusquin</surname> <given-names>M</given-names></name> <etal/></person-group>. <article-title>Cord blood metabolic signatures predictive of childhood overweight and rapid growth</article-title>. <source>Int J Obes</source>. (<year>2021</year>) <volume>45</volume>:<fpage>2252</fpage>&#x02013;<lpage>60</lpage>. <pub-id pub-id-type="doi">10.1038/s41366-021-00888-1</pub-id><pub-id pub-id-type="pmid">34253844</pub-id></citation></ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Z</given-names></name> <name><surname>Lu</surname> <given-names>Y</given-names></name> <name><surname>Guo</surname> <given-names>Y</given-names></name> <name><surname>Cao</surname> <given-names>H</given-names></name> <name><surname>Wang</surname> <given-names>Q</given-names></name> <name><surname>Shui</surname> <given-names>W</given-names></name></person-group>. <article-title>Comprehensive evaluation of untargeted metabolomics data processing software in feature detection, quantification and discriminating marker selection</article-title>. <source>Anal Chim Acta.</source> (<year>2018</year>) <volume>1029</volume>:<fpage>50</fpage>&#x02013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1016/j.aca.2018.05.001</pub-id><pub-id pub-id-type="pmid">30442403</pub-id></citation></ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hohrenk</surname> <given-names>LL</given-names></name> <name><surname>Itzel</surname> <given-names>F</given-names></name> <name><surname>Baetz</surname> <given-names>N</given-names></name> <name><surname>Tuerk</surname> <given-names>J</given-names></name> <name><surname>Vosough</surname> <given-names>M</given-names></name> <name><surname>Schmidt</surname> <given-names>TC</given-names></name></person-group>. <article-title>Comparison of software tools for liquid chromatography&#x02013;high-resolution mass spectrometry data processing in nontarget screening of environmental samples</article-title>. <source>Anal Chem.</source> (<year>2020</year>) <volume>92</volume>:<fpage>1898</fpage>&#x02013;<lpage>907</lpage>. <pub-id pub-id-type="doi">10.1021/acs.analchem.9b04095</pub-id><pub-id pub-id-type="pmid">31840499</pub-id></citation></ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chetnik</surname> <given-names>K</given-names></name> <name><surname>Petrick</surname> <given-names>L</given-names></name> <name><surname>Pandey</surname> <given-names>G</given-names></name></person-group>. <article-title>MetaClean: a machine learning-based classifier for reduced false positive peak detection in untargeted LC&#x02013;MS metabolomics data</article-title>. <source>Metabolomics.</source> (<year>2020</year>) <volume>16</volume>:<fpage>117</fpage>. <pub-id pub-id-type="doi">10.1007/s11306-020-01738-3</pub-id><pub-id pub-id-type="pmid">33085002</pub-id></citation></ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Do</surname> <given-names>KT</given-names></name> <name><surname>Wahl</surname> <given-names>S</given-names></name> <name><surname>Raffler</surname> <given-names>J</given-names></name> <name><surname>Molnos</surname> <given-names>S</given-names></name> <name><surname>Laimighofer</surname> <given-names>M</given-names></name> <name><surname>Adamski</surname> <given-names>J</given-names></name> <etal/></person-group>. <article-title>Characterization of missing values in untargeted MS-based metabolomics data and evaluation of missing data handling strategies</article-title>. <source>Metabolomics.</source> (<year>2018</year>) <volume>14</volume>:<fpage>128</fpage>. <pub-id pub-id-type="doi">10.1007/s11306-018-1420-2</pub-id><pub-id pub-id-type="pmid">30830398</pub-id></citation></ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="web"><person-group person-group-type="author"><collab><italic>Agilent Technologies Technical Overview Publication 5990-8341EN</italic></collab></person-group>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.agilent.com/cs/library/technicaloverviews/public/5990-8341EN.pdf">https://www.agilent.com/cs/library/technicaloverviews/public/5990-8341EN.pdf</ext-link> (accessed August 30, 2021).</citation>
</ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal"><person-group person-group-type="author"><collab>U.S. EPA</collab></person-group>. <source>Title 40: Protection of Environment; Part 136 &#x02013;Guidelines Establishing Test Procedures for the Analysis of Pollutants; Appendix B to Part 136 &#x02013; Definition and Procedure for the Determination of the Method Detection Limit &#x02013; Revision 2.</source> 82 FR 40939, August 28, 2017.</citation>
</ref>
</ref-list>
</back>
</article>