<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1244493</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2023.1244493</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>The draft genome of the microscopic <italic>Nemertoderma westbladi</italic> sheds light on the evolution of Acoelomorpha genomes</article-title>
<alt-title alt-title-type="left-running-head">Abalde et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2023.1244493">10.3389/fgene.2023.1244493</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Abalde</surname>
<given-names>Samuel</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1257300/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Tellgren-Roth</surname>
<given-names>Christian</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2409419/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Heintz</surname>
<given-names>Julia</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2417660/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Vinnere Pettersson</surname>
<given-names>Olga</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/547213/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Jondelius</surname>
<given-names>Ulf</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/317076/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Zoology</institution>, <institution>Swedish Museum of Natural History</institution>, <addr-line>Stockholm</addr-line>, <country>Sweden</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Immunology, Genetics and Pathology</institution>, <institution>SciLifeLab</institution>, <institution>Uppsala University</institution>, <addr-line>Uppsala</addr-line>, <country>Sweden</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Zoology</institution>, <institution>Stockholm University</institution>, <addr-line>Stockholm</addr-line>, <country>Sweden</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/779051/overview">Iker Irisarri</ext-link>, University of G&#xf6;ttingen, Germany</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1237867/overview">Joel Vizueta</ext-link>, University of Copenhagen, Denmark</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/182187/overview">Andreas Hejnol</ext-link>, University of Bergen, Norway</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Samuel Abalde, <email>saabalde@gmail.com</email>; Ulf Jondelius, <email>ulf.jondelius@zoologi.su.se</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>26</day>
<month>09</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1244493</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>06</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>12</day>
<month>09</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Abalde, Tellgren-Roth, Heintz, Vinnere Pettersson and Jondelius.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Abalde, Tellgren-Roth, Heintz, Vinnere Pettersson and Jondelius</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<bold>Background:</bold> Xenacoelomorpha is a marine clade of microscopic worms that is an important model system for understanding the evolution of key bilaterian novelties, such as the excretory system. Nevertheless, Xenacoelomorpha genomics has been restricted to a few species that either can be cultured in the lab or are centimetres long. Thus far, no genomes are available for Nemertodermatida, one of the group&#x2019;s main clades and whose origin has been dated more than 400 million years ago.</p>
<p>
<bold>Methods:</bold> DNA was extracted from a single specimen and sequenced with HiFi following the PacBio Ultra-Low DNA Input protocol. After genome assembly, decontamination, and annotation, the genome quality was benchmarked using two acoel genomes and one Illumina genome as reference. The gene content of three cnidarians, three acoelomorphs, four deuterostomes, and eight protostomes was clustered in orthogroups to make inferences of gene content evolution. Finally, we focused on the genes related to the ultrafiltration excretory system to compare patterns of presence/absence and gene architecture among these clades.</p>
<p>
<bold>Results:</bold> We present the first nemertodermatid genome sequenced from a single specimen of <italic>Nemertoderma westbladi</italic>. Although genome contiguity remains challenging (N50: 60&#xa0;kb), it is very complete (BUSCO: 80.2%, Metazoa; 88.6%, Eukaryota) and the quality of the annotation allows fine-detail analyses of genome evolution. Acoelomorph genomes seem to be relatively conserved in terms of the percentage of repeats, number of genes, number of exons per gene and intron size. In addition, a high fraction of genes present in both protostomes and deuterostomes are absent in Acoelomorpha. Interestingly, we show that all genes related to the excretory system are present in Xenacoelomorpha except <italic>Osr</italic>, a key element in the development of these organs and whose acquisition seems to be interconnected with the origin of the specialised excretory system.</p>
<p>
<bold>Conclusion:</bold> Overall, these analyses highlight the potential of the Ultra-Low Input DNA protocol and HiFi to generate high-quality genomes from single animals, even for relatively large genomes, making it a feasible option for sequencing challenging taxa, which will be an exciting resource for comparative genomics analyses.</p>
</abstract>
<kwd-group>
<kwd>Ultra-Low DNA Input</kwd>
<kwd>Xenacoelomorpha</kwd>
<kwd>HiFi</kwd>
<kwd>gene content</kwd>
<kwd>excretory system</kwd>
<kwd>Osr</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Evolutionary and Population Genetics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Access to a growing number of high-quality genomes from non-model animal species has helped us understand the origin of key evolutionary novelties (<xref ref-type="bibr" rid="B1">Albertin et al., 2015</xref>; <xref ref-type="bibr" rid="B20">Dunwell et al., 2017</xref>; <xref ref-type="bibr" rid="B64">Rubin et al., 2019</xref>). However, small yields of extracted DNA is a limiting factor in genome sequencing of small animals, also when using whole-body extractions. In this regard, the recent development of Ultra-Low DNA Input protocols has significantly reduced the amount of input DNA, enabling the sequencing of high-quality genomes from millimetric animals (<xref ref-type="bibr" rid="B41">Kingan et al., 2019</xref>; <xref ref-type="bibr" rid="B43">Korlach, 2020</xref>; <xref ref-type="bibr" rid="B69">Schneider et al., 2021</xref>)&#x2060;. Yet, this approach is recommended for genomes smaller than 500&#xa0;Mb (PacBio&#x2019;s library prep protocol), and it is unclear how well it performs beyond that limit, which is not a minor detail. Despite the general trend that miniaturised animals tend to have smaller genomes (<xref ref-type="bibr" rid="B48">Liu et al., 2012</xref>; <xref ref-type="bibr" rid="B29">Gross et al., 2019</xref>; <xref ref-type="bibr" rid="B79">Xu et al., 2021</xref>)&#x2060;, there are several animals, such as Xenacoelomorpha, whose genome size is comparable to that of larger animals (<xref ref-type="bibr" rid="B4">Arimoto et al., 2019</xref>; <xref ref-type="bibr" rid="B26">Gehrke et al., 2019</xref>; <xref ref-type="bibr" rid="B53">Martinez et al., 2022</xref>).</p>
<p>Xenacoelomorpha is a clade of mostly marine, microscopic worms consisting of the clades Acoela, Nemertodermatida, and their sister taxon <italic>Xenoturbella</italic>. Early molecular phylogenetic studies placed Xenacoelomorpha as the sister group of all other Bilateria (<xref ref-type="bibr" rid="B65">Ruiz-Trillo et al., 1999</xref>; <xref ref-type="bibr" rid="B36">Jondelius et al., 2002</xref>). This hypothesis received support from the simple morphology of Xenacoelomorpha, which lacks typical bilaterian structures such as excretory organs, through-gut and circulatory system (<xref ref-type="bibr" rid="B33">Haszprunar, 2016</xref>)&#x2060; and the name Nephrozoa was introduced for its sister group under this hypothesis (<xref ref-type="bibr" rid="B36">Jondelius et al., 2002</xref>)&#x2060;. The Nephrozoa hypothesis was further supported by analyses of gene content and phylogenomic inference (<xref ref-type="bibr" rid="B12">Cannon et al., 2016</xref>; <xref ref-type="bibr" rid="B37">Juravel et al., 2023</xref>)&#x2060;. However, an alternative hypothesis based on analyses of nucleotide sequence data places Xenacoelomorpha as sister group to Ambulacraria (echinoderms and hemichordates) within the deuterostomes (<xref ref-type="bibr" rid="B59">Philippe et al., 2019</xref>; <xref ref-type="bibr" rid="B38">Kapli and Telford, 2020</xref>). In either case, xenacoelomorphs offer a good opportunity for studying the origin of important animal novelties. Due to their lack of specialised excretory organs, xenacoelomorphs make a good comparison reference to better understand the evolution of this system. A recent study based on spatial transcriptomics has shown the expression in Xenacoelomorpha of several genes involved in the excretory process in other bilaterians, as well as several genes specifically related to the ultrafiltration excretory system (<italic>Nephrin</italic>, <italic>Kirrel</italic>, and <italic>ZO1</italic>; (<xref ref-type="bibr" rid="B3">Andrikou et al., 2019</xref>)), although their expression was observed throughout the body, unlike in other organisms with specialised excretory organs (<xref ref-type="bibr" rid="B24">G&#x105;siorowski et al., 2021</xref>). In addition to analysing their expression, the comparison of high-quality genomes from xenacoelomorphs, protostomes, and deuterostomes would offer a better understanding of the evolution of these genes, thanks to a more accurate assessment of gene presence/absence, the annotation of all gene copies in the genome, information about their distribution in the genomes, or comparisons of gene architecture, among other analyses. However, the set of available xenacoelomorph genomes is still limited.</p>
<p>Several xenacoelomorph species have drawn interest as model systems to study the evolution of body regeneration, the nervous system, and endosymbiosis (<xref ref-type="bibr" rid="B52">Mart&#xed;n-Dur&#xe1;n et al., 2018</xref>; <xref ref-type="bibr" rid="B3">Andrikou et al., 2019</xref>; <xref ref-type="bibr" rid="B26">Gehrke et al., 2019</xref>)&#x2060;, resulting in the generation of genomes from <italic>Xenoturbella</italic> [<italic>Xenoturbella bocki</italic>; (<xref ref-type="bibr" rid="B67">Schiffer et al., 2023</xref>)] and Acoela [<italic>Hofstenia miamia</italic> and the closely related acoel species <italic>Praesagittifera naikaiensis</italic> and <italic>Symsagittifera roscoffensis</italic>; (<xref ref-type="bibr" rid="B4">Arimoto et al., 2019</xref>; <xref ref-type="bibr" rid="B26">Gehrke et al., 2019</xref>; <xref ref-type="bibr" rid="B53">Martinez et al., 2022</xref>)]. Thus, to fully capture the diversity of Xenacoelomorpha it is necessary to generate new genomes from Nemertodermatida, the sister group of Acoela from which it diverged more than 400 MYBP (<xref ref-type="bibr" rid="B19">Dos Reis et al., 2015</xref>). This, however, is challenging due to their microscopic size. The four available xenacoelomorph genomes were sequenced from species that can be either cultured in the lab and/or are relatively big (<italic>X. bocki</italic> and <italic>H. miamia</italic> can reach four and 2&#xa0;cm in body length, respectively), but that is not the case for the vast majority of xenacoelomorphs, requiring more sophisticated methods. Despite their small size, the acoel genomes sequenced so far range between 700 and 1000&#xa0;Mb, two to three times larger than any other published genome sequenced with the Ultra-Low Input protocol (<xref ref-type="bibr" rid="B41">Kingan et al., 2019</xref>; <xref ref-type="bibr" rid="B43">Korlach, 2020</xref>; <xref ref-type="bibr" rid="B69">Schneider et al., 2021</xref>)&#x2060;, and thus represent a good opportunity for testing its performance in a challenging animal group. Here, we applied the PacBio Ultra-Low DNA Input protocol to sequence the genome of <italic>Nemertoderma westbladi</italic> from a single, microscopic worm, the first nemertodermatid and the longest genome sequenced with this protocol. We demonstrate the potential of this approach to generate relatively good-quality genomes through comparisons with other xenacoelomorphs. In addition, we explore the evolution of acoelomorph genomes, analyze the evolution of gene content in Bilateria and provide insights into the evolution of the genes related to the excretory system.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 DNA extractions, library preparation, and sequencing</title>
<p>High molecular weight DNA was extracted from single individuals of the nemertodermatid <italic>N. westbladi</italic> stored in either ethanol, RNAlater, or RNA Shield using two different methods: the salting-out protocol (<xref ref-type="bibr" rid="B30">Guimaraes, 2018</xref>) and the QIAamp Micro DNA kit (Qiagen). The Qubit dsDNA HS kit, a 2% agarose gel, and a Femto Pulse system were used to ensure the extraction met the minimum requirements for DNA yield and fragment size (the majority of gDNA over 20&#xa0;kb). These worms were isolated in dishes of seawater before their long-term preservation.</p>
<p>Library preparation and sequencing followed the PacBio Ultra-Low DNA Input protocol with small modifications. This protocol has been heavily tested at the Welcome Sanger Institute by Dr. Laumer, who kindly shared his experience and recommendations with us (Laumer, pers. comm.). Briefly, DNA was sheared to 10&#xa0;kb using Megaruptor 3 instead of Covaris g-TUBE (Covaris). After removing single-strand overhangs and repairing the fragment ends (SMRTbell Express Template Prep Kit 2.0; PacBio), DNA fragments were ligated to the amplification adapter and PCR amplified in two independent reactions (Reaction Mix 5A and 5B) for 15 cycles each (SMRTbell&#xae; gDNA Sample Amplification Kit; PacBio). Amplified DNA was purified using ProNex Beads (Promega), pooled in a single sample, damage-repaired for the second time, and ligated to the hairpin adapters. Size selection of the prepared SMRTbell library was done using a 35% dilution of AMPure PB beads (PacBio), which removed all fragments shorter than 3&#xa0;kb, instead of the BluePippin system (Sage Science). Finally, the library was sequenced in one SMRT cell on the Sequel IIe platform.</p>
</sec>
<sec id="s2-2">
<title>2.2 Data filtering, assembly, and decontamination</title>
<p>The &#x2018;Trim gDNA Amplification Adapters&#x2019; pipeline from SMRT Link v11 was used to remove the adapter sequences. Three genome assembly strategies were attempted and compared: the IPA HiFi Genome Assembler included in SMRT Link v11 (PacBio), Hifiasm v.0.7 (<xref ref-type="bibr" rid="B14">Cheng et al., 2021</xref>)&#x2060;, and Flye v.2.8.3 (<xref ref-type="bibr" rid="B42">Kolmogorov et al., 2019</xref>)&#x2060;. Default parameters were used in the three approaches, but with the &#x201c;--meta&#x201d; option activated in Flye. Based on genome length and completeness (measured with BUSCO and the metazoa odb10 database; <xref ref-type="sec" rid="s11">Supplementary Table S1</xref>), the Flye assembly was selected for downstream analyses, which included two additional scaffolding approaches. First, the two&#xa0;<italic>N. westbladi</italic> transcriptomes were mapped to the genome using HISAT2 v.2.0.5 (<xref ref-type="bibr" rid="B40">Kim et al., 2019</xref>)&#x2060; and fed to P_RNA_SCAFFOLDER (<xref ref-type="bibr" rid="B84">Zhu et al., 2018</xref>)&#x2060;. Second, the genome of <italic>S. roscoffensis</italic> was used as a reference to map the assembled genome with RagTag v.2.0.1 (<xref ref-type="bibr" rid="B2">Alonge et al., 2022</xref>)&#x2060;. Unfortunately, none of these attempts improved the genome contiguity any further. Finally, assembly redundancy was removed using the kmerDedup pipeline (<ext-link ext-link-type="uri" xlink:href="https://github.com/xiekunwhy/kmerDedup">https://github.com/xiekunwhy/kmerDedup</ext-link>). First, the Kmer spectra (k &#x3d; 21) from the trimmed reads and the assembly were extracted and then merged with Jellyfish v.2.3.0 (<xref ref-type="bibr" rid="B51">Mar&#xe7;ais and Kingsford, 2011</xref>). The list of kmers was converted to a fasta file and mapped to the assembly with Bowtie2 v.2.5.1 (<xref ref-type="bibr" rid="B45">Langmead and Salzberg, 2012</xref>), using the very-sensitive mode and the end-to-end approach, with a minimum alignment score of &#x201c;L, &#x2212;0.6, &#x2212;0.2&#x201d;. Then, kmerDedup was run with different minimum kmer coverage (mcv) and maximum duplication percentage (mpr), but the best combination according to BUSCO was to set both to 30%.</p>
<p>The raw assembly was decontaminated following the BlobTools2 pipeline (<xref ref-type="bibr" rid="B13">Challis et al., 2020</xref>)&#x2060;. Coverage data was calculated by mapping the filtered HiFi reads to the assembled genome using Minimap2 with default parameters (<xref ref-type="bibr" rid="B47">Li, 2021</xref>)&#x2060;, genome completeness inferred with BUSCO v.5.2.2 (<xref ref-type="bibr" rid="B70">Seppey et al., 2019</xref>) and the Metazoa odb10 database, and taxonomic information was identified through BLAST searches of the contigs <italic>versus</italic> the UniProt database (Release 2022_05) using diamond v.0.9.26.127 (<xref ref-type="bibr" rid="B11">Buchfink et al., 2021</xref>). Only the contigs identified as &#x201c;Metazoa&#x201d; were kept at this stage. Additionally, a BLAST search and a custom database were used to remove mitochondrial contigs. Finally, Minimap2 was used to map the reads back to the decontaminated genome to separate the nemertodermatid reads. The k-mer approaches GenomeScope v.2.0 (assuming a diploid genome and a 35x coverage) and SmudgePlot (<xref ref-type="bibr" rid="B63">Ranallo-Benavidez et al., 2020</xref>) were used to calculate the genome heterozygosity and ploidy before and after the decontamination step with a Kmer length of 21. To identify the contaminant contigs, the diamond output was used to extract the <italic>Taxid</italic> information of the hits, which is associated with a unique taxonomic category on the NCBI database.</p>
</sec>
<sec id="s2-3">
<title>2.3 Genome annotation</title>
<p>RepeatMasker v.4.1.2-p1 (<xref ref-type="bibr" rid="B73">Smit et al., 2015</xref>)&#x2060; was used to soft mask the repeats in the decontaminated genome with the rmblast engine, for which a custom repeat database was generated with RepeatModeler v.2.0.1 (<xref ref-type="bibr" rid="B74">Smit and Hubley, 2015</xref>) and the -LTRStruct option activated. Afterwards, the genome was annotated with BRAKER2 (<xref ref-type="bibr" rid="B9">Br&#x16f;na et al., 2021</xref>) using transcriptomic and proteomic evidence. The two available transcriptomes for <italic>N. westbladi</italic>, sequenced from a whole-body extraction, were downloaded and quality filtered in a two-step approach. Adapters removal and the first trimming were performed with Trimmomatic v.0.36 (as implemented in Trinity v2.6.6 with default parameters, (<xref ref-type="bibr" rid="B28">Grabherr et al., 2011</xref>)), followed by a more thorough cleaning with PRINSEQ v.0.20.3 (<xref ref-type="bibr" rid="B68">Schmieder and Edwards, 2011</xref>): trim all terminal bases with a quality below 30 and filter out reads whose mean quality is below 25, low complexity sequences (minimum entropy 50), and reads shorter than 75&#xa0;bp. Clean reads were mapped to the soft-masked genome with STAR v.2.7.9 (<xref ref-type="bibr" rid="B18">Dobin et al., 2013</xref>) and the options &#x201c;--sjdbOverhang 100 --genomeSAindexNbases 13 --genomeChrBinNbits 15&#x201d; and &#x201c;--chimSegmentMin 40 --twopassMode Basic&#x201d;. For the proteomes, the gene models from the acoel <italic>P. naikaiensis</italic> (<xref ref-type="bibr" rid="B4">Arimoto et al., 2019</xref>)&#x2060;, the BUSCO Metazoa odb10 database, and a custom set of single-copy orthogroups, inferred from published transcriptomes with OrthoFinder v.2.4.1 (<xref ref-type="bibr" rid="B22">Emms and Kelly, 2019</xref>), were concatenated and mapped to the <italic>N. westbladi</italic> genome using ProtHint v.2.6 (<xref ref-type="bibr" rid="B10">Br&#x16f;na et al., 2020</xref>). The inferred gene models were functionally annotated by pfam_scan v.1.6 (<xref ref-type="bibr" rid="B56">Mistry et al., 2007</xref>) and the PFAM 31.0 database.</p>
</sec>
<sec id="s2-4">
<title>2.4 Quality control</title>
<p>The quality of the decontaminated genome was assessed using QUAST v.5.2.0 (<xref ref-type="bibr" rid="B31">Gurevich et al., 2013</xref>) and the completeness of the genome and the annotation with BUSCO v.5.2.2 using the Metazoa and Eukaryota odb10 databases. Since all the metazoan contigs were kept during the decontamination step, two approaches were followed to ensure they belong to the nemertodermatid genome. First, a distance tree was inferred with FastMe v.2.1.5 (<xref ref-type="bibr" rid="B46">Lefort et al., 2015</xref>) based on a distance matrix calculated with Skmer (<xref ref-type="bibr" rid="B66">Sarmashghi et al., 2017</xref>), an alignment-free method designed to estimate genomic distances, over the <italic>N. westbladi</italic> genome and 18 metazoan genomes downloaded from GenBank (<xref ref-type="sec" rid="s11">Supplementary Table S2</xref>). Second, a phylogenetic tree was inferred from these genomes except for three for which the annotated proteome was not available. Briefly, orthogroups were inferred with OrthoFinder v.2.4.1 (<xref ref-type="bibr" rid="B22">Emms and Kelly, 2019</xref>) and cleaned from paralogs with PhyloPyPruner v.1.2.3 (<xref ref-type="bibr" rid="B76">Thal&#xe9;n et al., 2021</xref>) using the &#x201c;Largest Subtree&#x201d; method, collapsing nodes with bootstrap support lower than 60, and pruning branches more than five times longer than the standard deviation of all branch lengths in the tree. Then, orthogroups were aligned with MAFFT v.7.475 using the L-INS-i algorithm (<xref ref-type="bibr" rid="B39">Katoh and Standley, 2013</xref>), cleaned from poorly aligned sites with BMGE v.1.12 (<xref ref-type="bibr" rid="B16">Criscuolo and Gribaldo, 2010</xref>), tested for stationarity and homogeneity (symmetry tests) with IQ-TREE2 v.2.1.3 (<xref ref-type="bibr" rid="B55">Minh et al., 2020</xref>), and concatenated with FASconCAT v.1.05 (<xref ref-type="bibr" rid="B44">K&#xfc;ck and Longo, 2014</xref>). Finally, a phylogenetic tree was inferred using coalescence [ASTRAL; (<xref ref-type="bibr" rid="B82">Zhang et al., 2017</xref>)] and concatenation by maximum likelihood with a site-specific heterogeneous model (assuming 20 amino acid categories, C20) with IQ-TREE v.1.6.12 (<xref ref-type="bibr" rid="B58">Nguyen et al., 2015</xref>).</p>
<p>All the genome metrics, including length, contiguity, number of genes, and completeness, among others, were compared to the acoel genomes from <italic>P. naikaiensis</italic> (<xref ref-type="bibr" rid="B4">Arimoto et al., 2019</xref>)&#x2060; and <italic>S. roscoffensis</italic> (<xref ref-type="bibr" rid="B53">Martinez et al., 2022</xref>)&#x2060;, which were also tested for contaminants using BlobTools2, following the same pipeline and with the same filtering criteria. The genomes of <italic>H. miamia</italic> and <italic>X. bocki</italic> (<xref ref-type="bibr" rid="B26">Gehrke et al., 2019</xref>; <xref ref-type="bibr" rid="B67">Schiffer et al., 2023</xref>) were not considered because an annotation file with details of protein structure is not available for any of them. Additionally, a second&#xa0;<italic>N. westbladi</italic> genome sequenced in an Illumina HiSeq2500 platform was also included in the comparisons to estimate the improvement in genome quality with HiFi data relative to a short-read approach. Briefly, DNA was extracted from a pool of 12 individuals, collected in the same location at the same time, the sequencing library was prepared with a Rubicon kit, and the sequencing generated more than 385 million reads. The Illumina reads were assembled with SPAdes v.3.14.1 (<xref ref-type="bibr" rid="B6">Bankev et al., 2012</xref>), with four Kmer lengths (21, 33, 55, 75) and error correction activated. Finally, this genome was analysed with the same parameters as the HiFi genome to eliminate contamination contigs (but not redundancy), produce completeness stats, and annotate gene models.</p>
</sec>
<sec id="s2-5">
<title>2.5 Analysis of gene content</title>
<p>To analyse the evolution of gene content in Acoelomorpha, the annotated genomes of 18 metazoans were compared, including <italic>N. westbladi</italic> (Nemertodermatida) and <italic>P. naikaiensis</italic> and <italic>S. symsagittifera</italic> (Acoela) as representatives of Acoelomorpha, eight protostome genomes, four deuterostomes, and three cnidarians as the outgroup to Bilateria (<xref ref-type="sec" rid="s11">Supplementary Table S2</xref>). Redundancies in the gene models of all genomes were removed with CD-HIT (<xref ref-type="bibr" rid="B23">Fu et al., 2012</xref>), clustering all sequences more than 95% identical, and then functionally annotated with pfam_scan v.1.6 (<xref ref-type="bibr" rid="B56">Mistry et al., 2007</xref>) and the PFAM 31.0 database. The annotated proteins were clustered using OrthoFinder v.2.4.1 (<xref ref-type="bibr" rid="B22">Emms and Kelly, 2019</xref>) and used to calculate the number of genes specific to or shared among the four main clades of interest: Cnidaria, Acoelomorpha, Deuterostomia, and Protostomia. Genes present in at least one cnidarian and one bilaterian were considered to be shared across Metazoa, whereas genes present in at least two of Acoelomorpha, Deuterostomia, and Protostomia were considered to be shared across Bilateria. The proportion of &#x201c;metazoan&#x201d; and &#x201c;bilaterian&#x201d; genes absent from each of the three bilaterian clades was calculated based on these two datasets.</p>
</sec>
<sec id="s2-6">
<title>2.6 Annotation and comparison of the genes related to the ultrafiltration excretory system</title>
<p>This analysis was based on the results of G&#x105;siorowski et al. (<xref ref-type="bibr" rid="B24">G&#x105;siorowski et al., 2021</xref>), who used spatial transcriptomics to identify the genes involved in the development of the ultrafiltration excretory system in several protostomes and one hemichordate species. All the protein sequences annotated in that study were downloaded from GenBank except <italic>Hunchback</italic>, as they found no evidence of this gene being involved in nephridiogenesis, for a total of three structural proteins: <italic>Nephrin</italic>, <italic>Kirrel</italic>, and <italic>ZO1</italic>; and six transcription factors: <italic>Eya</italic>, <italic>Lhx1/5</italic>, <italic>Osr</italic>, <italic>POU3</italic>, <italic>Sall</italic>, and <italic>Six1</italic>. These genes were annotated in the same genomes used to analyse gene content evolution through BLAST searches with diamond v0.9.26.127 (<xref ref-type="bibr" rid="B11">Buchfink et al., 2021</xref>). The correct identification of these genes was later confirmed through phylogenetic analyses with IQ-TREE v.1.6.12 (<xref ref-type="bibr" rid="B58">Nguyen et al., 2015</xref>) and manual BLAST searches on the NCBI web server. The identification of the <italic>Lhx1/5</italic> and <italic>Six1</italic> transcription factors was not always straightforward, as they are thoroughly mixed in the phylogenetic tree with many other gene variants and sometimes different isoform names were proposed in the BLAST searches for the same sequence, and thus they represent a mixture of isoforms of the same gene. All genes were later confirmed to be correctly annotated by mapping them to the same proteins annotated in the <italic>N. westbladi</italic> transcriptome. A custom R script was written to locate the filtered genes in the GFF files and extract three metrics related to gene architecture: protein length, number of exons per protein, and average exon length per gene. Unfortunately, the GFF annotation file was not available for all these genomes, so not all of them could be included in this analysis (<xref ref-type="sec" rid="s11">Supplementary Table S2</xref>). To ameliorate the misleading effect of highly fragmented genes we filtered out all proteins shorter than half of the average protein length of the respective gene (a total of 10 proteins). To test if the observed differences in the three gene metrics were statistically significant, the Shapiro-Wilk&#x2019;s method and the Barlett test were used to check if they follow a normal distribution and the homogeneity of their variances, respectively. For each gene, the differences among clades were tested with either an ANOVA or a Kruskal-Wallis test, depending on the result of the normality and homoscedasticity tests. Finally, the Bonferroni correction (ANOVA) and the Dunn test (Kruskal-Wallis) were selected to run pairwise comparisons in all cases identified as statistically different. In all cases, a <italic>p</italic>-value of 0.05 was set as the significance threshold.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 The <italic>Nemertoderma westbladi</italic> genome</title>
<p>The best extraction was produced from a sample stored in RNAlater using the QIAamp Micro kit, obtaining a fragment size over 20&#xa0;kb and ca. 20&#xa0;ng of total DNA, which yielded 990&#xa0;ng after DNA shearing and whole-genome amplification. About half of this DNA was selected for sequencing. A total of 2,313,071 reads were produced during HiFi sequencing, later reduced to 2,297,478 after quality filtering with an average length of 6.6&#xa0;kb.</p>
<p>Flye produced the best assembly (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref>), which was 678.9&#xa0;Mb long and contained 26,880 contigs, and later reduced to 458&#xa0;Mb and 11,625 contigs when assembly redundancies were removed (<xref ref-type="fig" rid="F1">Figure 1A</xref>). The longest contig was 2&#xa0;Mb long, with an N50 of 57.1&#xa0;kb and contained 82.6% of the BUSCO Metazoa odb10. The assembly contained two repeats of 507 and 531 bp with 70,000 and 79,000 copies, respectively, corresponding to 11% of the assembled (raw) genome. BlobTools2 revealed the presence of many contaminants, with only 78.9% of the contigs identified as metazoan (<xref ref-type="sec" rid="s11">Supplementary Table S3</xref>). Thus, the decontaminated assembly was only 407.7 Mb, split into 9,167 contigs with an N50 of 60&#xa0;kb (<xref ref-type="fig" rid="F1">Figure 1A</xref>; <xref ref-type="table" rid="T1">Table 1</xref>), but 80.2% of the Metazoa and 88.6% of the Eukaryota BUSCO genes were still present (<xref ref-type="sec" rid="s11">Supplementary Figure S1</xref>). The two databases reported a relatively high proportion of duplicated BUSCO genes, 7.8% and 11% in the Metazoa and Eukaryota databases, respectively (<xref ref-type="sec" rid="s11">Supplementary Figure S1</xref>). The smudgeplot was markedly different before and after the decontamination step, as the inferred ploidy went from triploid to diploid after the decontamination (<xref ref-type="sec" rid="s11">Supplementary Figure S2</xref>). The genome size estimated by GenomeScope was 237.1 Mb, with an average coverage of 39.6, and high heterozygosity (3.99%), although these numbers must be taken cautiously given the poor fit of the model (39.3%; <xref ref-type="sec" rid="s11">Supplementary Figure S3</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Summary of the statistics calculated for the two&#xa0;<italic>N. westbladi</italic> genomes (sequenced with Illumina or HiFi), <italic>P. naikaiensis</italic>, and <italic>S. roscoffensis</italic>. <bold>(A)</bold> Cumulative genome length, sorted from the longest to the shortest contig, separating the raw assembly from the BlobTools decontamination. Due to the large number of contigs in the raw assembly, only the decontaminated version of the <italic>N. westbladi</italic> genome sequenced with Illumina is shown. <bold>(B)</bold> Summary of the number of genes per contig, <bold>(C)</bold> distribution of the intron length per species, and <bold>(D)</bold> number of exons per gene.</p>
</caption>
<graphic xlink:href="fgene-14-1244493-g001.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Statistics of the four genomes analysed in this study after the decontamination step. The <italic>N. westbladi</italic> genomes are presented as &#x201c;HiFi&#x201d; and &#x201c;Illumina&#x201d; to differentiate the two sequencing approaches.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Parameter</th>
<th align="left">Illumina</th>
<th align="left">HiFi</th>
<th align="left">Pnaikaiensis</th>
<th align="left">Sroscoffensis</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Length after BlobTools (Mb)</td>
<td align="right">62.229</td>
<td align="right">407.663</td>
<td align="right">581.371</td>
<td align="right">1064.926</td>
</tr>
<tr>
<td align="left">N&#x2019;s (count)</td>
<td align="right">49,310</td>
<td align="right">12,700</td>
<td align="right">7,367,142</td>
<td align="right">1,589,933</td>
</tr>
<tr>
<td align="left">N&#x2019;s (%)</td>
<td align="right">0.079</td>
<td align="right">0.003</td>
<td align="right">1.267</td>
<td align="right">0.149</td>
</tr>
<tr>
<td align="left">Number of contigs</td>
<td align="right">26,021</td>
<td align="right">9,167</td>
<td align="right">7,104</td>
<td align="right">2,730</td>
</tr>
<tr>
<td align="left">Longest contig (Kb)</td>
<td align="right">65.353</td>
<td align="right">601.587</td>
<td align="right">702.461</td>
<td align="right">8,003.794</td>
</tr>
<tr>
<td align="left">Average contig length (Kb)</td>
<td align="right">2.391</td>
<td align="right">44.471</td>
<td align="right">81.837</td>
<td align="right">390.083</td>
</tr>
<tr>
<td align="left">N50 (Kb)</td>
<td align="right">3.996</td>
<td align="right">59.976</td>
<td align="right">129.752</td>
<td align="right">1077.644</td>
</tr>
<tr>
<td align="left">Number of gene models</td>
<td align="right">23,120</td>
<td align="right">22,578</td>
<td align="right">20,303</td>
<td align="right">28,513</td>
</tr>
<tr>
<td align="left">Fuctionally annotated proteins</td>
<td align="right">14,486</td>
<td align="right">10,074</td>
<td align="right">13,708</td>
<td align="right">17,717</td>
</tr>
<tr>
<td align="left">Max. number of genes per contig</td>
<td align="right">33</td>
<td align="right">92</td>
<td align="right">37</td>
<td align="right">280</td>
</tr>
<tr>
<td align="left">Average number of genes per contig</td>
<td align="right">0.876</td>
<td align="right">2.352</td>
<td align="right">2.858</td>
<td align="right">12.281</td>
</tr>
<tr>
<td align="left">Max. number of exons per gene</td>
<td align="right">26</td>
<td align="right">349 &#x2a;</td>
<td align="right">512</td>
<td align="right">97</td>
</tr>
<tr>
<td align="left">Average number of exons per gene</td>
<td align="right">1.531</td>
<td align="right">3.284</td>
<td align="right">6.386</td>
<td align="right">4.244</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>&#x2a;One gene with &#x3e;1200 exons, thought to be an annotation artefact, was removed for this calculation</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The decontaminated Illumina genome was also relatively complete, with 76.8% of the metazoan BUSCO genes present in the assembly (with barely any duplicates, 0.6%), but much shorter (62.2&#xa0;Mb) and much more fragmented (49,310 contigs; N50: 4&#xa0;kb) (<xref ref-type="fig" rid="F1">Figure 1A</xref>). Despite being sequenced from cultured, starved and symbiont-free populations, BlobTools also identified some contaminants in the published genomes of <italic>P. naikaiensis</italic> and <italic>S. roscoffensis</italic>. The former went from 656.1 Mb and 12,525 contigs to 581.4 Mb and 7,104 contigs, whereas the latter went from 1103&#xa0;Mb to 3,460 contigs to 1064.9 Mb and 2,730 contigs (<xref ref-type="fig" rid="F1">Figure 1A</xref>). The N50 of the two genomes rose from 127 to 130&#xa0;kb in <italic>P. naikaiensis</italic>, and from 1.04 to 1.08&#xa0;Mb in <italic>S. roscoffensis</italic>. Despite the observed differences in genome size and contiguity, the four genomes show very similar completeness results. Around 90% of the Eukaryota BUSCO genes were identified in the decontaminated genomes of all species, with <italic>P. naikaiensis</italic> presenting the fewest genes (14.9% of missing genes) (<xref ref-type="sec" rid="s11">Supplementary Figure S1A</xref>). Differences among genomes were slightly higher with the Metazoa database, with almost a 10% difference between the most (<italic>S. roscoffensis</italic>; 18.5% missing genes) and the least (<italic>P. naikaiensis</italic>; 27%) complete genomes. In <italic>N. westbladi</italic>, the HiFi genome was almost as complete as <italic>S. roscoffensis</italic> (19.8% missing genes), whereas the Illumina genome was in an intermediate position (23.1%) (<xref ref-type="sec" rid="s11">Supplementary Figure S1B</xref>).</p>
<p>The number of gene models in the four genomes ranged from 20,303 (<italic>P. naikaiensis</italic>) to 28,513 (<italic>S, roscoffensis</italic>), although the differences were reduced when only functionally annotated genes were considered: 10,074 (<italic>N. westbladi</italic>, HiFi), 13,708 (<italic>P. naikaiensis</italic>), 14,486 (<italic>N. westbladi</italic>, Illumina), and 17,717 (<italic>S. roscoffensis</italic>) (<xref ref-type="table" rid="T1">Table 1</xref>). The organisation of these genes in the genome somehow reflected the differences observed in genome contiguity. In the <italic>N. westbladi</italic> genome sequenced with Illumina, the average number of genes per contig was just 0.876, with a single gene in almost 90% of the contigs (<xref ref-type="fig" rid="F1">Figure 1B</xref>), and the contig with the highest number of genes presented 33 gene models (<xref ref-type="table" rid="T1">Table 1</xref>). In the HiFi-sequenced <italic>N. westbladi</italic> genome, up to 92 genes were found in a single contig, with an average of 2.4 genes per contig. Similarly, an average of 2.9 genes per contig were annotated in the <italic>P. naikaiensis</italic> genome, but in this case, the maximum number of genes in one contig was only 37. The <italic>S. roscoffensis</italic> genome stands out, with a maximum of 280 genes in a single contig and more than 10 genes in almost 40% of the contigs (<xref ref-type="fig" rid="F1">Figure 1B</xref>; <xref ref-type="table" rid="T1">Table 1</xref>). This trend, however, was not observed in gene architecture. The gene models in <italic>P. naikaiensis</italic>, <italic>S. roscoffensis</italic>, and the HiFi genome of <italic>N. westbladi</italic> were similar, ranging between an average of 3.2&#x2013;6.3 exons per gene, whereas almost all the genes presented a single exon in the Illumina genome (average 1.5) (<xref ref-type="fig" rid="F1">Figure 1D</xref>). One gene with over 1200 exons was annotated in the <italic>N. westbladi</italic> genome. This is likely an annotation error and it was not included in the calculation of these metrics, but it is shown in <xref ref-type="fig" rid="F1">Figure 1D</xref>. The intron size was very variable in all genomes, ranging from 6 (<italic>P. naikaiensis</italic>) to 193,733 (<italic>S. roscoffensis</italic>) bp. The intron size distribution was similar between <italic>N. westbladi</italic> and <italic>P. naikaiensis</italic>, but with generally longer introns in <italic>S. roscoffensis</italic> (<xref ref-type="fig" rid="F1">Figure 1C</xref>). Nevertheless, the intron size range was similar in the three genomes, but visibly smaller in the <italic>N. westbladi</italic> Illumina genome.</p>
<p>According to RepeatMasker, the <italic>N. westbladi</italic> genome is very repetitive, masking up to 59.66% of the genome (<xref ref-type="sec" rid="s11">Supplementary Table S4</xref>). The majority of these repeats are interspersed throughout the genome (57.84%) and more than a fifth (22.61%) were not classified into any known repeat family. Among the classified repeats, the most common ones are retroelements (31.19%), particularly the long terminal repeats (LTR, 19.32%) and long interspersed nuclear elements (LINEs, 11.53%). The Illumina genome presents a sharp contrast, with just 16.40% of the genome masked as repetitive, although LINEs (4.28%) and LTR (3.43%) are still the most abundant repeat elements (<xref ref-type="sec" rid="s11">Supplementary Table S4</xref>).</p>
</sec>
<sec id="s3-2">
<title>3.2 Identification of the contaminant contigs</title>
<p>More than half of the taxonomic groups identified within the set of contaminant contigs were bacteria, including several of the major taxonomic groups: Bacteroidetes, Tectomicrobia, Proteobacteria (including Alpha-, Beta-, Delta/Epsilon-, and Gammaproteobacteria), Planctomycetes, Actinobacteria, Cyanobacteria, and Firmicutes. None of the &#x201c;Candidate Phyla Radiation&#x201d; phyla (<xref ref-type="bibr" rid="B8">Brown et al., 2015</xref>) were identified. More specifically, there are nine genera that have been reported as statistically more abundant in the microbiome of microscopic animals than in environmental samples (<xref ref-type="bibr" rid="B7">Boscaro et al., 2022</xref>) and thus might be part of the <italic>Nemertoderma</italic> microbiome: <italic>Algoriphagus</italic>, <italic>Alteromonas</italic>, <italic>Francisella</italic>, <italic>Photobacterium</italic>, <italic>Roseobacter</italic>, <italic>Shewanella</italic>, and <italic>Vibrio</italic>. Other important sources of contamination besides bacteria are algae (Chlorophyta, Rhodophyta, and Streptophyta), and fungi (Ascomycota, Basidiomycota, Microsporidia, Mucoromycota, and Zoopagomycota). These groups accumulate 87% of the taxonomic diversity within the contaminants. In addition, we also found Protista (Amoebozoa, Perkinsozoa, Endomyxa, and Oomycota), and Virus (Uroviricota). A complete description of these results is provided in <xref ref-type="sec" rid="s11">Supplementary Table S5</xref>.</p>
</sec>
<sec id="s3-3">
<title>3.3 Gene content evolution</title>
<p>The comparison of 18 animal genomes, representing Acoelomorpha, Cnidaria, Deuterostomia, and Protostomia revealed a high degree of specificity in gene content: 17.2% of all orthogroups present in Cnidaria are exclusive to this phylum (1754 out of 10,172), 23.1% in Acoelomorpha (2,101 out of 9,080), 45.3% in Deuterostomia (7,976 out of 17,593), and 48.7% in Protostomia (9,573 out of 19,669; <xref ref-type="fig" rid="F2">Figure 2A</xref>). Hence, only 33.4% (10,736 out of 32,141) of all orthogroups were annotated in at least two of the four groups. Among these, more than half (53.4%) were present in at least one species of each clade, whereas only 3.8% were present in all bilaterian clades but Cnidaria. A total of 8,418 genes were identified as shared across Metazoa (present in Cnidaria and at least one Bilateria), and 2,318 for Bilateria (present in at least two bilaterian clades). Acoelomorpha had 71.5% of the metazoan genes and 41.5% of the bilaterian ones, contrasting with deuterostomes (91.3% and 83.1%) and protostomes (94.4% and 92.9%) (<xref ref-type="fig" rid="F2">Figure 2C</xref>). The proportion of missing BUSCO genes was below 11% in all four groups (<xref ref-type="fig" rid="F2">Figure 2B</xref>), and so genome completeness does not explain this pattern. Within Acoelomorpha, almost half (43.2%) of the genes were shared between Acoela and Nemertodermatida (<xref ref-type="fig" rid="F2">Figure 2A</xref>), 41.5% were unique to Acoela, and 15.3% to Nemertodermatida.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The gene content of the three acoelomorph genomes was compared to 15 genomes from several phyla, including three cnidarians, four deuterostomes (three chordates and one echinoderm), and eight protostomes. <bold>(A)</bold> Number of unique and shared genes among acoelomorphs, cnidarians, deuterostomes, and protostomes. In the inset, the number of shared genes between the two acoel genomes and <italic>N. westbladi</italic>. <bold>(B)</bold> BUSCO scores of each of the four main clades. <bold>(C)</bold> Percentage of missing genes observed in acoelomorphs, deuterostomes, and protostomes. The set of &#x201c;metazoan genes&#x201d; was defined as all genes shared between at least one cnidarian and one bilaterian species; whereas the &#x201c;bilaterian genes&#x201d; are those shared between at least two of the three bilaterian clades. The silhouettes in <bold>(B,C)</bold> were downloaded from PhyloPic (Nemertodermatida, Andreas Hejnol; <italic>Chrysaora</italic>, Levi Simons; Asteroidea, Fernando Carezzano; and <italic>Tricolia</italic>, Tauana Cunha).</p>
</caption>
<graphic xlink:href="fgene-14-1244493-g002.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>3.4 Ultrafiltration excretory system</title>
<p>Nine genes, selected because of their participation in the development of the ultrafiltration excretory system, were investigated: three structural proteins (<italic>Nephrin</italic>, <italic>Kirrel</italic>, and <italic>ZO1</italic>), and six transcription factors (<italic>Eya</italic>, <italic>Lhx1/5</italic>, <italic>Osr</italic>, <italic>POU3</italic>, <italic>Sall</italic>, and <italic>Six1</italic>). All of them were annotated in both protostomes and deuterostomes. In Acoelomorpha, all genes but <italic>Osr</italic> were annotated, whereas only three out of the nine genes were found in the two cnidarian species (<italic>ZO1</italic>, <italic>Six</italic>, and <italic>Lhx</italic>; <xref ref-type="fig" rid="F3">Figure 3A</xref>). According to GenBank, three more genes (<italic>Nephrin</italic>, <italic>Eya</italic>, and <italic>POU3</italic>) are also present in this phylum (<xref ref-type="fig" rid="F3">Figure 3A</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>
<bold>(A)</bold> Presence of the nine genes related to the ultrafiltration excretory system annotated in this study (blue), complemented with information from GenBank (black). The phyla investigated here are highlighted in bold, whereas the others were studied in <xref ref-type="bibr" rid="B24">G&#x105;siorowski et al. (2021)</xref>. The cladogram topology is based on (<xref ref-type="bibr" rid="B27">Giribet and Edgecombe, 2020</xref>), including the two alternative positions of Xenacoelomorpha as a dashed line. <bold>(B)</bold> Boxplot comparing the three metrics related to gene architecture, separating the four main clades analysed per colour. Only the comparisons significantly different are shown, but the full result is included in <xref ref-type="sec" rid="s11">Supplementary Figure S5</xref>. In the X-axis, below the boxplots, the brackets summarise the pairwise comparisons, clustering the clades with no significant differences within the same brackets.</p>
</caption>
<graphic xlink:href="fgene-14-1244493-g003.tif"/>
</fig>
<p>The gene architecture (in terms of protein length, number of exons per gene, and average exon length) was compared for the nine genes among four clades: Cnidaria, Acoelomorpha, Deuterostomia, and Protostomia. Almost half of the 27 comparisons returned statistically significant differences among clades, most of them related to acoelomorphs (<xref ref-type="fig" rid="F3">Figure 3B</xref>). Despite the evident variation in protein length, both within and among clades, only three out of the nine genes were considered to be statistically significant: <italic>Kirrel</italic>, which is significantly longer in acoelomorphs; <italic>ZO1</italic>, longer in deuterostomes; and <italic>Lhx</italic>, but in this case the differences were only significant between acoelomorphs (longer) and protostomes (shorter). As for the number of exons per gene, <italic>ZO1</italic> and <italic>Eya</italic> presented fewer exons in acoelomorphs than in both deuterostomes and protostomes. Finally, the last gene with a significantly different number of exons is <italic>POU3</italic>. This is a relatively short protein, on average shorter than 500 amino acids in all clades, and with very few exons: only one exon in all deuterostomes but <italic>Branchiostoma floridae</italic> (three), between one and three in protostomes, and between one and four in acoelomorphs. Only the differences between deuterostomes and acoelomorphs were statistically significant. Two remarkable outliers were found when comparing the number of exons per gene. Three chordate <italic>ZO1</italic> sequences were divided into more than 80 exons (average 29.5) and one of the <italic>POU3</italic> sequences annotated in <italic>P. naikaiensis</italic> presented 15 exons (average in Acoelomorpha: 2.6). Nonetheless, these proteins were roughly of the same size as the others and their identity to the most similar protein was above 90%.</p>
<p>In an attempt to avoid the misleading effect of errors in the annotation (partial proteins will be generally shorter and with fewer exons), the average exon length was also considered. In this case, six out of the nine proteins were significantly different among clades. The average exon length was significantly longer in acoelomorphs in three genes (<italic>Kirrel</italic>, <italic>Eya</italic>, and <italic>Lhx</italic>), and two in deuterostomes (<italic>Sall</italic> and <italic>Osr</italic>, although the latter was only present in deuterostomes and protostomes). The only instance with significantly shorter exon lengths is the protostome&#x2019;s <italic>ZO1</italic> gene. Finally, among the nine comparisons including at least one cnidarian species (three genes, three metrics) no significant differences were found but in the average exon length of <italic>Lhx</italic>, which is significantly shorter than that of acoelomorphs, as also observed in deuterostomes and protostomes.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<sec id="s4-1">
<title>4.1 Performance of the Ultra-Low DNA Input protocol for sequencing large genomes</title>
<p>The steady development of sequencing technologies is allowing the generation of genomes spanning the diversity of life, which now includes minute organisms. Indeed, thanks to the latest low and ultra-low DNA input protocols, sequencing high-quality genomes from millimetric animals is now possible (<xref ref-type="bibr" rid="B80">Yoshida et al., 2018</xref>; <xref ref-type="bibr" rid="B69">Schneider et al., 2021</xref>; <xref ref-type="bibr" rid="B49">Lord et al., 2023</xref>). In this study, we used the Pacbio Ultra-Low DNA Input protocol to sequence the genome of <italic>N. westbladi</italic>, reporting the first nemertodermatid genome, sequenced from a single microscopic worm. The estimated genome length is shorter than in any sequenced acoel, and considerably shorter than <italic>S. roscoffensis</italic> and <italic>H. miamia</italic> (<xref ref-type="bibr" rid="B26">Gehrke et al., 2019</xref>)&#x2060;. Although the <italic>P. naikaiensis</italic> genome is slightly more contiguous than <italic>N. westbladi</italic>, all the metrics compared are similar between the two genomes. In contrast, both <italic>S. roscoffensis</italic> and <italic>H. miamia</italic> were scaffolded using proximity ligation data, and hence both show much higher contiguity. Beyond the differences in contiguity, annotation metrics are comparable among <italic>N.</italic> westbladi, <italic>P. naikaiensis</italic>, and <italic>S. roscoffensis</italic>. In this case, <italic>N. westbladi</italic> is more similar to <italic>S. roscoffensis</italic> than to <italic>P. naikaiensis</italic>, which shows the lowest genome completeness and number of gene models. In particular, the analysis of gene architecture shows that the number of exons per gene and intron size is also comparable, likely meaning that the annotated proteins are complete or nearly complete, facilitating the study of gene properties, such as intron-exon structure. Likewise, all genomes are similarly repetitive: <italic>N. westbladi</italic> 59.66%; <italic>P. naikaiensis</italic> 69.8%; <italic>S. roscoffensis</italic> 61.14%; and <italic>H. miamia</italic> 53%, but this is where the difference between the short- and long-read genomes of <italic>N. westbladi</italic> strikes the most. Although they have similar completeness and number of gene models, the Illumina genome is only 62.2&#xa0;Mb long and only 16.4% repeats, which is probably explained by the difficulty to assemble repetitive areas of the genome (<xref ref-type="bibr" rid="B77">T&#xf8;rresen et al., 2019</xref>).</p>
<p>It is obvious from the comparisons above that achieving a highly contiguous genome from single-millimetre worms is still challenging. One potential explanation for this is genome size. Although the <italic>N. westbladi</italic> genome is shorter than the maximum genome size advised by PacBio, due to the contaminants the raw assembly is still almost 700&#xa0;Mb long. The Ultra-Low DNA Input protocol has insofar been tested in animals whose genome size ranges between 200 and 300&#xa0;Mb, returning significantly more contiguous genomes than that of <italic>N. westbladi</italic> (<xref ref-type="bibr" rid="B41">Kingan et al., 2019</xref>; <xref ref-type="bibr" rid="B43">Korlach, 2020</xref>; <xref ref-type="bibr" rid="B69">Schneider et al., 2021</xref>)&#x2060;. The generally lower coverage of the raw assembly of the nemertodermatid genome, due to its larger size, could have resulted in a more fragmented assembly. Yet, sequencing a second HiFi SMRT cell was not feasible due to the low DNA yield. This protocol has been successfully tested on <italic>C. elegans</italic> and freshwater flatworms, demonstrating its good performance (Laumer, pers. comm.). However, it has never been used for sequencing &#x3e;300&#xa0;Mb genomes. Thus, it would be interesting to sequence the genome from a novel species with a similar genome size to confirm low coverage, and not unexpected artefacts, explain the fragmentation observed. Alternatively, one straightforward solution to improve genome contiguity is complementing this approach with ligation data, which has shown great results both in <italic>S. roscoffensis</italic> and <italic>H. miamia</italic> (<xref ref-type="bibr" rid="B26">Gehrke et al., 2019</xref>; <xref ref-type="bibr" rid="B53">Martinez et al., 2022</xref>)&#x2060;. However, this approach would require pooling tens of individuals to obtain the required amount of DNA, which is not feasible for all animals. <italic>N. westbladi</italic> cannot be cultured in the lab and collecting worms in enough numbers is challenging. Interestingly, the <italic>P. naikaiensis</italic> genome (the most similar to <italic>N. westbladi</italic>) was sequenced from a pool of individuals in 52 SMRT Cells (<xref ref-type="bibr" rid="B4">Arimoto et al., 2019</xref>)&#x2060;, whereas the <italic>N. westbladi</italic> genome comes from a single worm and one HiFi SMRT Cell. Altogether, these results highlight the potential of combining this protocol and HiFi reads to generate good-quality genomes from single, microscopic organisms, even for relatively large genomes.</p>
<p>The BlobTools analysis identified a high degree of contamination in the raw assembly of <italic>N. westbladi</italic>, which is to be expected from a microscopic organism caught in the wild. Although <italic>N. westbladi</italic> is known to not carry internal symbionts (based on hundreds of observations), a TEM analysis revealed the presence of gram-negative bacteria throughout the epidermal cilia (<xref ref-type="bibr" rid="B50">Lundin, 1998</xref>). All bacteria identified during the decontamination are gram-negative, matching this observation. Thus far, DNA extraction was performed from a whole specimen, thus sequencing the gut microbiome, and other contaminants might have been transferred from the DNA suspended in the seawater. A common practice to limit the presence of contaminants in the organism, and that was applied to both <italic>P. naikaiensis</italic> and <italic>S. roscoffensis</italic>, but not <italic>N. westbladi</italic>, is to starve the animals before DNA extraction. Besides, the two acoel genomes were sequenced from juveniles, before they incorporate the symbiotic algae, and rinsed with filtered seawater (<xref ref-type="bibr" rid="B4">Arimoto et al., 2019</xref>; <xref ref-type="bibr" rid="B53">Martinez et al., 2022</xref>). However, as seen here this is not enough to prevent the presence of contaminants. This was particularly problematic in the case of <italic>P. naikaiensis</italic>, as almost 4% of the contigs (75&#xa0;Mb, over 10% of the genome) were identified as bacterial contigs. It is important to notice that a big fraction of the genomes did not have any hit against the Uniprot database (<italic>N. westbladi</italic> 18.1%; <italic>P. naikaiensis</italic> 8.4%; <italic>S. roscoffensis</italic> 1.9%; <xref ref-type="sec" rid="s11">Supplementary Table S3</xref>), showing the importance of sequencing underrepresented groups to improve the reference databases.</p>
</sec>
<sec id="s4-2">
<title>4.2 Evolution of Acoelomorpha genomes</title>
<p>The increasing availability of animal genomes has unveiled a remarkable diversity in genome sizes, ranging from 15.3&#xa0;Mb in the orthonectid <italic>Intoshia variabilis</italic> to the 43&#xa0;Gb of the lungfish genome (<xref ref-type="bibr" rid="B72">Slyusarev et al., 2020</xref>; <xref ref-type="bibr" rid="B54">Meyer et al., 2021</xref>)&#x2060;. It has been observed that miniaturised animals tend to have smaller genomes, which has been noted both in vertebrates and invertebrates (<xref ref-type="bibr" rid="B48">Liu et al., 2012</xref>; <xref ref-type="bibr" rid="B17">Decena-Segarra et al., 2020</xref>; <xref ref-type="bibr" rid="B79">Xu et al., 2021</xref>), but with notable exceptions to this rule, as observed in nematodes and platyhelminths (<xref ref-type="bibr" rid="B15">Consortium, 2019</xref>). Genome length in the latter ranges between 700 and 1200&#xa0;Mb, the same size range as birds, some gastropods, and many freshwater fish, among others (<xref ref-type="bibr" rid="B83">Zhang et al., 2014</xref>; <xref ref-type="bibr" rid="B57">Nam et al., 2017</xref>; <xref ref-type="bibr" rid="B81">Yuan et al., 2018</xref>). Similarly, acoelomorph genomes vary between 407 (the genome of <italic>N. westbladi</italic> is the shortest reported for Acoelomorpha) and 1059&#xa0;Mb but contrast with the chromosome-level genome of <italic>X. bocki</italic>, estimated at 110&#xa0;Mb (<xref ref-type="bibr" rid="B67">Schiffer et al., 2023</xref>). Comparisons of eukaryotic genomes proposed that variations in genome sizes and proportion of repeat elements are correlated (<xref ref-type="bibr" rid="B21">Elliott and Gregory, 2015</xref>; <xref ref-type="bibr" rid="B71">Shah et al., 2020</xref>), which might also apply within Xenacoelomorpha. Acoelomorph genomes show a much higher repeat content than the small genome of <italic>Xenoturbella</italic> (<xref ref-type="bibr" rid="B67">Schiffer et al., 2023</xref>).</p>
<p>In turn, acoelomorph genomes seem to be characterised by an important reduction of gene content. Indeed, almost 60% of the genes shared between protostomes and deuterostomes are missing in acoelomorphs, which could be explained by the morphological simplicity of these worms compared with other bilaterians, but the evolutionary interpretation depends on the phylogenetic hypothesis. Under the Xenambulacraria hypothesis, their absence must be explained by massive secondary losses. The Nephrozoa hypothesis, on the other hand, suggests that the evolution of the genes exclusively shared by deuterostomes and protostomes occurred in the stem line of Nephrozoa and no <italic>ad hoc</italic> hypotheses of gene loss are required.</p>
</sec>
<sec id="s4-3">
<title>4.3 Evolution of the genes related to the ultrafiltration excretory system</title>
<p>Despite the absence of a specialised excretory system in Xenacoelomorpha, <xref ref-type="bibr" rid="B3">Andrikou et al. (2019)</xref> detected active excretion in this group through the digestive tissue and annotated several genes known to participate in the excretory mechanisms of nephrozoan animals. Here, we annotated in the genomes of Acoela and Nemertodermatida seven of the nine genes involved in the development of the nephridia and one more (<italic>Sall</italic>) in Acoela. Regardless of their phylogenetic position, whether as a sister to Ambulacraria or Nephrozoa, the presence of these genes might be explained by their participation in other important functions. A spatial transcriptomics analysis in the acoel <italic>Isodiametra pulchra</italic> and the nemertodermatid <italic>Meara stichopi</italic> located the expression of <italic>Nephrin</italic> in the brain and the nerve cords (<xref ref-type="bibr" rid="B3">Andrikou et al., 2019</xref>), which resembles observations in mammals and <italic>Drosophila</italic>, the latter through the <italic>Nephrin</italic> homolog <italic>Sns</italic> (<xref ref-type="bibr" rid="B61">Putaala et al., 2000</xref>; <xref ref-type="bibr" rid="B62">Putaala et al., 2001</xref>; <xref ref-type="bibr" rid="B5">Bali et al., 2022</xref>). In contrast, no homologs to the <italic>Osr</italic> gene (named <italic>Odd</italic> in <italic>Drosophila</italic>) could be annotated in any of the acoelomorph genomes. A BLAST search over the two <italic>Xenoturbella</italic> transcriptomes failed to annotate this gene in these species, confirming its absence is a general trait of Xenacoelomorpha. This is noteworthy, as <italic>Osr</italic> is essential in the formation of the excretory organs: in vertebrates, it participates in the formation of the pronephros, the first stage in kidney formation, and its knock-out results in the absence of kidneys (<xref ref-type="bibr" rid="B35">James et al., 2006</xref>); whereas in <italic>Drosophila</italic>, <italic>Odd</italic> participates in the embryogenesis of the tubules of Malpigi (<xref ref-type="bibr" rid="B75">Tena et al., 2007</xref>). Overall, it seems that the molecular machinery that participates in the functioning of a complex ultrafiltration excretory system is present in acoelomorphs, but they lack the one gene necessary to promote the formation of discrete excretory organs.</p>
<p>This pattern fits well within the Nephrozoa hypothesis. In this scenario, the origin of the molecular machinery associated with the excretory organs would be the result of gene co-option, a common phenomenon in the origin of key innovations, such as the development of the radula and shell evolution in molluscs (<xref ref-type="bibr" rid="B34">Hilgers et al., 2018</xref>) or the multiple origins of cnidarian eyes (<xref ref-type="bibr" rid="B60">Picciani et al., 2018</xref>). Interestingly, six of the nine genes investigated have been annotated in different cnidarian species, strengthening the idea of these genes pre-dating the appearance of this specialised excretory system (<xref ref-type="bibr" rid="B24">G&#x105;siorowski et al., 2021</xref>). Thus far, <italic>Osr</italic> has not been annotated in any phylum outside of Nephrozoa, supporting the origin of this gene in the ancestor of this clade. Nevertheless, given the ongoing debate around the phylogenetic position of xenacoelomorphs, the Xenambulacraria hypothesis also needs to be taken into consideration. If Xenacoelomorpha is the sister group of Ambulacraria, additional <italic>ad hoc</italic> hypotheses have to be invoked: either the <italic>Osr</italic> gene was independently gained in Protostomia, Ambulacraria, and Chordata or it was lost in Xenacoelomorpha. The <italic>Drosophila Odd</italic> gene has been shown to activate the formation of kidney tissue in vertebrates (<xref ref-type="bibr" rid="B75">Tena et al., 2007</xref>), which suggests a common origin of both genes in protostomes and deuterostomes. Likewise, the function of this gene is not limited to the development of the excretory organs, but it participates in the development of the foregut in vertebrates (<xref ref-type="bibr" rid="B32">Han et al., 2017</xref>) and it is known to be expressed in the digestive tract of spiralians and hemichordates (<xref ref-type="bibr" rid="B24">G&#x105;siorowski et al., 2021</xref>). Although its general anatomy is variable, the presence of a sack-like gut is considered a plesiomorphy within Xenacoelomorpha (<xref ref-type="bibr" rid="B25">Gavil&#xe1;n et al., 2019</xref>) and the involvement of <italic>Osr</italic> in its development could be expected. In this light, the reduction of the excretory organs alone would not explain the secondary loss of <italic>Osr</italic>, as it would need to be completely nonfunctionalized before that. Regardless of which of the two phylogenetic hypotheses is true, the acquisition of <italic>Osr</italic> and the development of a discrete excretory system seem to be interconnected. The two are a nephrozoan novelty and their origin likely dates back to the most recent common ancestor of protostomes and deuterostomes. More importantly, knock-out experiments demonstrate that there are no other current transcription factors capable of replacing <italic>Osr</italic> in the early formation of the pronephros (<xref ref-type="bibr" rid="B35">James et al., 2006</xref>), which suggests this gene must have assumed this function rather early.</p>
<p>We found statistically significant differences in the gene architecture of all genes but <italic>Nephrin</italic> and <italic>Six</italic>, six of them related to the average exon length. Acoelomorpha is responsible for two-thirds of the differences observed, which fits with the co-option of these genes into the development of the excretory system in the ancestor of Nephrozoa. Changes in gene structure are a strong generator of diversity, particularly after gene duplication, as part of the neofunctionalization of proteins (<xref ref-type="bibr" rid="B78">Xu et al., 2012</xref>). Alternatively, the differences observed might simply be explained by changes in the selective pressures during the acquisition or the reduction of this system, something that might be supported by the observations in Bryozoa. Within protostomes, Bryozoa, which also lack an excretory system, is responsible for most of the variation observed. Notably, half of the gene metrics that are visibly different in this phylum are shared with acoelomorphs: <italic>ZO1</italic> and <italic>Lhx</italic> length, <italic>ZO1</italic> number of exons, and <italic>Sall</italic> average exon length. However, the variation does not always go in the same direction (e.g., the number of exons in <italic>ZO1</italic> increases in Acoelomorpha, but decreases in Bryozoa), likely because the absence of the excretory organs in the two animal groups represents two independent evolutionary events. Some authors have argued that the rapid evolutionary rates observed in Acoelomorpha might be associated with other traits observed in this group, such as chromosomic rearrangements or changes in gene content, misleading comparative analyses and making <italic>Xenoturbella</italic> a better model for studying the evolution of Xenacoelomorpha (<xref ref-type="bibr" rid="B59">Philippe et al., 2019</xref>; <xref ref-type="bibr" rid="B67">Schiffer et al., 2023</xref>). Unfortunately, the genomic data of <italic>X. bocki</italic> is yet not available so we have inferred a gene tree for each of the nine genes analysed and compared the differences in branch lengths among clades to explore this possibility (<xref ref-type="sec" rid="s11">Supplementary Figure S4</xref>). Although branch lengths are indeed significantly longer in acoelomorphs than in any other clade (except in <italic>Lhx</italic> and <italic>Six</italic>), they are also longer in deuterostomes compared to protostomes despite the similarities between the two clades. In more detail, protostomes present the shortest branches in the gene trees, while Bryozoa is one of the phyla with the most changes in gene architecture. Hence, the accelerated evolutionary rates of Acoelomorpha do not seem to be the main factor underlying the differences observed in these genes, although it would be interesting to confirm this once all the data from the <italic>Xenoturbella</italic> genome is publicly available.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In this study, we have generated the first draft of a nemertodermatid genome, sequenced from a single, microscopic individual using the Ultra-Low DNA Input protocol and HiFi. We show that this approach is capable of producing genomes of relatively good quality even from small organisms with large genomes. The main drawback is genome contiguity, which remains the main challenge and one of the avenues in genome sequencing that need the most attention. Nevertheless, genome quality is good enough to annotate full proteins, allowing detailed analysis of gene architecture. We prove this by analysing the genes related to the ultrafiltration excretory system. We observe that the molecular machinery related to this system predates its origin, as most of the genes were present in Urbilateria or even in the cnidarian-bilaterian ancestor. Interestingly, all genes but <italic>Osr</italic>, the one gene triggering the formation of these organs, were annotated in Xenacoelomorpha. Thus far, gene architecture is markedly different in Acoelomorpha, which cannot be explained either by the accelerated evolution of this clade or the lack of the excretory system alone. All these findings are more easily explained under the Nephrozoa hypothesis.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/genbank/">https://www.ncbi.nlm.nih.gov/genbank/</ext-link>, GenBank Bioproject PRJNA981986 <ext-link ext-link-type="uri" xlink:href="https://figshare.com/projects/2023_Nemertoderma_westbladi_genome/169818">https://figshare.com/projects/2023_Nemertoderma_westbladi_genome/169818</ext-link>.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>SA, OV, and UJ conceived the project; SA performed DNA extractions; JH was responsible for library preparations and sequencing; CT-R carried out the post-sequencing analyses, from quality filtering to genome assembly; SA decontaminated and annotated the genome and performed comparative analyses; SA and UJ led the writing of the manuscript. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>This project was funded by the VR project 2018-05191, granted to UJ, and the &#x201c;2021 Riksmusei V&#x00E4;nner&#x201d; and &#x201c;2020 Helge Ax:son Johnsons stiftelse&#x201d; stipends to SA. Work performed at NGI/Uppsala Genome Center has been funded by RFI/VR and Science for Life Laboratory, Sweden. Analyses and data handling were enabled by resources in projects SNIC 2020/15-191 and SNIC 2021/22-562 provided by the National Academic Infrastructure for Supercomputing in Sweden (NAISS) at UPPMAX, funded by the Swedish Research Council through grant agreement no. 2018-05191.</p>
</sec>
<ack>
<p>We are thankful to C. Laumer (NHM) for his advice during the early stages of this project. The authors would like to acknowledge support of the National Genomics Infrastructure (NGI)/Uppsala Genome Center for providing assistance in massive parallel sequencing and computational infrastructure.</p>
</ack>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2023.1244493/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2023.1244493/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table2.XLSX" id="SM1" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table3.XLSX" id="SM2" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image5.PNG" id="SM3" mimetype="application/PNG" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image4.PNG" id="SM4" mimetype="application/PNG" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table4.XLSX" id="SM5" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image2.PNG" id="SM6" mimetype="application/PNG" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table1.XLSX" id="SM7" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image1.PNG" id="SM8" mimetype="application/PNG" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table5.XLSX" id="SM9" mimetype="application/XLSX" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image3.PNG" id="SM10" mimetype="application/PNG" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Albertin</surname>
<given-names>C. B.</given-names>
</name>
<name>
<surname>Simakov</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Mitros</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z. Y.</given-names>
</name>
<name>
<surname>Pungor</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Edsinger-Gonzales</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>The octopus genome and the evolution of cephalopod neural and morphological novelties</article-title>. <source>Nature</source> <volume>524</volume>, <fpage>220</fpage>&#x2013;<lpage>224</lpage>. <pub-id pub-id-type="doi">10.1038/nature14668</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alonge</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lebeigle</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kirsche</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Aganezov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lippman</surname>
<given-names>Z. B.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Automated assembly scaffolding using RagTag elevates a new tomato system for high-throughput genome editing</article-title>. <source>Genome Biol.</source> <volume>23</volume>, <fpage>258</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-022-02823-7</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andrikou</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Thiel</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ruiz-Santiesteban</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Hejnol</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Active mode of excretion across digestive tissues predates the origin of excretory organs</article-title>. <source>PLoS Biol.</source> <volume>17</volume>, <fpage>e3000408</fpage>&#x2013;<lpage>e3000422</lpage>. <pub-id pub-id-type="doi">10.1371/journal.pbio.3000408</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arimoto</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hikosaka-Katayama</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hikosaka</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tagawa</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Inoue</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ueki</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>A draft nuclear-genome assembly of the acoel flatworm Praesagittifera naikaiensis</article-title>. <source>Gigascience</source> <volume>8</volume>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1093/gigascience/giz023</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bali</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>H. K.</given-names>
</name>
<name>
<surname>Zinn</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Sticks and Stones, a conserved cell surface ligand for the Type IIa RPTP Lar, regulates neural circuit wiring in Drosophila</article-title>. <source>Elife</source> <volume>11</volume>, <fpage>714699</fpage>&#x2013;<lpage>e71530</lpage>. <pub-id pub-id-type="doi">10.7554/eLife.71469</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bankevich</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nurk</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Antipov</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Gurevich</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Dvorkin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kulikov</surname>
<given-names>A. S.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>SPAdes: A new genome assembly algorithm and its applications to single-cell sequencing</article-title>. <source>J. Comput. Biol.</source> <volume>19</volume>, <fpage>455</fpage>&#x2013;<lpage>477</lpage>. <pub-id pub-id-type="doi">10.1089/cmb.2012.0021</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boscaro</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Holt</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Van Steenkiste</surname>
<given-names>N. W. L.</given-names>
</name>
<name>
<surname>Herranz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Irwin</surname>
<given-names>N. A. T.</given-names>
</name>
<name>
<surname>&#xc0;lvarez-Campos</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Microbiomes of microscopic marine invertebrates do not reveal signatures of phylosymbiosis</article-title>. <source>Nat. Microbiol.</source> <volume>7</volume>, <fpage>810</fpage>&#x2013;<lpage>819</lpage>. <pub-id pub-id-type="doi">10.1038/s41564-022-01125-9</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brown</surname>
<given-names>C. T.</given-names>
</name>
<name>
<surname>Hug</surname>
<given-names>L. A.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>B. C.</given-names>
</name>
<name>
<surname>Sharon</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Castelle</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Unusual biology across a group comprising more than 15% of domain Bacteria</article-title>. <source>Nature</source> <volume>523</volume>, <fpage>208</fpage>&#x2013;<lpage>211</lpage>. <pub-id pub-id-type="doi">10.1038/nature14486</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Br&#x16f;na</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hoff</surname>
<given-names>K. J.</given-names>
</name>
<name>
<surname>Lomsadze</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Stanke</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Borodovsky</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>BRAKER2: automatic eukaryotic genome annotation with GeneMark-ep&#x2b; and AUGUSTUS supported by a protein database</article-title>. <source>Nar. Genomics Bioinforma.</source> <volume>3</volume>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1093/nargab/lqaa108</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Br&#x16f;na</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lomsadze</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Borodovsky</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>GeneMark-EP&#x2b;: eukaryotic gene prediction with self-training in the space of genes and proteins</article-title>. <source>Nar. Genomics Bioinforma.</source> <volume>2</volume>, <fpage>lqaa026</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1093/nargab/lqaa026</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Buchfink</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Reuter</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Drost</surname>
<given-names>H. G.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Sensitive protein alignments at tree-of-life scale using DIAMOND</article-title>. <source>Nat. Methods</source> <volume>18</volume>, <fpage>366</fpage>&#x2013;<lpage>368</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-021-01101-x</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cannon</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Vellutini</surname>
<given-names>B. C.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ronquist</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Jondelius</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Hejnol</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Xenacoelomorpha is the sister group to Nephrozoa</article-title>. <source>Nature</source> <volume>530</volume>, <fpage>89</fpage>&#x2013;<lpage>93</lpage>. <pub-id pub-id-type="doi">10.1038/nature16520</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Challis</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Richards</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Rajan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cochrane</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Blaxter</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>BlobToolKit - interactive quality assessment of genome assemblies</article-title>. <source>G3 Genes, Genomes, Genet.</source> <volume>10</volume>, <fpage>1361</fpage>&#x2013;<lpage>1374</lpage>. <pub-id pub-id-type="doi">10.1534/g3.119.400908</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Concepcion</surname>
<given-names>G. T.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Haplotype-resolved de novo assembly using phased assembly graphs with hifiasm</article-title>. <source>Nat. Methods</source> <volume>18</volume>, <fpage>170</fpage>&#x2013;<lpage>175</lpage>. <pub-id pub-id-type="doi">10.1038/s41592-020-01056-5</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Consortium</surname>
<given-names>I. H. G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Comparative genomics of the major parasitic worms</article-title>. <source>Nat. Genet.</source> <volume>51</volume>, <fpage>163</fpage>&#x2013;<lpage>174</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-018-0262-1</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Criscuolo</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gribaldo</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>BMGE (block mapping and gathering with entropy): A new software for selection of phylogenetic informative regions from multiple sequence alignments</article-title>. <source>BMC Evol. Biol.</source> <volume>10</volume>, <fpage>210</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2148-10-210</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Decena-Segarra</surname>
<given-names>L. P.</given-names>
</name>
<name>
<surname>Bizjak-Mali</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kladnik</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sessions</surname>
<given-names>S. K.</given-names>
</name>
<name>
<surname>Rovito</surname>
<given-names>S. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Miniaturization, genome size, and biological size in a diverse clade of salamanders</article-title>. <source>Am. Nat.</source> <volume>196</volume>, <fpage>634</fpage>&#x2013;<lpage>648</lpage>. <pub-id pub-id-type="doi">10.1086/711019</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dobin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Davis</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Schlesinger</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Drenkow</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zaleski</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jha</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Star: ultrafast universal RNA-seq aligner</article-title>. <source>Bioinformatics</source> <volume>29</volume>, <fpage>15</fpage>&#x2013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bts635</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dos Reis</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Thawornwattana</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Angelis</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Telford</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Donoghue</surname>
<given-names>P. C. J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Uncertainty in the timing of origin of animals and the limits of precision in molecular timescales</article-title>. <source>Curr. Biol.</source> <volume>25</volume>, <fpage>2939</fpage>&#x2013;<lpage>2950</lpage>. <pub-id pub-id-type="doi">10.1016/j.cub.2015.09.066</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dunwell</surname>
<given-names>T. L.</given-names>
</name>
<name>
<surname>Paps</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Holland</surname>
<given-names>P. W. H.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Novel and divergent genes in the evolution of placental mammals</article-title>. <source>Proc. R. Soc. B Biol. Sci.</source> <volume>284</volume>, <fpage>20171357</fpage>. <pub-id pub-id-type="doi">10.1098/rspb.2017.1357</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elliott</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Gregory</surname>
<given-names>T. R.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>What&#x2019;s in a genome? The C-value enigma and the evolution of eukaryotic genome content</article-title>. <source>Philos. Trans. R. Soc. B Biol. Sci.</source> <volume>370</volume>, <fpage>20140331</fpage>. <pub-id pub-id-type="doi">10.1098/rstb.2014.0331</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Emms</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Kelly</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>OrthoFinder: phylogenetic orthology inference for comparative genomics</article-title>. <source>Genome Biol.</source> <volume>20</volume>, <fpage>238</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1832-y</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Niu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>CD-HIT: accelerated for clustering the next-generation sequencing data</article-title>. <source>Bioinformatics</source> <volume>28</volume>, <fpage>3150</fpage>&#x2013;<lpage>3152</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bts565</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>G&#x105;siorowski</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Andrikou</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Janssen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Bump</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Budd</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Lowe</surname>
<given-names>C. J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Molecular evidence for a single origin of ultrafiltration-based excretory organs</article-title>. <source>Curr. Biol.</source> <volume>31</volume>, <fpage>3629</fpage>&#x2013;<lpage>3638.e2</lpage>. <pub-id pub-id-type="doi">10.1016/j.cub.2021.05.057</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gavil&#xe1;n</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sprecher</surname>
<given-names>S. G.</given-names>
</name>
<name>
<surname>Hartenstein</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Martinez</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>The digestive system of xenacoelomorphs</article-title>. <source>Cell Tissue Res.</source> <volume>377</volume>, <fpage>369</fpage>&#x2013;<lpage>382</lpage>. <pub-id pub-id-type="doi">10.1007/s00441-019-03038-2</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gehrke</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Neverett</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>Y. J.</given-names>
</name>
<name>
<surname>Brandt</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ricci</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hulett</surname>
<given-names>R. E.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Acoel genome reveals the regulatory landscape of whole-body regeneration</article-title>. <source>Science</source> <volume>80</volume>, <fpage>363</fpage>. <pub-id pub-id-type="doi">10.1126/science.aau6173</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Giribet</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Edgecombe</surname>
<given-names>G. D.</given-names>
</name>
</person-group> (<year>2020</year>). <source>The invertebrate tree of life</source>. <publisher-loc>Princeton, New Jersey</publisher-loc>: <publisher-name>Princeton University Press</publisher-name>. <pub-id pub-id-type="doi">10.2307/j.ctvscxrhm</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grabherr</surname>
<given-names>M. G.</given-names>
</name>
<name>
<surname>Haas</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Yassour</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Levin</surname>
<given-names>J. Z.</given-names>
</name>
<name>
<surname>Thompson</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Amit</surname>
<given-names>I.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Full-length transcriptome assembly from RNA-Seq data without a reference genome</article-title>. <source>Nat. Biotechnol.</source> <volume>29</volume>, <fpage>644</fpage>&#x2013;<lpage>652</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.1883</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gross</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Treffkorn</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Reichelt</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Epple</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>L&#xfc;ter</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mayer</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Miniaturization of tardigrades (water bears): morphological and genomic perspectives</article-title>. <source>Arthropod Struct. Dev.</source> <volume>48</volume>, <fpage>12</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1016/j.asd.2018.11.006</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Guimaraes</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <source>DNA extraction (Salting out) V.4</source>. <publisher-loc>California</publisher-loc>: <publisher-name>Protocols.io</publisher-name>. <pub-id pub-id-type="doi">10.17504/protocols.io.vwfe7bn</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gurevich</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Saveliev</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Vyahhi</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Tesler</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Quast: quality assessment tool for genome assemblies</article-title>. <source>Bioinformatics</source> <volume>29</volume>, <fpage>1072</fpage>&#x2013;<lpage>1075</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btt086</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Grigg</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Slack</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chaturvedi</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Osr1 functions downstream of Hedgehog pathway to regulate foregut development</article-title>. <source>Dev. Biol.</source> <volume>427</volume>, <fpage>72</fpage>&#x2013;<lpage>83</lpage>. <pub-id pub-id-type="doi">10.1016/j.ydbio.2017.05.005</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haszprunar</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Review of data for a morphological look on Xenacoelomorpha (Bilateria incertae sedis)</article-title>. <source>Org. Divers. Evol.</source> <volume>16</volume>, <fpage>363</fpage>&#x2013;<lpage>389</lpage>. <pub-id pub-id-type="doi">10.1007/s13127-015-0249-z</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hilgers</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hartmann</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hofreiter</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>von Rintelen</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Novel genes, ancient genes, and gene Co-option contributed to the genetic basis of the radula, a Molluscan innovation</article-title>. <source>Mol. Biol. Evol.</source> <volume>35</volume>, <fpage>1638</fpage>&#x2013;<lpage>1652</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msy052</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>James</surname>
<given-names>R. G.</given-names>
</name>
<name>
<surname>Kamei</surname>
<given-names>C. N.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Schulthesis</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Schultheiss</surname>
<given-names>T. M.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Odd-skipped related 1 is required for development of the metanephric kidney and regulates formation and differentiation of kidney precursor cells</article-title>. <source>Dev. Dis.</source> <volume>133</volume>, <fpage>2995</fpage>&#x2013;<lpage>3004</lpage>. <pub-id pub-id-type="doi">10.1242/dev.02442</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jondelius</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Ruiz-Trillo</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Bagu&#xf1;&#xe0;</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Riutort</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>The Nemertodermatida are basal bilaterians and not members of the Platyhelminthes</article-title>. <source>Zool. Scr.</source> <volume>31</volume>, <fpage>201</fpage>&#x2013;<lpage>215</lpage>. <pub-id pub-id-type="doi">10.1046/j.1463-6409.2002.00090x</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Juravel</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Porras</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>H&#xf6;hna</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pisani</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>W&#xf6;rheide</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Exploring genome gene content and morphological analysis to test recalcitrant nodes in the animal phylogeny</article-title>. <source>PLoS One</source> <volume>18</volume>, <fpage>e0282444</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0282444</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kapli</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Telford</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Topology-dependent asymmetry in systematic errors affects phylogenetic placement of Ctenophora and Xenacoelomorpha</article-title>. <source>Sci. Adv.</source> <volume>6</volume>, <fpage>eabc5162</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1126/sciadv.abc5162</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Katoh</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Standley</surname>
<given-names>D. M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>MAFFT multiple sequence alignment software version 7: improvements in performance and usability</article-title>. <source>Mol. Biol. Evol.</source> <volume>30</volume>, <fpage>772</fpage>&#x2013;<lpage>780</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/mst010</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Paggi</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bennett</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Salzberg</surname>
<given-names>S. L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Graph-based genome alignment and genotyping with HISAT2 and HISAT-genotype</article-title>. <source>Nat. Biotechnol.</source> <volume>37</volume>, <fpage>907</fpage>&#x2013;<lpage>915</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-019-0201-4</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kingan</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Heaton</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Cudini</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lambert</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Baybayan</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Galvin</surname>
<given-names>B. D.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>A high-quality de novo genome assembly from a single mosquito using pacbio sequencing</article-title>. <source>Genes (Basel)</source> <volume>10</volume>, <fpage>62</fpage>. <pub-id pub-id-type="doi">10.3390/genes10010062</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kolmogorov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Pevzner</surname>
<given-names>P. A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Assembly of long, error-prone reads using repeat graphs</article-title>. <source>Nat. Biotechnol.</source> <volume>37</volume>, <fpage>540</fpage>&#x2013;<lpage>546</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-019-0072-8</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Korlach</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <source>A high-quality PacBio insect genome from 5 ng of input. DNA</source>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>K&#xfc;ck</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Longo</surname>
<given-names>G. C.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>FASconCAT-G: extensive functions for multiple sequence alignment preparations concerning phylogenetic studies</article-title>. <source>Front. Zool.</source> <volume>11</volume>, <fpage>81</fpage>&#x2013;<lpage>88</lpage>. <pub-id pub-id-type="doi">10.1186/s12983-014-0081-x</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Langmead</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Salzberg</surname>
<given-names>S. L.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Fast gapped-read alignment with Bowtie 2</article-title>. <source>Nat. Methods</source> <volume>9</volume>, <fpage>357</fpage>&#x2013;<lpage>359</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.1923</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lefort</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Desper</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gascuel</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>FastME 2.0: A comprehensive, accurate, and fast distance-based phylogeny inference program</article-title>. <source>Mol. Biol. Evol.</source> <volume>32</volume>, <fpage>2798</fpage>&#x2013;<lpage>2800</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msv150</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>New strategies to improve minimap2 alignment accuracy</article-title>. <source>Bioinformatics</source> <volume>37</volume>, <fpage>4572</fpage>&#x2013;<lpage>4574</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab705</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hui</surname>
<given-names>T. H.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Chromosome evolution and genome miniaturization in minifish</article-title>. <source>PLoS One</source> <volume>7</volume>, <fpage>e37305</fpage>&#x2013;<lpage>e37307</lpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0037305</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lord</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cunha</surname>
<given-names>T. J.</given-names>
</name>
<name>
<surname>de Medeiros</surname>
<given-names>B. A. S.</given-names>
</name>
<name>
<surname>Sato</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Khost</surname>
<given-names>D. E.</given-names>
</name>
<name>
<surname>Sackton</surname>
<given-names>T. B.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Expanding on our knowledge of ecdysozoan genomes, a contiguous assembly of the meiofaunal prapulan Tubiluchus corallicola</article-title>. <source>Genome Biol. Evol.</source>, <volume>15</volume> (<issue>6</issue>) <fpage>evad103</fpage>. <pub-id pub-id-type="doi">10.1093/gbe/evad103</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lundin</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Symbiotic bacteria on the epidermis of species of the Nemertodermatida (Platyhelminthes, Acoelomorpha)</article-title>. <source>Acta Zool.</source> <volume>79</volume>, <fpage>187</fpage>&#x2013;<lpage>191</lpage>. <pub-id pub-id-type="doi">10.1111/j.1463-6395.1998.tb01157x</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mar&#xe7;ais</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kingsford</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>A fast, lock-free approach for efficient parallel counting of occurrences of k-mers</article-title>. <source>Bioinformatics</source> <volume>27</volume>, <fpage>764</fpage>&#x2013;<lpage>770</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btr011</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mart&#xed;n-Dur&#xe1;n</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>B&#xf8;rve</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>L&#xea;</surname>
<given-names>H. S.</given-names>
</name>
<name>
<surname>Furu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cannon</surname>
<given-names>J. T.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Convergent evolution of bilaterian nerve cords</article-title>. <source>Nature</source> <volume>553</volume>, <fpage>45</fpage>&#x2013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1038/nature25030</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Martinez</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ustyantsev</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Biryukov</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mouton</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Glasenburg</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Sprecher</surname>
<given-names>S. G.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Genome assembly of the acoel flatworm Symsagittifera roscoffensis, a model for research on body plan evolution and photosymbiosis</article-title>. <source>G3 Genes&#x7c;Genomes&#x7c;Genetics</source> <volume>13</volume>. <pub-id pub-id-type="doi">10.1093/g3journal/jkac336</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meyer</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schloissnig</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Franchini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Woltering</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Irisarri</surname>
<given-names>I.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Giant lungfish genome elucidates the conquest of land by vertebrates</article-title>. <source>Nature</source> <volume>590</volume>, <fpage>284</fpage>&#x2013;<lpage>289</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-021-03198-8</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Minh</surname>
<given-names>B. Q.</given-names>
</name>
<name>
<surname>Schmidt</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>Chernomor</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Schrempf</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Woodhams</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Von Haeseler</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>IQ-TREE 2: new models and efficient methods for phylogenetic inference in the genomic era</article-title>. <source>Mol. Biol. Evol.</source> <volume>37</volume>, <fpage>1530</fpage>&#x2013;<lpage>1534</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msaa015</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mistry</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bateman</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Finn</surname>
<given-names>R. D.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Predicting active site residue annotations in the Pfam database</article-title>. <source>BMC Bioinforma.</source> <volume>8</volume>, <fpage>298</fpage>&#x2013;<lpage>314</lpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-8-298</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nam</surname>
<given-names>B. H.</given-names>
</name>
<name>
<surname>Kwak</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>Y. O.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>D. G.</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>W. J.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Genome sequence of pacific abalone (Haliotis discus hannai): the first draft genome in family haliotidae</article-title>. <source>Gigascience</source> <volume>6</volume>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1093/gigascience/gix014</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>L. T.</given-names>
</name>
<name>
<surname>Schmidt</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>von Haeseler</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Minh</surname>
<given-names>B. Q.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>IQ-TREE: A fast and effective stochastic algorithm for estimating maximum-likelihood phylogenies</article-title>. <source>Mol. Biol. Evol.</source> <volume>32</volume>, <fpage>268</fpage>&#x2013;<lpage>274</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msu300</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Philippe</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Poustka</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Chiodin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hoff</surname>
<given-names>K. J.</given-names>
</name>
<name>
<surname>Dessimoz</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tomiczek</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Mitigating anticipated effects of systematic errors supports sister-group relationship between Xenacoelomorpha and Ambulacraria</article-title>. <source>Curr. Biol.</source> <volume>29</volume>, <fpage>1818</fpage>&#x2013;<lpage>1826</lpage>. <pub-id pub-id-type="doi">10.1016/j.cub.2019.04.009</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Picciani</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kerlin</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Sierra</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Swafford</surname>
<given-names>A. J. M.</given-names>
</name>
<name>
<surname>Ramirez</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Roberts</surname>
<given-names>N. G.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Prolific origination of eyes in Cnidaria with Co-option of non-visual opsins</article-title>. <source>Curr. Biol.</source> <volume>28</volume>, <fpage>2413</fpage>&#x2013;<lpage>2419</lpage>. <pub-id pub-id-type="doi">10.1016/j.cub.2018.05.055</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Putaala</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sainio</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Sariola</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Tryggvason</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Primary structure of mouse and rat nephrin cDNA and structure and expression of the mouse gene</article-title>. <source>J. Am. Soc. Nephrol.</source> <volume>11</volume>, <fpage>991</fpage>&#x2013;<lpage>1001</lpage>. <pub-id pub-id-type="doi">10.1681/asn.v116991</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Putaala</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Soininen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kilpel&#xe4;inen</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wartiovaara</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tryggvason</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>The murine nephrin gene is specifically expressed in kidney, brain and pancreas: inactivation of the gene leads to massive proteinuria and neonatal death</article-title>. <source>Hum. Mol. Genet.</source> <volume>10</volume>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1093/hmg/10.1.1</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ranallo-Benavidez</surname>
<given-names>T. R.</given-names>
</name>
<name>
<surname>Jaron</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Schatz</surname>
<given-names>M. C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>GenomeScope 2.0 and Smudgeplot for reference-free profiling of polyploid genomes</article-title>. <source>Nat. Commun.</source> <volume>11</volume>, <fpage>1432</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-020-14998-3</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rubin</surname>
<given-names>B. E. R.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>B. M.</given-names>
</name>
<name>
<surname>Hunt</surname>
<given-names>B. G.</given-names>
</name>
<name>
<surname>Kocher</surname>
<given-names>S. D.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Rate variation in the evolution of non-coding DNA associated with social evolution in bees</article-title>. <source>Philos. Trans. R. Soc. B Biol. Sci.</source> <volume>374</volume>, <fpage>20180247</fpage>. <pub-id pub-id-type="doi">10.1098/rstb.2018.0247</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ruiz-Trillo</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Riutort</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Timothy</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Littlewood</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Herniou</surname>
<given-names>E. A.</given-names>
</name>
<name>
<surname>Bagu&#xf1;&#xe0;</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Acoel flatworms: earliest extant bilaterian metazoans, not members of platyhelminthes</article-title>. <source>Sci. (80-. )</source> <volume>283</volume>, <fpage>1919</fpage>&#x2013;<lpage>1923</lpage>. <pub-id pub-id-type="doi">10.1126/science.283.5409.1919</pub-id>
</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sarmashghi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bohmann</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Gilbert</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bafna</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Mirarab</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Assembly-free and alignment-free sample identification using genome skims</article-title>. <source>Genome Biol.</source> <volume>20</volume>, <fpage>1</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1101/230409</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schiffer</surname>
<given-names>P. H.</given-names>
</name>
<name>
<surname>Natsidis</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Leite</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Robertson</surname>
<given-names>H. E.</given-names>
</name>
<name>
<surname>Lapraz</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Marl&#x00E9;taz</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <source>The slow evolving genomes of the xenacoelomorph worm Xenoturbella bocki</source>. <publisher-name>bioRxiv</publisher-name>. <pub-id pub-id-type="doi">10.1101/2022.06.24.497508</pub-id>
</citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schmieder</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Edwards</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Quality control and preprocessing of metagenomic datasets</article-title>. <source>Bioinformatics</source> <volume>27</volume>, <fpage>863</fpage>&#x2013;<lpage>864</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btr026</pub-id>
</citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schneider</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Woehle</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Greve</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>D&#x2019;Haese</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Wolf</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hiller</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Two high-quality de novo genomes from single ethanol-preserved specimens of tiny metazoans (Collembola)</article-title>. <source>Gigascience</source> <volume>10</volume>, <fpage>giab035</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1093/gigascience/giab035</pub-id>
</citation>
</ref>
<ref id="B70">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Seppey</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Manni</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zdobnov</surname>
<given-names>E. M.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>BUSCO: assessing genome assembly and annotation completeness</article-title>,&#x201d; in <source>Gene prediction. Methods in molecular biology</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>Humana Press</publisher-name>).</citation>
</ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shah</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hoffman</surname>
<given-names>J. I.</given-names>
</name>
<name>
<surname>Schielzeth</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Comparative analysis of genomic repeat content in gomphocerine grasshoppers reveals expansion of satellite DNA and helitrons in species with unusually large genomes</article-title>. <source>Genome Biol. Evol.</source> <volume>12</volume>, <fpage>1180</fpage>&#x2013;<lpage>1193</lpage>. <pub-id pub-id-type="doi">10.1093/GBE/EVAA119</pub-id>
</citation>
</ref>
<ref id="B72">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Slyusarev</surname>
<given-names>G. S.</given-names>
</name>
<name>
<surname>Starunov</surname>
<given-names>V. V.</given-names>
</name>
<name>
<surname>Bondarenko</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Zorina</surname>
<given-names>N. A.</given-names>
</name>
<name>
<surname>Bondarenko</surname>
<given-names>N. I.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Extreme genome and nervous system streamlining in the invertebrate parasite Intoshia variabili</article-title>. <source>Curr. Biol.</source> <volume>30</volume>, <fpage>1292</fpage>&#x2013;<lpage>1298</lpage>. <pub-id pub-id-type="doi">10.1016/j.cub.2020.01.061</pub-id>
</citation>
</ref>
<ref id="B73">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Smit</surname>
<given-names>A. F. A.</given-names>
</name>
<name>
<surname>Hubley</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2015</year>). <source>RepeatMasker open-4.0</source>.</citation>
</ref>
<ref id="B74">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Smit</surname>
<given-names>A. F. A.</given-names>
</name>
<name>
<surname>Hubley</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2015</year>). <source>RepeatModeler open-1.0</source>.</citation>
</ref>
<ref id="B75">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tena</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Neto</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>de la Calle-Mustienes</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Bras-Pereira</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Casares</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>G&#xf3;mez-Skarmeta</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Odd-skipped genes encode repressors that control kidney development</article-title>. <source>Dev. Biol.</source> <volume>301</volume>, <fpage>518</fpage>&#x2013;<lpage>531</lpage>. <pub-id pub-id-type="doi">10.1016/j.ydbio.2006.08.063</pub-id>
</citation>
</ref>
<ref id="B76">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Thal&#xe9;n</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Kocot</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Haddock</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <source>PhyloPyPruner: Tree-based orthology inference for phylogenomics</source>.</citation>
</ref>
<ref id="B77">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>T&#xf8;rresen</surname>
<given-names>O. K.</given-names>
</name>
<name>
<surname>Star</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Mier</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Andrade-Navarro</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Bateman</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jarnot</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Tandem repeats lead to sequence assembly errors and impose multi-level challenges for genome and protein databases</article-title>. <source>Nucleic Acids Res.</source> <volume>47</volume>, <fpage>10994</fpage>&#x2013;<lpage>11006</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz841</pub-id>
</citation>
</ref>
<ref id="B78">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Divergence of duplicate genes in exon-intron structure</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>109</volume>, <fpage>1187</fpage>&#x2013;<lpage>1192</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1109047109</pub-id>
</citation>
</ref>
<ref id="B79">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y. H.</given-names>
</name>
<name>
<surname>Mei</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Comparative genomics sheds light on the convergent evolution of miniaturized wasps</article-title>. <source>Mol. Biol. Evol.</source> <volume>38</volume>, <fpage>5539</fpage>&#x2013;<lpage>5554</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msab273</pub-id>
</citation>
</ref>
<ref id="B80">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yoshida</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Konno</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nishino</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Murai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tomita</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Arakawa</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Ultralow input genome sequencing library preparation from a single tardigrade specimen</article-title>. <source>J. Vis. Exp.</source>, <fpage>57615</fpage>&#x2013;<lpage>57618</lpage>. <pub-id pub-id-type="doi">10.3791/57615</pub-id>
</citation>
</ref>
<ref id="B81">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Dunham</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Comparative genome analysis of 52 fish species suggests differential associations of repetitive elements with their living aquatic environments</article-title>. <source>BMC Genomics</source> <volume>19</volume>, <fpage>141</fpage>&#x2013;<lpage>210</lpage>. <pub-id pub-id-type="doi">10.1186/s12864-018-4516-1</pub-id>
</citation>
</ref>
<ref id="B82">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sayyari</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Mirarab</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>ASTRAL-III: increased scalability and impacts of contracting low support branches</article-title>,&#x201d; in <source>Comparative genomics. RECOMB-CG 2017. Lecture notes in computer science</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>).</citation>
</ref>
<ref id="B83">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Larkin</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Comparative genomics reveals insights into avian genome evolution and adaptation</article-title>. <source>Sci. (80-</source> <volume>346</volume>, <fpage>1311</fpage>&#x2013;<lpage>1320</lpage>. <pub-id pub-id-type="doi">10.1126/science.1251385</pub-id>
</citation>
</ref>
<ref id="B84">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>B. H.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xue</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>G. C.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>M. Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J. T.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>P_RNA_scaffolder: A fast and accurate genome scaffolder using paired-end RNA-sequencing reads</article-title>. <source>BMC Genomics</source> <volume>19</volume>, <fpage>175</fpage>&#x2013;<lpage>213</lpage>. <pub-id pub-id-type="doi">10.1186/s12864-018-4567-3</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>