<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Microbiol.</journal-id>
<journal-title>Frontiers in Microbiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Microbiol.</abbrev-journal-title>
<issn pub-type="epub">1664-302X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmicb.2022.855666</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Microbiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Full-Length Genome of an <italic>Ogataea polymorpha</italic> Strain CBS4732 <italic>ura3</italic>&#x0394; Reveals Large Duplicated Segments in Subtelomeric Regions</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Chang</surname> <given-names>Jia</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Bei</surname> <given-names>Jinlong</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/487658/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Shao</surname> <given-names>Qi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1693757/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Hemu</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Fan</surname> <given-names>Huan</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Yau</surname> <given-names>Tung On</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/529148/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Bu</surname> <given-names>Wenjun</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Ruan</surname> <given-names>Jishou</given-names></name>
<xref ref-type="aff" rid="aff8"><sup>8</sup></xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Wei</surname> <given-names>Dongsheng</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/767603/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Gao</surname> <given-names>Shan</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/512809/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Key Laboratory of Molecular Microbiology and Technology, Ministry of Education, College of Life Science, Nankai University</institution>, <addr-line>Tianjin</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Agro-Biological Gene Research Center, Guangdong Academy of Agricultural Sciences, Guangdong Provincial Key Laboratory for Crop Germplasm Resources Preservation and Utilization</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Guangdong Laboratory for Lingnan Modern Agriculture</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Tianjin Hemu Health Biotechnological Co., Ltd.</institution>, <addr-line>Tianjin</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>Tianjin Institute of Animal Husbandry and Veterinary Research</institution>, <addr-line>Tianjin</addr-line>, <country>China</country></aff>
<aff id="aff6"><sup>6</sup><institution>John Van Geest Cancer Research Centre, School of Science and Technology, Nottingham Trent University</institution>, <addr-line>Nottingham</addr-line>, <country>United Kingdom</country></aff>
<aff id="aff7"><sup>7</sup><institution>Department of Rural Land Use, Scotland&#x2019;s Rural College</institution>, <addr-line>Aberdeen</addr-line>, <country>United Kingdom</country></aff>
<aff id="aff8"><sup>8</sup><institution>School of Mathematical Sciences, Nankai University</institution>, <addr-line>Tianjin</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Feng Gao, Tianjin University, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Fabio Mitsuo Lima, Centro Universit&#x00E1;rio S&#x00E3;o Camilo, Brazil; Sara Hanson, Colorado College, United States; Yanzhu Ji, Institute of Zoology (CAS), China; Dzmitry Padhorny, Stony Brook University, United States</p></fn>
<corresp id="c001">&#x002A;Correspondence: Dongsheng Wei, <email>weidongsheng@nankai.edu.cn</email></corresp>
<corresp id="c002">Shan Gao, <email>gao_shan@mail.nankai.edu.cn</email></corresp>
<fn fn-type="equal" id="fn002"><p><sup>&#x2020;</sup>These authors have contributed equally to this work</p></fn>
<fn fn-type="other" id="fn004"><p>This article was submitted to Evolutionary and Genomic Microbiology, a section of the journal Frontiers in Microbiology</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>04</day>
<month>04</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>855666</elocation-id>
<history>
<date date-type="received">
<day>15</day>
<month>01</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>25</day>
<month>02</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2022 Chang, Bei, Shao, Wang, Fan, Yau, Bu, Ruan, Wei and Gao.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Chang, Bei, Shao, Wang, Fan, Yau, Bu, Ruan, Wei and Gao</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Currently, methylotrophic yeasts (e.g., <italic>Pichia pastoris</italic>, <italic>Ogataea polymorpha</italic>, and <italic>Candida boindii</italic>) are subjects of intense genomics studies in basic research and industrial applications. In the genus <italic>Ogataea</italic>, most research is focused on three basic <italic>O. polymorpha</italic> strains-CBS4732, NCYC495, and DL-1. However, the relationship between CBS4732, NCYC495, and DL-1 remains unclear, as the genomic differences between them have not be exactly determined without their high-quality complete genomes. As a nutritionally deficient mutant derived from CBS4732, the <italic>O. polymorpha</italic> strain CBS4732 <italic>ura3</italic>&#x0394; (named HU-11) is being used for high-yield production of several important proteins or peptides. HU-11 has the same reference genome as CBS4732 (noted as HU-11/CBS4732), because the only genomic difference between them is a 5-bp insertion.</p>
</sec>
<sec>
<title>Results</title>
<p>In the present study, we have assembled the full-length genome of <italic>O. polymorpha</italic> HU-11/CBS4732 using high-depth PacBio and Illumina data. Long terminal repeat retrotransposons (LTR-rts), rDNA, 5&#x2032; and 3&#x2032; telomeric, subtelomeric, low complexity and other repeat regions were exactly determined to improve the genome quality. In brief, the main findings include complete rDNAs, complete LTR-rts, three large duplicated segments in subtelomeric regions and three structural variations between the HU-11/CBS4732 and NCYC495 genomes. These findings are very important for the assembly of full-length genomes of yeast and the correction of assembly errors in the published genomes of <italic>Ogataea</italic> spp. HU-11/CBS4732 is so phylogenetically close to NCYC495 that the syntenic regions cover nearly 100% of their genomes. Moreover, HU-11/CBS4732 and NCYC495 share a nucleotide identity of 99.5% through their whole genomes. CBS4732 and NCYC495 can be regarded as the same strain in basic research and industrial applications.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>The present study preliminarily revealed the relationship between CBS4732, NCYC495, and DL-1. Our findings provide new opportunities for in-depth understanding of genome evolution in methylotrophic yeasts and lay the foundations for the industrial applications of <italic>O. polymorpha</italic> CBS4732, NCYC495, DL-1, and their derivative strains. The full-length genome of <italic>O. polymorpha</italic> HU-11/CBS4732 should be included into the NCBI RefSeq database for future studies of <italic>Ogataea</italic> spp.</p>
</sec>
</abstract>
<kwd-group>
<kwd>methylotrophic yeast</kwd>
<kwd><italic>Hansenula polymorpha</italic></kwd>
<kwd>rDNA quadruple</kwd>
<kwd>genome expansion</kwd>
<kwd>long terminal repeat</kwd>
</kwd-group>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content></contract-sponsor>
<counts>
<fig-count count="4"/>
<table-count count="2"/>
<equation-count count="0"/>
<ref-count count="20"/>
<page-count count="11"/>
<word-count count="8082"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>Introduction</title>
<p>Currently, methylotrophic yeasts (e.g., <italic>Pichia pastoris</italic>, <italic>Hansenula polymorpha</italic>, and <italic>Candida boindii</italic>) are subjects of intense genomics studies in basic research and industrial applications. However, genomic research on <italic>Ogataea (Hansenula) polymorpha</italic> trails behind that on <italic>P. pastoris</italic> (<xref ref-type="bibr" rid="B10">Ravin et al., 2013</xref>), although they both are widely used species of methylotrophic yeasts. In the genus <italic>Ogataea</italic>, most research is focused on three basic <italic>O. polymorpha</italic> strains&#x2014;CBS4732 (synonymous to NRRL Y-5445 or ATCC34438), NCYC495 (synonymous to NRRL Y-1798, ATCC14754, or CBS1976), and DL-1 (synonymous to NRRL Y-7560 or ATCC26012). These three strains are of independent geographic and ecological origins: CBS4732 was originally isolated from soil irrigated with waste water from a distillery in Pernambuco, Brazil in 1959 (<xref ref-type="bibr" rid="B8">Morais and Maia, 1959</xref>); NCYC495 is identical to a strain first isolated from spoiled concentrated orange juice in Florida and initially designated as <italic>Hansenula angusta</italic> by <xref ref-type="bibr" rid="B16">Wickerham (1951)</xref>; DL-1 was isolated from soil by <xref ref-type="bibr" rid="B6">Levine and Cooney (1973)</xref>. CBS4732 and its derivatives&#x2014;LR9 and RB11&#x2014;have been developed as genetically engineered strains to produce many heterologous proteins, including enzymes (e.g., feed additive phytase), anticoagulants (e.g., hirudin and saratin), and an efficient vaccine against hepatitis B infection (<xref ref-type="bibr" rid="B7">Massoud et al., 2003</xref>). As a nutritionally deficient mutant derived from CBS4732, the <italic>O. polymorpha</italic> strain HU-11 (CBS4732 <italic>ura3</italic>&#x0394;) (<xref ref-type="bibr" rid="B12">Wang et al., 2007</xref>) is being used for high-yield production of several important proteins or peptides, particularly including recombinant hepatitis B surface antigen (HBsAg) vaccine (<xref ref-type="bibr" rid="B14">Wang H. et al., 2016</xref>) and hirudin (<xref ref-type="bibr" rid="B13">Wang et al., 2011</xref>). HU-11 has the same reference genome as CBS4732 (noted as HU-11/CBS4732), as the only genomic difference between them is a 5-bp insertion caused by frame-shift mutation of its <italic>URA3</italic> gene, which encodes orotidine 5&#x2032;-phosphate decarboxylase. Although CBS4732 and NCYC495 are classified as <italic>O. polymorpha</italic>, and DL-1 is reclassified as <italic>O. parapolymorpha</italic> (<xref ref-type="bibr" rid="B4">Hanson et al., 2017</xref>), the relationship between CBS4732, NCYC495, and DL-1 remains unclear, as the genomic differences between them have not been exactly determined due to lack of their high-quality complete genomes. Thus, knowledge obtained from any of the three strains can&#x2019;t be used to investigate the other two strains.</p>
<p>To facilitate genomic research of yeasts, genome sequences have been increasingly submitted to the Genome-NCBI datasets. Among the genomes of 34 species in the <italic>Ogataea</italic> or <italic>Candida</italic> genus (<xref ref-type="supplementary-material" rid="DS1">Supplementary File 1</xref>), those of NCYC495 and DL-1 have been assembled at chromosome levels. However, the other genomes have been assembled at contig or scaffold levels. Furthermore, the genome sequence of CBS4732 was not available in the Genome-NCBI datasets until this manuscript was drafted. Among the genomes of 33 <italic>Komagataella</italic> (<italic>Pichia</italic>) spp., the genome of the <italic>P. pastoris</italic> strain GS115 is the only genome assembled at the chromosome level. The main problem of the <italic>Ogataea</italic>, <italic>Candida</italic>, and <italic>Pichia</italic> genome data is their incomplete sequences and poor annotations. For example, the rDNA sequence (GenBank: FN392325) of <italic>P. pastoris</italic> GS115 cannot be well aligned to its genome (GenBank assembly: GCA_001708105). Most genome sequences do not contain complete subtelomeric regions and, as a result, subtelomeres are often overlooked in comparative genomics (<xref ref-type="bibr" rid="B2">Brown, 2010</xref>). For example, the genome of DL-1 has been analyzed for better understanding the phylogenetics and molecular basis of <italic>O. polymorpha</italic> (<xref ref-type="bibr" rid="B10">Ravin et al., 2013</xref>); however, it does not contain complete subtelomeric regions due to assembly using short sequences. Another problem of the <italic>Ogataea</italic>, <italic>Candida</italic>, and <italic>Pichia</italic> genome data is that the mitochondrial (mt) genome sequence is not simultaneously released with the corresponding nuclear genome sequences. The only exception is the <italic>O. polymorpha</italic> DL-1 mt genome (RefSeq: NC_014805). Therefore, more complete genome sequences of <italic>Ogataea</italic> spp. are being accomplished to bridge the gap between basic research and industrial applications. For example, a new project has been conducted to provide a high-quality complete genome of DL-1 (GenBank: CP080316-22) based on Nanopore technology.</p>
<p>In the present study, we have assembled the full-length genome of <italic>O. polymorpha</italic> HU-11/CBS4732 using high-depth PacBio and Illumina data, and conducted the annotation and analysis to achieve the following research goals: (1) to provide a high-quality and well-curated reference genome for future studies of <italic>Ogataea</italic> spp.; (2) to determine the relationship between CBS4732, NCYC495, and DL-1; and (3) to discover important genomic features (e.g., high yield) of <italic>Ogataea</italic> spp. for basic research (e.g., synthetic biology) and industrial applications.</p>
</sec>
<sec id="S2" sec-type="results|discussion">
<title>Results and Discussion</title>
<sec id="S2.SS1">
<title>Genome Sequencing, Assembly and Annotation</title>
<p>One 500 bp and one 10 Kbp DNA library were prepared using fresh cells of <italic>O. polymorpha</italic> HU-11 and sequenced on the Illumina HiSeq X Ten and PacBio Sequel platforms, respectively, for <italic>de novo</italic> assembly of a high-quality genome. Firstly, 18,319,084,791 bp cleaned PacBio DNA-seq data were used to assembled the complete genome, except the rDNA region, with an extremely high depth of &#x223C;1800X. However, the draft genome using high-depth PacBio data still contained two types of errors in the low complexity (<xref ref-type="fig" rid="F1">Figure 1A</xref>) and the short tandem repeat (STR) regions, respectively (<xref ref-type="fig" rid="F1">Figure 1B</xref>). Then, 6,628,480,424 bp cleaned Illumina DNA-seq data were used to polish the complete genome of HU-11/CBS4732 to remove the two types of errors. However, Illumina DNA-seq data contained errors in the long (&#x003E;10 G or C) poly(GC) regions. Following this, the poly(GC) regions, polished using Illumina DNA-seq data, were curated using PacBio subreads (<xref ref-type="fig" rid="F1">Figure 1C</xref>). Finally, 5&#x2032; and 3&#x2032; telomeric, subtelomeric, rDNA, Long Terminal Repeat retrotransposons (LTR-rts), low complexity, and other repeat regions were exactly determined and confirmed by human curation (see section &#x201C;Materials and Methods&#x201D;). The complete <italic>O. polymorpha</italic> HU-11/CBS4732 genome is a full-length genome, which is defined to has sequences ending at the 5&#x2032; and 3&#x2032; telomeric sites without gaps and ambiguous nucleotides (<xref ref-type="bibr" rid="B18">Xu et al., 2020</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>Errors in PacBio data and Illumina data. The errors in the low complexity and short tandem repeat (STR) regions can be corrected during the genome polishing using Illumina data, while the errors in the long (&#x003E;10 G or C) poly(GC) regions need be curated using PacBio data after the genome polishing. <bold>(A)</bold> An example to show that the assembled genomes using high-depth PacBio data still contain errors in the low complexity regions. <bold>(B)</bold> An example to show that the assembled genomes using high-depth PacBio data still contain errors in the STR regions. <bold>(C)</bold> An example to show that the genome polishing using Illumina data causes errors in the long poly (GC) regions.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-855666-g001.tif"/>
</fig>
<p>The full-length <italic>O. polymorpha</italic> HU-11/CBS4732 genome includes the complete sequences of all seven chromosomes (<xref ref-type="fig" rid="F2">Figure 2A</xref>), which were named as 1 to 7 from the smallest to the largest, respectively (<xref ref-type="table" rid="T1">Table 1</xref>). As the 5&#x2032; and 3&#x2032; telomeric regions vary in lengths, they were not included into the seven linear chromosomes of HU-11/CBS4732. Analysis of long PacBio subreads revealed that the telomeric regions at 5&#x2032; and 3&#x2032; ends of each chromosome consist of tandem repeats (TRs) [ACCCCGCC]<sub><italic>n</italic></sub> and [GGCGGGGT]<sub><italic>n</italic></sub> (<italic>n</italic> is the copy number) with average lengths of 166 bp and 168 bp (&#x223C;20 copy numbers), respectively. Finally, we released the data of the nuclear genome (GenBank: CP073033-39) with a summed sequence length of 9.1 Mbp and the mt genome (GenBank: CP073040) with a sequence length of 59,496 bp (<xref ref-type="table" rid="T1">Table 1</xref>). For the submission to the GenBank database, the sequence of circular mt genome (<xref ref-type="fig" rid="F2">Figure 2B</xref>) was anticlockwise linearized, starting at the first nt of large subunit ribosomal RNA (rrnL). The complete genome sequence of the <italic>O. polymorpha</italic> strain CBS4732 ura3&#x0394; (named HU-11) is available at the NCBI GenBank database, which need be included into the NCBI RefSeq database to facilitate future studies on <italic>O. polymorpha</italic> CBS4732 and its derivatives- LR9, RB11, and HU-11.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Full-length genome of <italic>Ogataea polymorpha</italic> HU-11/CBS4732. <bold>(A)</bold> The full-length <italic>O. polymorpha</italic> HU-11/CBS4732 genome includes the complete sequences of seven linear chromosomes, which were named as 1&#x2013;7 from the smallest to the largest. The 5&#x2032; and 3&#x2032; telomeric regions were not included. The minimum, Q<sub>90</sub>, Q<sub>75</sub>, Q<sub>50</sub>, Q<sub>25</sub>, Q<sub>10</sub>, and maximum of GC contents (%) are 0.08, 0.408 0.436, 0.472, 0.514, 0.554, and 0.732. The GC contents (%) were calculated by 500-bp sliding windows and then trimmed between Q<sub>10</sub> and Q<sub>90</sub> for plotting the heatmaps. Long terminal repeat retrotransposons (LTR-rts) and markers genes are indicated by arrows (red and green colors represent sense and antisense strands) in the chromosomes. Markers genes of chromosome 1&#x2013;7 include URA3 (encoding orotidine 5&#x2032;-phosphate decarboxylase), rDNA, HARS (Hansenula autonomously replicating sequence), FGH (S-formylglutathione hydrolase), MOX (methanol oxidase), FDH (Formate dehydrogenase) and TERT (telomerase reverse transcriptase), respectively. <bold>(B)</bold> For the data submission to the GenBank database, the genome sequence of circular mitochondrion was anticlockwise linearized, starting at the first nt (indicated by a red arrow) of rrnL, which may include a part of the control region. SSU, small subunit; RPS3, ribosomal protein S3; rrnL, large subunit ribosomal RNA; rrnS, small subunit ribosomal RNA.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-855666-g002.tif"/>
</fig>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p>Genomes of three basic <italic>Ogataea polymorpha</italic> strains.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Chromosome</td>
<td valign="top" align="left">CBS4732/HU-11</td>
<td valign="top" align="left">NCYC495</td>
<td valign="top" align="left">DL-1</td>
<td valign="top" align="center">HU-11 Size (bp)</td>
<td valign="top" align="center">Marker</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Chr1</td>
<td valign="top" align="left">CP073033</td>
<td valign="top" align="left">NW_017264703</td>
<td valign="top" align="left">NC_027865</td>
<td valign="top" align="center">1,000,895</td>
<td valign="top" align="center">URA3</td>
</tr>
<tr>
<td valign="top" align="left">Chr2</td>
<td valign="top" align="left">CP073034</td>
<td valign="top" align="left">NW_017264704</td>
<td valign="top" align="left">NC_027866</td>
<td valign="top" align="center">1,125,341</td>
<td valign="top" align="center">rDNA</td>
</tr>
<tr>
<td valign="top" align="left">Chr3</td>
<td valign="top" align="left">CP073035</td>
<td valign="top" align="left">NW_017264702</td>
<td valign="top" align="left">NC_027864</td>
<td valign="top" align="center">1,265,401</td>
<td valign="top" align="center">HARS</td>
</tr>
<tr>
<td valign="top" align="left">Chr4</td>
<td valign="top" align="left">CP073036</td>
<td valign="top" align="left">NW_017264701</td>
<td valign="top" align="left">NC_027863</td>
<td valign="top" align="center">1,315,956</td>
<td valign="top" align="center">FGH</td>
</tr>
<tr>
<td valign="top" align="left">Chr5</td>
<td valign="top" align="left">CP073037</td>
<td valign="top" align="left">NW_017264700</td>
<td valign="top" align="left">NC_027862</td>
<td valign="top" align="center">1,357,435</td>
<td valign="top" align="center">MOX</td>
</tr>
<tr>
<td valign="top" align="left">Chr6</td>
<td valign="top" align="left">CP073038</td>
<td valign="top" align="left">NW_017264698</td>
<td valign="top" align="left">NC_027860</td>
<td valign="top" align="center">1,513,391</td>
<td valign="top" align="center">FDH</td>
</tr>
<tr>
<td valign="top" align="left">Chr7</td>
<td valign="top" align="left">CP073039</td>
<td valign="top" align="left">NW_017264699</td>
<td valign="top" align="left">NC_027861</td>
<td valign="top" align="center">1,525,912</td>
<td valign="top" align="center">TERT</td>
</tr>
<tr>
<td valign="top" align="left">ChrM</td>
<td valign="top" align="left">CP073040</td>
<td valign="top" align="left">NA</td>
<td valign="top" align="left">NC_014805</td>
<td valign="top" align="center">59,496</td>
<td valign="top" align="center">COIII</td>
</tr>
<tr>
<td valign="top" align="left">Total (Mbp)</td>
<td valign="top" align="left">9.1</td>
<td valign="top" align="left">8.97</td>
<td valign="top" align="left">8.87</td>
<td/>
<td/>
</tr>
<tr>
<td valign="top" align="left">GC%</td>
<td valign="top" align="left">47.76</td>
<td valign="top" align="left">47.86</td>
<td valign="top" align="left">47.83</td>
<td/>
<td/>
</tr>
<tr>
<td valign="top" align="left">Gene<xref ref-type="table-fn" rid="t1fn1"><sup>#</sup></xref></td>
<td valign="top" align="left">5453<xref ref-type="table-fn" rid="t1fn1">&#x002A;</xref></td>
<td valign="top" align="left">5454<xref ref-type="table-fn" rid="t1fn1">&#x002A;</xref></td>
<td valign="top" align="left">5309<xref ref-type="table-fn" rid="t1fn1">&#x002A;</xref></td>
<td/>
<td/>
</tr>
<tr>
<td valign="top" align="left">tRNA<xref ref-type="table-fn" rid="t1fn1"><sup>#</sup></xref></td>
<td valign="top" align="left">80</td>
<td valign="top" align="left">80</td>
<td valign="top" align="left">80</td>
<td/>
<td/>
</tr>
<tr>
<td valign="top" align="left">rRNA<xref ref-type="table-fn" rid="t1fn1"><sup>#</sup></xref></td>
<td valign="top" align="left">4 &#x00D7; 20</td>
<td valign="top" align="left">4 &#x00D7; 6</td>
<td valign="top" align="left">4 &#x00D7; 25</td>
<td/>
<td/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="t1fn1"><p><italic>HU-11 has the same reference genome as CBS4732 (noted as HU-11/CBS4732), because the only genomic difference between them is a 5-bp insertion. <sup>#</sup>The numbers of mitochondrial genes in the LTR-rts were not counted. &#x002A;The genome sequences of NCYC495 and DL-1 with annotations were corrected, so the numbers of protein-coding genes (&#x003E;150 bp) are different from their original records. The full-length O. polymorpha HU-11/CBS4732 genome includes the complete sequences of all seven chromosomes, which were named as 1&#x2013;7 from the smallest to the largest, respectively. As the 5&#x2032; and 3&#x2032; telomeric regions vary in lengths, they were not included into the seven linear chromosomes of HU-11/CBS4732. The accession numbers of <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NCYC495">NCYC495</ext-link> and DL-1 were mapped to the chromosome numbers of HU-11, according to the marker genes of seven chromosomes. They are: URA3 (encoding orotidine 5&#x2032;-phosphate decarboxylase), rDNA, HARS(Hansenula autonomously replicating sequence), FGH (S-formylglutathione hydrolase), MOX(methanol oxidase), FDH (Formate dehydrogenase), TERT (telomerase reverse transcriptase) and COIII (cytochrome c oxidase subunit 3).</italic></p></fn>
</table-wrap-foot>
</table-wrap>
<p>The HU-11/CBS4732 (nuclear) genome has a summed length of 9.1 Mbp that is close to the estimated length of the <italic>O. polymorpha</italic> DL-1 genome (<xref ref-type="bibr" rid="B10">Ravin et al., 2013</xref>), while the NCYC495 (RefSeq: NW_017264698-704) and DL-1 genomes (RefSeq: NC_027860-66) have shorter lengths of 8.97 and 8.87 Mbp, respectively (<xref ref-type="table" rid="T1">Table 1</xref>), as both of them are incomplete and have many errors at 5&#x2032; and 3&#x2032; ends of their chromosomes (<xref ref-type="supplementary-material" rid="DS1">Supplementary File 1</xref>). The GC contents of the HU-11, NCYC495, and DL-1 genomes are very close (&#x223C;48%). Syntenic comparison (see section &#x201C;Materials and Methods&#x201D;) revealed that <italic>O. polymorpha</italic> HU-11/CBS4732 is so phylogenetically close to NCYC495 that the syntenic regions cover nearly 100% of their genomes, however, HU-11/CBS4732 and NCYC495 are significantly distinct from DL-1. Then, we discovered large duplicated segments (LDSs) in the subtelomeric regions and exactly determined all the structural variations (SVs) between HU-11/CBS4732 and NCYC495 (Detailed later). We improved the annotations of NCYC495 protein-coding genes (&#x003E;150 bp) using a high quality RNA-seq data of NCYC495 (NCBI SRA: SRP124832), and then HU-11/CBS4732 and DL-1 genes by gene ID mapping (<xref ref-type="supplementary-material" rid="DS2">Supplementary File 2</xref>). As a result, we updated the annotations (<xref ref-type="table" rid="T1">Table 1</xref>) of: (1) 5,453 protein-coding genes of HU-11/CBS4732, including 5,021 single-exon genes, and 432 multi-exon genes; (2) 5,454 protein-coding genes of NCYC495, including 5,022 single-exon genes, and 432 multi-exon genes; (3) 5,309 protein-coding genes of DL-1, including 4,843 single-exon genes, and 464 multi-exon genes; and (4) 80 identical tRNA genes of HU-11/CBS4732, NCYC495, and DL-1. For 432 multi-exon genes of HU-11/CBS4732, only the longest splicing isoforms of them were annotated. Then, 5,453 CDSs (<xref ref-type="supplementary-material" rid="DS3">Supplementary File 3</xref>) were identified from 5,453 genes of HU-11/CBS4732. Furthermore, only 13 single-exon genes of HU-11/CBS4732 or NCYC495 were not detected to be expressed using the RNA-seq data SRP124832, while 87 genes (data not shown) of DL-1 have been reported to be not expressed in the previous study (<xref ref-type="bibr" rid="B10">Ravin et al., 2013</xref>).</p>
</sec>
<sec id="S2.SS2">
<title>Organization of rDNA Genes</title>
<p>An rDNA TR of HU-11/CBS4732, NCYC495, or DL-1 encodes 5S, 18S, 5.8S, and 25S rRNAs (named as quadruple in the present study), with a length of 8,145 bp (<xref ref-type="supplementary-material" rid="DS1">Supplementary File 1</xref>). As the largest TR region (&#x223C;162 Kbp) in the HU-11/CBS4732 genome, the only rDNA locus is in chromosome 2. The organization of rDNA TRs is conserved in the <italic>Ogataea</italic> genus. TRs of HU-11/CBS4732 and NCYC495 rDNAs share a very high nucleotide (nt) sequence identity of 99.5% (8,115/8,152), while TRs of HU-11/CBS4732 and DL-1 rDNAs share a comparatively low nt sequence identity of 97% (7,530/7,765). The major difference of rDNAs among <italic>Ogataea</italic> spp. lie in their copy numbers. The copy number of rDNA TRs was estimated as 20 in the HU-11/CBS4732 genome (<xref ref-type="fig" rid="F3">Figure 3A</xref>), while that was estimated as 6 and 25 in NCYC495 and DL-1, respectively (<xref ref-type="bibr" rid="B10">Ravin et al., 2013</xref>). An rDNA TR of <italic>Saccharomyces cerevisiae</italic> also contains 5S, 18S, 5.8S, and 25S rDNAs as a quadruple, repeating two times in chromosome 7 of its genome (<xref ref-type="fig" rid="F3">Figure 3B</xref>), however, four other 5S rDNAs are located separately away from the rDNA quadruples in <italic>S. cerevisiae</italic>. Different from <italic>O. polymorpha</italic> or <italic>S. cerevisiae</italic> with only one rDNA locus, <italic>Pichia pastoris</italic> GS115 carries several rDNA loci, which are interspersed in three of its four chromosomes. Since the genome of <italic>P. pastoris</italic> GS115 (GenBank assembly: GCA_001708105) is incomplete and poorly annotated, we estimated the copy number of its rDNAs as three or more. In animals, rDNAs encoding 18S, 5.8S, and 28S rRNAs are also organized in TRs and transcribed into a single RNA precursor by RNA polymerase I. As a typical example, human has approximately 200&#x2013;600 rDNA copies (<xref ref-type="fig" rid="F3">Figure 3C</xref>) distributed in short arms of the five acrocentric chromosomes (chromosomes 13, 14, 15, 21, and 22) (<xref ref-type="bibr" rid="B1">Agrawal and Ganley, 2018</xref>). In prokaryotic cells, 5S, 23S, and 16S rRNA genes are typically organized as a co-transcribed operon. There may be one or more copies of the operon dispersed in the genome and the copy numbers typically range from 1 to 15 in bacteria. As a typical example, <italic>Ochrobactrum quorumnocens</italic> has four copies of co-transcribed operons at two rDNA loci (<xref ref-type="fig" rid="F3">Figure 3D</xref>) in chromosome 1 (GenBank: CP022603) and 2 (GenBank: CP022604). Compared to <italic>S. cerevisiae</italic>, human, and <italic>O. quorumnocens</italic> rDNAs (<xref ref-type="fig" rid="F3">Figures 3B&#x2013;D</xref>), the rDNAs of <italic>O. polymorpha</italic> are more closely organized. By the organization in TR, 20 copies of <italic>O. polymorpha</italic> rDNA quadruples are transcribed in a large (&#x003E;162 Kbp) co-transcribed operon, suggesting that their transcription can be regulated with higher efficiency. This genomic feature may contribute to the high yield characteristics of <italic>O. polymorpha</italic>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>Organization of rDNA genes in yeasts, human and bacteria. <bold>(A)</bold> The only rDNA locus is in chromosome 2 (GenBank: CP073034) of the <italic>Ogataea polymorpha</italic> HU-11/CBS4732 genome, containing 20 copies of tandem repeats (TRs). Here only six copies of TRs are shown. <bold>(B)</bold> An rDNA TR of Saccharomyces cerevisiae also contains 5S, 18S, 5.8S, and 25S rDNAs as a quadruple, repeating 2 times in chromosome 7 of its genome. Four other 5S rDNAs are located separately away from the rDNA quadruples in <italic>S. cerevisiae</italic>. <bold>(C)</bold> Each human rDNA unit has an rRNA region and an intergenic spacer (IGS). Here only eight units are shown. ITS, internal transcribed spacer; ETS, external transcribed spacers. <bold>(D)</bold> There are four copies of rRNA regions at two rDNA loci in chromosome 1 (GenBank: CP022603) and 2 (GenBank: CP022604) of the <italic>Ochrobactrum quorumnocens</italic> genome. &#x002A;Indicate the 5S rRNA.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-855666-g003.tif"/>
</fig>
<p>Besides the high similarity of genomic organization, the rDNAs of <italic>O. polymorpha</italic> HU-11/CBS4732 and <italic>S. cerevisiae</italic> share high nt sequence identities of 95.3% (1720/1805), 96.2% (152/158), 92% (3,111/3,381), and 96.7% (117/121) for 18S, 5.8S, 25S, and 5S rDNAs, respectively. These identities are little lower than those of <italic>O. polymorpha</italic> NCYC495 and DL-1, indicated that the rDNA genes are more conservative than the protein-coding genes. Therefore, rDNA is an important genomic feature for the detection, identification, classification and phylogenetic analysis of <italic>Ogataea</italic> spp. Unexpectedly, we found that the rDNAs (GenBank: FN392325) of <italic>O. polymorpha</italic> HU-11/CBS4732 and <italic>P. pastoris</italic> GS115 have nt sequence identities of 87.3% (1477/1691), 80% (84/105), and 80.5% (2,073/2,576) for 18S, 5.8S, and 25S rDNAs, respectively, which are much lower than those of <italic>S. cerevisiae</italic>. These results are not consistent with those of a previous study (<xref ref-type="bibr" rid="B10">Ravin et al., 2013</xref>), in which phylogenetic analysis using 153 protein-coding genes showed that <italic>O. polymorpha</italic> and <italic>Pichia pastoris</italic> GS115 are members of a clade that is distinct from the one that <italic>S. cerevisiae</italic> belongs to. Based on the previous study, HU-11/CBS4732 is phylogenetically closer to <italic>P. pastoris</italic> GS115 than <italic>S. cerevisiae</italic>. The nt sequence identities of rDNAs between HU-11/CBS4732 and <italic>P. pastoris</italic> GS115 are supposed to be higher than those between HU-11/CBS4732 and <italic>S. cerevisiae</italic>.</p>
</sec>
<sec id="S2.SS3">
<title>Long Terminal Repeat Retrotransposons</title>
<p>LTRs with lengths of 322 bp were discovered in all seven chromosomes of HU-11/CBS4732. These LTRs with the low GC content of 29% (94/322) are flanked by TCTTG and CAACA at their 5&#x2032; and 3&#x2032; ends (<xref ref-type="fig" rid="F4">Figure 4A</xref>). In HU-11/CBS4732, a total of 14 LTRs are present in seven copies of intact LTR-rts (<xref ref-type="supplementary-material" rid="DS1">Supplementary File 1</xref>), which were identified as components of Tpa5 LTR-rts (GenBank: AJ439553) from <italic>Pichia angusta</italic> CBS4732 (a former name of <italic>O. polymorpha</italic> CBS4732) in a previous study (<xref ref-type="bibr" rid="B9">Neuveglise et al., 2002</xref>). A LTR-rt consists of 5&#x2032; LTR, 3&#x2032; LTR, and a single open reading frame (ORF) encoding a putative polyprotein (<xref ref-type="fig" rid="F4">Figure 4A</xref>). This polyprotein, if translated, can be processed into truncated Gag (GAG), protease (PR), integrase (IN), reverse transcriptase (RT), and RNase H (RH). Based on the gene order (PR, IN, RT, and RH), the LTR-rts of HU-11/CBS4732 were classified into the Ty5 type of the Ty1/copia group (Ty1, 2, 4, and 5 types) (<xref ref-type="bibr" rid="B1">Agrawal and Ganley, 2018</xref>).</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>Long terminal repeat (LTR) retrotransposons, large duplicated segments and structural variations. <bold>(A)</bold> NCYC495 and HU-11/CBS4732 share identical 322-bp LTRs, which are flanked by TCTTG and CAACA at their 5&#x2032; and 3&#x2032; ends. Three of seven LTR-rts of HU-11/CBS4732 have no homologs in the NCYC495 genome due to misassembly. A LTR-rt consists of 5&#x2032; LTR, 3&#x2032; LTR and a single open reading frame (ORF) encoding a putative polyprotein. This polyprotein, if translated, can be processed into trunculated gag (GAG), protease (PR), integrase (IN), reverse transcriptase (RT) and RNase H (RH). <bold>(B)</bold> Chr1, 2, 4, 5, and 7 represent the chromosomes (GenBank: CP073033, 34, 36, 37, and 39) of the HU-11/CBS4732 genome. Three large duplicated segments (LDSs) named LDS1 (in yellow color), 2 (in green color) and 3 (in blue color) are supposed to be included in both NCYC495 and HU-11/CBS4732 genomes. However, LDS2 and a 14,090 bp part of LDS1 (indicated by black slashs) were not assembled into chromosome 2 of the NCYC495 genome. The genomic differences between the HU-11/CBS4732 and NCYC495 only include three structural variations (SVs), named SV1, 2 and 3 (in red color). The three SVs are located in three large syntenic regions (SRs) of HU-11/CBS4732, NCYC495, and DL-1 genomes with very high nt sequence identities, named SR1, 2 and 3. SV4 is a 22.6-Kb DNA region which functions in the determination of the yeast mating-type (MAT). <bold>(C)</bold> The graphic elements used to represent the genomes and genes were originally used in the previous study (<xref ref-type="bibr" rid="B4">Hanson et al., 2017</xref>). The HU-11 genome (GenBank: CP073033-40) contains a 22.6-Kb MAT region where MAT&#x03B1; can be transcribed, while the NCYC495 genome (RefSeq: NW_017264698-704) contains an identical 22.6-Kb MAT region where MATa can be transcribed.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmicb-13-855666-g004.tif"/>
</fig>
<p>With the length corrected from 4,883 bp to 4,882 bp, a sequence (GenBank: AJ439558) was used as the reference of Tpa5 LTR-rts to search for homologs. The results confirmed that HU-11/CBS4732 is phylogenetically closest to NCYC495 and they share identical LTRs. However, the 322-bp LTRs of HU-11/CBS4732 and NCYC495 are quite distinct from the 282-bp LTRs of DL-1, which were reported as 290-bp solo LTRs in the previous study (<xref ref-type="bibr" rid="B10">Ravin et al., 2013</xref>). In addition, the amino acid (aa) sequences of the polyprotein with the length of 1417 aa in HU-11/CBS4732 and NCYC495 LTR-rts are distinct from those in DL-1. Based on the records in the UniProt Knowledgebase (UniProtKB), <italic>O. polymorpha</italic> strains DL-1, ATCC26012, BCRC20466, JCM22074, and NRRL Y-7560 have nearly the same aa sequences (UniProt: W1QI12) of the polyprotein. The above results suggest that the LTR-rt is another important genomic feature for the detection, identification, classification and phylogenetic analysis of <italic>Ogataea</italic> spp. Using RNA-seq data of NCYC495 (SRA: SRP124832), we discovered that the genes encoding the polyproteins in the LTR-rts of <italic>O. polymorpha</italic> are transcribed. If these putative polyproteins can be translated merits further studies.</p>
<p>In the previous study, 50,000 fragments of 13 <italic>Hemiascomycetes</italic> species were used to identify LTR-rts. However, the analysis was probably biased as it was based on only random sequences (approximately 1 kb on average) without the related genome information (<xref ref-type="bibr" rid="B9">Neuveglise et al., 2002</xref>). In the present study, seven copies of intact LTR-rts (As described above) were precisely located in the HU-11/CBS4732 genome (<xref ref-type="fig" rid="F2">Figure 2A</xref>), with five in the sense strands of chromosome 1, 2, 3, 6, and 7 (named LTR-rt1, 2, 3, 6, and 7) and two in the antisense strands of chromosome 5 and 7 (named LTR-rt5R and 7R). LTR-rt1, 3, and 6 share very high nt identities of 99.9% with each other. LTR-rt1 or 3 contains a single ORF encoding a polyprotein with the same aa sequence, while LTR-rt6 contains a single ORF with a 42-bp insertion (encoding RSSLFDVPCSPTVD), compared to LTR-rt1 and 3. LTR-rt2, 5R, 7, and 7R contain several single nucleotide polymorphisms (SNPs), small insertions and deletions (InDels), which break the single ORFs into several ORFs. Genome comparison revealed that the homologs of LTR-rt2, 3, and 5R in HU-11/CBS4732 are present in the NCYC495 genome with very high nt identities of 99.9%, while the homologs of LTR-rt1, 7, and 7R, however, are absent in the NCYC495 genome. Further analysis determined that their absence in the NCYC495 genome resulted from misassembly.</p>
</sec>
<sec id="S2.SS4">
<title>Large Duplicated Segments in Subtelomeric Regions</title>
<p>Syntenic comparison revealed that <italic>O. polymorpha</italic> HU-11/CBS4732 is so phylogenetically close to NCYC495 that the syntenic regions cover nearly 100% of their genomes, however, HU-11/CBS4732 and NCYC495 are significantly distinct from DL-1. HU-11/CBS4732 and NCYC495 share a nt identity of 99.5% through their whole genomes, including the rDNA regions and LTR-rts. In contrast, HU-11/CBS4732 and DL-1 share a comparatively low nt identity (&#x003C;95%). Subsequently, the detection of structural variations (SVs) was performed between the HU-11/CBS4732 and NCYC495 genomes. Further analysis revealed that most of detected SVs are errors in the assembly of NCYC495 genome (<xref ref-type="fig" rid="F4">Figure 4B</xref>), particularly including: (1) LTR-rt1, 7, and 7R (absent in NCYC495) which need be included in the NCYC495 genome; (2) two large deletions (absent in NCYC495) which need be added at 5&#x2032; and 3&#x2032; ends of chromosome 2 of NCYC495; and (3) an over-assembled large segment (absent in HU-11/CBS4732) at 3&#x2032; end of chromosome 6 (NW_017264698:1509870-1541475), which need be removed from chromosome 6 of NCYC495. Before the correction of above errors, (1), (2), and (3) were confirmed by long PacBio subreads. Particularly, (3) was confirmed as the telomeric region at the 3&#x2032; end of chromosome 6, which was wrongly assembled as the junction region at 5&#x2032; end of the over-assembled segment in the previous study and the reason is that the copy number of TRs [GGCGGGGT]<sub><italic>n</italic></sub> (NW_017264698:1509840-1509869) in this telomeric region was under-estimated using short sequencing data.</p>
<p>Two large deletions [one type of SV (<xref ref-type="bibr" rid="B19">Zhang et al., 2016</xref>)] in the NCYC495 genome (As described above) are &#x201C;false-positive&#x201D; SVs caused by the misassembly of LDSs (<xref ref-type="supplementary-material" rid="DS1">Supplementary File 1</xref>) in the subtelomeric regions. In contrast, all LDSs were correctly assembled in the HU-11/CBS4732 genome. Using long (&#x003E;30 Kb) PacBio subreads, human curation (see section &#x201C;Materials and Methods&#x201D;) was performed to verify the locations of the LDSs, particularly three LDSs named LDS1, 2 and 3 with lengths of &#x223C;27,850, &#x223C;5,100, and &#x223C;2,500 bp, respectively (<xref ref-type="fig" rid="F4">Figure 4B</xref>): (1) LDS1 and its paralog are present at 3&#x2032; ends of chromosome 2 and 1 in the HU-11/CBS4732 genome, respectively, while the paralog of LDS1 was correctly assembled into 3&#x2032; end of chromosome 1 of NCYC495, but a 14,090 bp part of LDS1 was not assembled into 3&#x2032; end of chromosome 2, which corresponds to a large deletion; and (2) LDS2 and its paralog are present at 5&#x2032; ends of both chromosomes 2 and 5 in the HU-11/CBS4732 genome, while the paralog of LDS2 was correctly assembled into 5&#x2032; end of chromosome 5, but LDS2 was not assembled into 5&#x2032; end of chromosome 2 of NCYC495, which corresponds to the other large deletion. Different from LDS1 and LDS2, LDS3 and its paralog were correctly assembled in the NCYC495 genome. LDS3 is downstream of LDS2 in chromosome 2, and the paralog of LDS3 is present at 5&#x2032; end of chromosome 7 (<xref ref-type="fig" rid="F4">Figure 4B</xref>). LDS1 and 2 had not been discovered before the present study, mainly because they are nearly identical to their paralogs. Particularly, there are only four mismatches and one 1-bp gap between LDS1 and its paralog. As an important finding, telomeric TR-like sequences [ACCCCGCC]<sub><italic>n</italic></sub> or [ACCCGCC]<sub><italic>n</italic></sub> (n &#x003E; 2) were discovered at 3&#x2032; ends of LDS2 and its paralog (located on both chromosomes 2 and 5), and at 3&#x2032; end of LDS3&#x2032;s paralog (located in chromosome 7). The discovery of these sequences (<xref ref-type="supplementary-material" rid="DS1">Supplementary File 1</xref>) indicated that these LDSs were integrated at 5&#x2032; ends of the old telomeric regions by their 3&#x2032; ends.</p>
</sec>
<sec id="S2.SS5">
<title>Structural Variations Between HU-11/CBS4732 and NCYC495</title>
<p>The major errors in the assembly of NCYC495 genome (RefSeq: NW_017264698-704) include incomplete rDNAs, misassembly of LTR-rts, under-assembly of LDSs and over-assembly of a large segment. After we corrected these errors using the syntenic regions, only four SVs between HU-11/CBS4732 and NCYC495 remained and were named as SV1, SV2, SV3, and SV4 (<xref ref-type="fig" rid="F4">Figure 4B</xref>). Further analysis revealed that SV4 is a very special &#x201C;false-positive&#x201D; SV. Actually, SV4 is a 22.6-Kb DNA region (<xref ref-type="fig" rid="F4">Figure 4C</xref>) which functions in the determination of the yeast mating-type (MAT). According to previous studies (<xref ref-type="bibr" rid="B4">Hanson et al., 2017</xref>), yeast mating generally occurs between two haploid cells with opposite genotypes (MATa and MAT&#x03B1;) at this locus, to form a diploid zygote (MATa/&#x03B1;). <italic>Ogataea</italic> spp. contain both a MATa locus and a MAT&#x03B1; locus in chromosome 5, approximately 19 Kb apart (<xref ref-type="fig" rid="F4">Figure 4C</xref>). The two MAT loci are beside two copies of an identical 2-Kb DNA sequence, which form two inverted repeats (IRs). During MAT switching, the two copies of the IR recombine, inverting the orientation of the 19-Kb region relative to the rest of the chromosome. The MAT locus proximal to the centromere is not transcribed, probably due to silencing by centromeric heterochromatin, whereas the distal MAT locus is transcribed. The HU-11 genome contains a 22.6-Kb MAT region (MAT-HU11) where MAT&#x03B1; can be transcribed, while the NCYC495 genome contains a 22.6-Kb MAT region (MAT-NCYC495) where MATa can be transcribed. There is only one 1-bp gap between the large segments MAT-HU11 and the reverse-complimentary sequence of MAT-NCYC495 (<xref ref-type="supplementary-material" rid="DS1">Supplementary File 1</xref>). Therefore, the HU-11 (GenBank: CP073033-40) and NCYC495 (RefSeq: NW_017264698-704) genomes represent genomes of <italic>O. polymorpha</italic> MAT&#x03B1; and MATa cells, respectively. MAT regions can&#x2019;t be used as a genomic marker to characterize different <italic>O. polymorpha</italic> strains, as MAT switching can be induced without environmental signals. For example, we found that MAT switching of HU-11 even occur under optimal growth conditions (pH = 5.5, <italic>T</italic> = 32&#x00B0;C), although the frequency is extremely low (1/264).</p>
<p>Only three SVs (SV1, SV2, and SV3) are true-positive. SV1 and SV2 are present at 5&#x2032; and 3&#x2032; ends of chromosome 4, respectively, while the location of SV3 is close to 5&#x2032; ends of chromosome 5 (<xref ref-type="fig" rid="F4">Figure 4C</xref>). Five sequences involved in these three SVs are SV1-CBS4732 and SV2-CBS4732 in the HU-11/CBS4732 genome and SV1-NCYC495, SV2-NCYC495, and SV3-NCYC495 in the NCYC495 genome (<xref ref-type="supplementary-material" rid="DS1">Supplementary File 1</xref>). These five sequences can be used to identify <italic>O. polymorpha</italic> strains, particularly CBS4732, NCYC495, and their derivative strains. Blasting the five sequences to the NCBI NT database, we found that SV1-CBS4732 and SV2-NCYC495 are nearly identical (&#x003E;98%) to their orthologs at 5&#x2032; and 3&#x2032; ends of chromosome 4 in the DL-1 genome (GenBank: CP080319), respectively, while SV1-NCYC495 and SV2-CBS4732 have no homologs in chromosome 4 of DL-1. As an insertion into the NCYC495 genome, SV3-NCYC495 has a very high nt sequence identity (&#x003E;91%) to its homolog in the DL-1 genome. Further analysis showed that the three SVs are located in three large syntenic regions (SRs) of HU-11/CBS4732, NCYC495, and DL-1 genomes with very high nt sequence identities (&#x003E;95%). Three SRs are: (1) SR1 with a length of 161,844 bp at 5&#x2032; ends of chromosome 4; (2) SR2 with a length of 81,748 bp at 3&#x2032; ends of chromosome 4; and (3) SR3 with a length of 11,087 bp close to 5&#x2032; ends of chromosome 5. The above results revealed that many recombination events occurred in chromosome 4 and 5 of CBS4732 and NCYC495&#x2032; ancestors after their divergence, particularly: (1) recombination events occurred at 5&#x2032; end of chromosome 4 of the NCYC495&#x2032; ancestor, resulting in the acquisition of SV1-NCYC495; (2) recombination events occurred at 3&#x2032; end of chromosome 4 of the CBS4732&#x2032; ancestor, resulting in the acquisition of SV2-CBS4732; (3) recombination events occurred close to 5&#x2032; end of chromosome 5 of the CBS4732&#x2032; ancestor, resulting in the loss of SV3-CBS4732 (the hypothetical homolog of SV3-NCYC495).</p>
<p>Only a few genes (predicted as 25) were involved in the three SVs between HU-11/CBS4732 and NCYC495 (<xref ref-type="table" rid="T2">Table 2</xref>). Among the 25 genes (<xref ref-type="table" rid="T2">Table 2</xref>), 10 genes (OGAPO_03766-67 and OGAPO_00003-08) of HU-11/CBS4732 and 11 genes (OGAPODRAFT_24127, 16381, 24129, 12876, 16382, 76936, 16706, 37951, 93168, 75778, and 75779) of NCYC495 have no orthologs in NCYC495 and HU-11/CBS4732, respectively and two genes (OGAPODRAFT_13497 and 15973) in NCYC495 were significantly changed into two other ones (OGAPO_13497 and 15973) in HU-11/CBS4732, resulting in different aa sequences. Blasting the proteins encoded by these 25 genes (<xref ref-type="supplementary-material" rid="DS1">Supplementary File 1</xref>) to the UniProt database, we found that the proteins encoded by five genes (OGAPODRAFT_24127, 16381, 24129, 12876, and 16382) in SV1-NCYC495 have the highest sequence similarities to their homologs encoded by six genes (HPODL_02401, 02402, 02403, 02404, 02405, and 02398) at 3&#x2032; end of chromosome 6 (RefSeq: NC_027860) in the DL-1 genome. The above results suggest that SV1-NCYC495 from chromosome 6 of NCYC495&#x2032; ancestor was acquired by chromosome 4 of NCYC495 <italic>via</italic> translocation. The proteins encoded by six genes (OGAPO_00003-08) in a major part (more than 80%) of SV2-CBS4732 have no homologs in <italic>Ogataea polymorpha</italic> NCYC495 or DL-1, but have the highest sequence similarities to their homologs in <italic>O. thermophila</italic>, followed by <italic>O. philodendri</italic> and <italic>O. haglerorum</italic>. These six proteins also have homologs in the <italic>Cyberlindnera jadinii</italic> strain NRRL Y-1542. Furthermore, we found that the proteins encoded by two genes (OGAPO_00001-02) in the minor part of SV2-CBS4732 (from chromosome 4) have the highest sequence similarities to their homologs encoded by genes in other chromosomes. These findings revealed more combination events occurred between chromosome 4 and other chromosomes within the genome of NCYC495&#x2032; ancestor or CBS4732&#x2032; ancestor.</p>
<table-wrap position="float" id="T2">
<label>TABLE 2</label>
<caption><p>Twenty five different genes between HU-11/CBS4732 and NCYC495.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Gene</td>
<td valign="top" align="center">Locus</td>
<td valign="top" align="left">Orthologs (CBS4732/NCYC495/DL-1)</td>
<td valign="top" align="left">Function</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">OGAPO_03767</td>
<td valign="top" align="center">SV1- CBS4732</td>
<td valign="top" align="left">OGAPO_03767/-/HPODL_03767</td>
<td valign="top" align="left">12-oxophytodienoate reductase 3 (OPR3)</td>
</tr>
<tr>
<td valign="top" align="left">OGAPO_03766</td>
<td valign="top" align="center">SV1- CBS4732</td>
<td valign="top" align="left">OGAPO_03766/-/HPODL_03766</td>
<td valign="top" align="left">Aminotriazole resistance protein</td>
</tr>
<tr>
<td valign="top" align="left">OGAPODRAFT_24127</td>
<td valign="top" align="center">SV1-NCYC495</td>
<td valign="top" align="left">-/OGAPODRAFT_24127/-</td>
<td valign="top" align="left">Myo-inositol transporter 1</td>
</tr>
<tr>
<td valign="top" align="left">OGAPODRAFT_16381</td>
<td valign="top" align="center">SV1-NCYC495</td>
<td valign="top" align="left">-/OGAPODRAFT_16381/-</td>
<td valign="top" align="left">Aldo keto reductase (ARK)</td>
</tr>
<tr>
<td valign="top" align="left">OGAPODRAFT_24129</td>
<td valign="top" align="center">SV1-NCYC495</td>
<td valign="top" align="left">-/OGAPODRAFT_24129/-</td>
<td valign="top" align="left">Amidase</td>
</tr>
<tr>
<td valign="top" align="left">OGAPODRAFT_12876</td>
<td valign="top" align="center">SV1-NCYC495</td>
<td valign="top" align="left">-/OGAPODRAFT_12876/-</td>
<td valign="top" align="left">MFS transporter</td>
</tr>
<tr>
<td valign="top" align="left">OGAPODRAFT_16382</td>
<td valign="top" align="center">SV1-NCYC495</td>
<td valign="top" align="left">-/OGAPODRAFT_16382/-</td>
<td valign="top" align="left">NADP-dependent alcohol dehydrogenase 6</td>
</tr>
<tr>
<td valign="top" align="left">OGAPO_00001</td>
<td valign="top" align="center">SV2- CBS4732</td>
<td valign="top" align="left">OGAPO_00001/-/-</td>
<td valign="top" align="left">Aminotriazole resistance protein</td>
</tr>
<tr>
<td valign="top" align="left">OGAPO_00002</td>
<td valign="top" align="center">SV2- CBS4732</td>
<td valign="top" align="left">OGAPO_00002/-/-</td>
<td valign="top" align="left">Aryl-alcohol dehydrogenase</td>
</tr>
<tr>
<td valign="top" align="left">OGAPO_00003</td>
<td valign="top" align="center">SV2- CBS4732</td>
<td valign="top" align="left">OGAPO_00005/-/-</td>
<td valign="top" align="left">Sterol regulatory element-binding protein ECM22</td>
</tr>
<tr>
<td valign="top" align="left">OGAPO_00004</td>
<td valign="top" align="center">SV2- CBS4732</td>
<td valign="top" align="left">OGAPO_00007/-/-</td>
<td valign="top" align="left">Agmatine ureohydrolase</td>
</tr>
<tr>
<td valign="top" align="left">OGAPO_00005</td>
<td valign="top" align="center">SV2- CBS4732</td>
<td valign="top" align="left">OGAPO_00008/-/-</td>
<td valign="top" align="left">P-loop containing nucleoside triphosphate hydrolase protein</td>
</tr>
<tr>
<td valign="top" align="left">OGAPO_00006</td>
<td valign="top" align="center">SV2- CBS4732</td>
<td valign="top" align="left">OGAPO_00009/-/-</td>
<td valign="top" align="left">MFS general substrate transporter</td>
</tr>
<tr>
<td valign="top" align="left">OGAPO_00007</td>
<td valign="top" align="center">SV2- CBS4732</td>
<td valign="top" align="left">OGAPO_00010/-/-</td>
<td valign="top" align="left">Acetylornithine aminotransferase, mitochondrial (ARG8)</td>
</tr>
<tr>
<td valign="top" align="left">OGAPO_00008</td>
<td valign="top" align="center">SV2- CBS4732</td>
<td valign="top" align="left">OGAPO_00012/-/-</td>
<td valign="top" align="left">Aldo keto reductase (ARK)</td>
</tr>
<tr>
<td valign="top" align="left">OGAPODRAFT_13497<xref ref-type="table-fn" rid="t2fn1">&#x002A;</xref></td>
<td valign="top" align="center">SV2-NCYC495</td>
<td valign="top" align="left">OGAPO_13497/<xref ref-type="table-fn" rid="t2fn1">&#x002A;</xref>/HPODL_00892</td>
<td valign="top" align="left">Basic amino-acid permease</td>
</tr>
<tr>
<td valign="top" align="left">OGAPODRAFT_76936</td>
<td valign="top" align="center">SV2-NCYC495</td>
<td valign="top" align="left">-/OGAPODRAFT_76936/HPODL_00891</td>
<td valign="top" align="left">Transcriptional activator protein DAL81</td>
</tr>
<tr>
<td valign="top" align="left">OGAPODRAFT_16706</td>
<td valign="top" align="center">SV2-NCYC495</td>
<td valign="top" align="left">-/OGAPODRAFT_16706/HPODL_00890</td>
<td valign="top" align="left">DUF1479-domain-containing protein</td>
</tr>
<tr>
<td valign="top" align="left">OGAPODRAFT_37951</td>
<td valign="top" align="center">SV2-NCYC495</td>
<td valign="top" align="left">-/OGAPODRAFT_37951/HPODL_02394</td>
<td valign="top" align="left">MFS domain-containing protein</td>
</tr>
<tr>
<td valign="top" align="left">OGAPODRAFT_93168</td>
<td valign="top" align="center">SV3-NCYC495</td>
<td valign="top" align="left">-/OGAPODRAFT_93168/HPODL_04518</td>
<td valign="top" align="left">MFS domain-containing protein</td>
</tr>
<tr>
<td valign="top" align="left">OGAPODRAFT_15973<xref ref-type="table-fn" rid="t2fn1">&#x002A;</xref></td>
<td valign="top" align="center">SV3-NCYC495</td>
<td valign="top" align="left">OGAPO_15973/<xref ref-type="table-fn" rid="t2fn1">&#x002A;</xref>/HPODL_04520</td>
<td valign="top" align="left">MFS sugar transporter</td>
</tr>
<tr>
<td valign="top" align="left">OGAPODRAFT_75778</td>
<td valign="top" align="center">SV3-NCYC495</td>
<td valign="top" align="left">-/OGAPODRAFT_75778/HPODL_04517</td>
<td valign="top" align="left">Adenosine deaminase</td>
</tr>
<tr>
<td valign="top" align="left">OGAPODRAFT_75779</td>
<td valign="top" align="center">SV3-NCYC495</td>
<td valign="top" align="left">-/OGAPODRAFT_75779/HPODL_04516</td>
<td valign="top" align="left">Zn(2)-C6 fungal-type domain-containing protein</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="t2fn1"><p><italic>The genomic differences between Ogataea polymorpha HU-11/CBS4732 and NCYC495 include SNPs, small InDels, and only three SVs (<xref ref-type="fig" rid="F4">Figure 4B</xref>). Five sequences (SV1-CBS4732, SV2-CBS4732, SV1-NCYC495, SV2-NCYC495, and SV3-NCYC495) were involved in these three SVs. Only 25 genes (<xref ref-type="supplementary-material" rid="DS1">Supplementary File 1</xref>) were involved in the three SVs between HU-11/CBS4732 and NCYC495. 10 genes (OGAPO_03766-67 and OGAPO_00003-08) of HU-11/CBS4732 and 11 genes (OGAPODRAFT_24127, 16381, 24129, 12876, 16382, 76936, 16706, 37951, 93168, 75778, and 75779) of NCYC495 have no orthologs in NCYC495 and HU-11/CBS4732, respectively. &#x002A;Two genes (OGAPODRAFT_13497 and 15973) in NCYC495 were significantly changed into two other ones (OGAPO_13497 and 15973) in HU-11/CBS4732. Six genes (OGAPO_00003-08) in a major part of SV2-CBS4732 have no homologs in NCYC495 or DL-1. OGAPO_, OGAPODRAFT_, and HPODL_ are prefix of gene IDs of HU-11/CBS4732, NCYC495, and DL-1, respectively.</italic></p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec id="S3" sec-type="conclusion">
<title>Conclusion</title>
<p>The <italic>O. polymorpha</italic> strain CBS4732 <italic>ura3</italic>&#x0394; (named HU-11) is a nutritionally deficient mutant derived from CBS4732 by a 5-bp insertion of &#x201C;GAAGT&#x201D; into the 32nd position of the <italic>URA3</italic> CDS; this insertion causes a frame-shift mutation of the <italic>URA3</italic> CDS, resulting in the loss of the <italic>URA3</italic> functions. Since the difference between the genomes of CBS4732 and HU-11 is only five nts, HU-11 has the same reference genome as CBS4732 (noted as HU-11/CBS4732). In the present study, we have assembled the full-length genome of <italic>O. polymorpha</italic> HU-11/CBS4732 using high-depth PacBio and Illumina data. 5&#x2032; and 3&#x2032; telomeric, subtelomeric, rDNA, LTR-rts, low complexity, and other repeat regions were curated to improve the genome quality. In brief, the main findings include complete rDNAs, complete LTR-rts, three LDSs in subtelomeric regions and three SVs between the HU-11/CBS4732 and NCYC495 genomes. SV1, LDS1, LDS2, and LDS3 were validated using the strand-specific RNA-seq data of NCYC495 (SRA: SRP124832). These findings are very important for the assembly of full-length genomes of yeast and the correction of assembly errors in the published genomes of <italic>Ogataea</italic> spp.</p>
<p>The present study preliminarily revealed the relationship between <italic>O. polymorpha</italic> CBS4732, NCYC495, and DL-1. HU-11/CBS4732 is so phylogenetically close to NCYC495 that the syntenic regions cover nearly 100% of their genomes. Moreover, HU-11/CBS4732 and NCYC495 share a nucleotide identity of 99.5% through their whole genomes, including the rDNA regions and LTR-rts. The genomic differences between HU-11/CBS4732 and NCYC495 include SNPs, small InDels, and only three SVs. CBS4732 and NCYC495 can be regarded as the same strain in basic research and industrial applications. The HU-11 (GenBank: CP073033-40) and NCYC495 (RefSeq: NW_017264698-704) genomes represent genomes of <italic>O. polymorpha</italic> MAT&#x03B1; and MATa cells, respectively. Large segments SV1-CBS4732, SV2-CBS4732, SV1-NCYC495, SV2-NCYC495, and SV3-NCYC495 involved in the three SVs can be used to identify <italic>O. polymorpha</italic> strains, particularly CBS4732, NCYC495, and their derivative strains. Among these five large segments, SV2-CBS4732 merits further investigation. The proteins encoded by six genes in a major part of SV2-CBS4732 have no homologs in <italic>Ogataea polymorpha</italic> NCYC495 or DL-1, but have the highest sequence similarities to their homologs in <italic>O. thermophila</italic>, followed by <italic>O. philodendri</italic> and <italic>O. haglerorum</italic>. These six proteins also have homologs in the <italic>Cyberlindnera jadinii</italic> strain NRRL Y-1542. As most genome sequences do not contain complete subtelomeric regions where the six genes locate, the origin of these genes are still not determined.</p>
<p>Only with the exact sequences of subtelomeric regions, can we discover the SV1 and SV2. Furthermore, we reported for the first time LDSs in the subtelomeric regions of <italic>Ogataea</italic> genomes. LDS1 and LDS2 had not been discovered before the present study, mainly because they are nearly identical to their paralogs. A computational study (<xref ref-type="bibr" rid="B2">Brown, 2010</xref>) showed that subtelomeric gene families are evolving and expanding much faster than gene families which do not contain subtelomeric genes in yeasts. This previous study also concluded that the extraordinary instability of eukaryotic subtelomeres supports rapid adaptation to novel niches by promoting gene recombination and duplication followed by the functional divergence of the alleles. Our results indicated that large segment duplication in subtelomeric regions occurs in a size to the extent of &#x223C;27,850 bp and suggests that the genome expansion in methylotrophic yeasts is mainly driven by large segment duplication in subtelomeric regions. The discovery of telomeric TR-like sequences at 3&#x2032; ends of the LDSs indicated that they were integrated at 5&#x2032;&#x2032; ends of the old telomeric regions by their 3&#x2032; ends. The exact LDS and telomeric TR-like sequences are very important for the investigation of the molecular mechanism (if <italic>via</italic> recombination or not) that underlies large segment duplication in subtelomeric regions.</p>
</sec>
<sec id="S4" sec-type="materials|methods">
<title>Materials and Methods</title>
<p>The <italic>Ogataea polymorpha</italic> strain HU-11 (CGMCC No. 1218) was preserved in the China General Microbiological Culture Collection Center (CGMCC). DNA extraction and quality control were performed as described in our previous study (<xref ref-type="bibr" rid="B15">Wang Y. et al., 2016</xref>). A 500 bp DNA library was constructed as described in our previous study (<xref ref-type="bibr" rid="B15">Wang Y. et al., 2016</xref>) and sequenced on the Illumina HiSeq X Ten platform. A 10 Kb DNA library was constructed and sequenced on the PacBio Sequel platforms, according to the manufacturer&#x2032;s instruction. The cleaning and quality control of PacBio data was performed with the software SMRTlink v5.0 (&#x2013;minLength = 50, &#x2013;minReadScore = 0.8), while the cleaning and quality control of Illumina data was performed with the software Fastq_clean (<xref ref-type="bibr" rid="B20">Zhang et al., 2014</xref>) v2.0. PacBio data was used to assemble the HU-11/CBS4732 draft genome with the software MECAT (<xref ref-type="bibr" rid="B17">Xiao et al., 2016</xref>) v1.2. To polish the draft genome, the software BWA was used to align Illumina data to the HU-11/CBS4732 draft genome. Then, the software samtools was used to obtain the BAM and pileup files from the alignment results. Perl scripts were used to extract the consensus sequence from the pileup file. This polishing procedure was repeatedly performed until human curation started. The reference genomes of <italic>O. polymorpha</italic> HU-11/CBS4732, NCYC495 and DL-1 are available at the NCBI GenBank or RefSeq database under the accession numbers <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP073033-40">CP073033-40</ext-link>, <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NW_017264698-704">NW_017264698-704</ext-link> and <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="NC_027860-66">NC_027860-66</ext-link>. Another genome of <italic>O. polymorpha</italic> DL-1 (GenBank: CP080316-22) was also used for syntenic comparison and SV detection, as this genome is more complete than the genome (RefSeq: NC_027860-66).</p>
<p>The genome sequences of 34 species in the <italic>Ogataea</italic> or <italic>Candida</italic> genus were downloaded from the Genome-NCBI datasets and their accession numbers were included in <xref ref-type="supplementary-material" rid="DS1">Supplementary File 1</xref>. Syntenic comparison of genomes was performed using the CoGe website,<sup><xref ref-type="fn" rid="footnote1">1</xref></sup> only for visualization. The detailed syntenic comparison and SV detection were performed locally using the software blast v2.9.0 and Perl scripts. Using the software IGV (<xref ref-type="bibr" rid="B11">Thorvaldsd&#x00F3;ttir et al., 2013</xref>) v2.0.34, human curation of the poly(GC) regions, 5&#x2032; and 3&#x2032; telomeric, subtelomeric, rDNA, LTR-rts, low complexity, and other repeat regions was performed with 103,345 long (&#x003E;20 Kbp) PacBio subreads. The curation criteria is: (1) the junctions between large segments (e.g., LTR-rts, LDSs, or SVs) must be spanned by a long PacBio subread; and (2) the corrected nucleotides must be confirmed by more than 5 long PacBio subreads. To estimate the frequency of MAT switching, each long PacBio subread was counted with human curation. Statistical computation and plotting were performed using the software R v2.15.3 with the Bioconductor packages (<xref ref-type="bibr" rid="B3">Gao et al., 2014</xref>). Prediction of protein-coding genes (&#x003E;150 bp) was performed using the software AUGUSTUS (<xref ref-type="bibr" rid="B5">Hoff and Stanke, 2013</xref>) v2.7.0. Strand-specific RNA-seq data (SRA: SRP124832) was used to curate gene annotations of HU-11/CBS4732, NCYC495 and DL-1. As the reads in the data SRP124832 correspond to the reverse-complementary counterpart of transcripts, they were transformed into their reverse-complementary sequences for all the analyses in the present study.</p>
</sec>
<sec id="S5" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>The complete genome sequence of the <italic>O. polymorpha</italic> HU-11/CBS4732 is available at the NCBI GenBank database under the accession number <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="CP073033-40">CP073033-40</ext-link>, in the project <ext-link ext-link-type="DDBJ/EMBL/GenBank" xlink:href="PRJNA687834">PRJNA687834</ext-link>.</p>
</sec>
<sec id="S6">
<title>Author Contributions</title>
<p>SG conceived the project and drafted the manuscript. SG and DW supervised the present study. JC assembled the HU-11/CBS4732 genome, prepared the figures, tables, and <xref ref-type="supplementary-material" rid="DS1">Supplementary Material</xref>. JB and HF executed the experiments. SG, QS, and TY analyzed the data. SG, HW, WB, and JR revised the manuscript. All authors have read and approved the manuscript.</p>
</sec>
<sec id="conf1" sec-type="COI-statement">
<title>Conflict of Interest</title>
<p>HW was employees by Tianjin Hemu Health Biotechnological Co., Ltd. The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="pudiscl1" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<sec id="S7" sec-type="funding-information">
<title>Funding</title>
<p>This work was supported by the Natural Science Foundation of China (31872388) to HF, Natural Science Foundation of Guangdong Province of China (2021A1515011072) to JB, and Tianjin Key Research and Development Program of China (19YFZCSY00500) to SG. The funding bodies played no role in the design of the study and collection, analysis, and interpretation of data and in writing the manuscript.</p>
</sec>
<ack><p>We appreciate the help equally from the people listed below. They are Dawei Huang, Huaijun Xue, Yanqiang Liu, Bingjun He, Qiang Zhao, and Zhen Ye from College of Life Sciences, Nankai University.</p>
</ack>
<sec id="S9" sec-type="supplementary-material">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmicb.2022.855666/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmicb.2022.855666/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.ZIP" id="DS1" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_2.XLSX" id="DS2" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_3.ZIP" id="DS3" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Agrawal</surname> <given-names>S.</given-names></name> <name><surname>Ganley</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <article-title>The conservation landscape of the human ribosomal RNA gene repeats.</article-title> <source><italic>PLoS One</italic></source> <volume>13</volume>:<issue>e0207531</issue>. <pub-id pub-id-type="doi">10.1371/journal.pone.0207531</pub-id> <pub-id pub-id-type="pmid">30517151</pub-id></citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname> <given-names>C. A.</given-names></name></person-group> (<year>2010</year>). <article-title>Rapid expansion and functional divergence of subtelomeric gene families in yeast.</article-title> <source><italic>Curr. Biol.</italic></source> <volume>20</volume> <fpage>895</fpage>&#x2013;<lpage>903</lpage>. <pub-id pub-id-type="doi">10.1016/j.cub.2010.04.027</pub-id> <pub-id pub-id-type="pmid">20471265</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>S.</given-names></name> <name><surname>Ou</surname> <given-names>J.</given-names></name> <name><surname>Xiao</surname> <given-names>K.</given-names></name></person-group> (<year>2014</year>). <source><italic>R Language and Bioconductor in Bioinformatics Applications(Chinese Edition).</italic></source> <publisher-loc>Tianjin</publisher-loc>: <publisher-name>Tianjin Science and Technology Translation Publishing Ltd</publisher-name>.</citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hanson</surname> <given-names>S. J.</given-names></name> <name><surname>Byrne</surname> <given-names>K. P.</given-names></name> <name><surname>Wolfe</surname> <given-names>K. H.</given-names></name> <name><surname>Joseph</surname> <given-names>H.</given-names></name></person-group> (<year>2017</year>). <article-title>Flip/flop mating-type switching in the methylotrophic yeast <italic>Ogataea polymorpha</italic> is regulated by an Efg1-Rme1-Ste12 pathway.</article-title> <source><italic>PLoS Genet.</italic></source> <volume>13</volume>:<issue>e1007092</issue>. <pub-id pub-id-type="doi">10.1371/journal.pgen.1007092</pub-id> <pub-id pub-id-type="pmid">29176810</pub-id></citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hoff</surname> <given-names>K. J.</given-names></name> <name><surname>Stanke</surname> <given-names>M.</given-names></name></person-group> (<year>2013</year>). <article-title>WebAUGUSTUS&#x2013;a web service for training AUGUSTUS and predicting genes in eukaryotes.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>41</volume> <fpage>123</fpage>&#x2013;<lpage>128</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkt418</pub-id> <pub-id pub-id-type="pmid">23700307</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Levine</surname> <given-names>D. W.</given-names></name> <name><surname>Cooney</surname> <given-names>C. L.</given-names></name></person-group> (<year>1973</year>). <article-title>Isolation and characterization of a thermotolerant methanol-utilizing yeast.</article-title> <source><italic>Appl. Environ. Microbiol.</italic></source> <volume>26</volume> <fpage>982</fpage>&#x2013;<lpage>990</lpage>. <pub-id pub-id-type="doi">10.1128/am.26.6.982-990.1973</pub-id> <pub-id pub-id-type="pmid">4767300</pub-id></citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Massoud</surname> <given-names>R. R.</given-names></name> <name><surname>Hollenberg</surname> <given-names>C. P.</given-names></name> <name><surname>Juergen</surname> <given-names>L.</given-names></name> <name><surname>Holger</surname> <given-names>W.</given-names></name> <name><surname>Eike</surname> <given-names>G.</given-names></name> <name><surname>Christian</surname> <given-names>W.</given-names></name><etal/></person-group> (<year>2003</year>). <article-title>The <italic>Hansenula polymorpha</italic> (strain CBS4732) genome sequencing and analysis.</article-title> <source><italic>FEMS Yeast Res.</italic></source> <volume>4</volume> <fpage>207</fpage>&#x2013;<lpage>215</lpage>. <pub-id pub-id-type="doi">10.1016/S1567-1356(03)00125-9</pub-id> <pub-id pub-id-type="pmid">14613885</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Morais</surname> <given-names>J. O. F.</given-names></name> <name><surname>Maia</surname> <given-names>M. H. D.</given-names></name></person-group> (<year>1959</year>). <article-title>Estudos de microorganismos encontrados em leitos de despejos de caldas de destilarias de Pernambuco. II. Uma nova especie de Hansenula: <italic>H. polymorpha</italic>.</article-title> <source><italic>An. Esc. Super. Quim. Univ. Recife</italic></source> <volume>1</volume> <fpage>15</fpage>&#x2013;<lpage>20</lpage>.</citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Neuveglise</surname> <given-names>C.</given-names></name> <name><surname>Feldmann</surname> <given-names>H.</given-names></name> <name><surname>Bon</surname> <given-names>E.</given-names></name> <name><surname>Gaillardin</surname> <given-names>C.</given-names></name> <name><surname>Casaregola</surname> <given-names>S.</given-names></name></person-group> (<year>2002</year>). <article-title>Genomic evolution of the long terminal repeat retrotransposons in hemiascomycetous yeasts.</article-title> <source><italic>Genome Res.</italic></source> <volume>12</volume> <fpage>930</fpage>&#x2013;<lpage>943</lpage>. <pub-id pub-id-type="doi">10.1101/gr.219202</pub-id> <pub-id pub-id-type="pmid">12045146</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ravin</surname> <given-names>N. V.</given-names></name> <name><surname>Eldarov</surname> <given-names>M. A.</given-names></name> <name><surname>Kadnikov</surname> <given-names>V. V.</given-names></name> <name><surname>Beletsky</surname> <given-names>A. V.</given-names></name> <name><surname>Skryabin</surname> <given-names>K. G.</given-names></name></person-group> (<year>2013</year>). <article-title>Genome sequence and analysis of methylotrophic yeast <italic>Hansenula polymorpha</italic> DL1.</article-title> <source><italic>BMC Genomics</italic></source> <volume>14</volume>:<issue>837</issue>. <pub-id pub-id-type="doi">10.1186/1471-2164-14-837</pub-id> <pub-id pub-id-type="pmid">24279325</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thorvaldsd&#x00F3;ttir</surname> <given-names>H.</given-names></name> <name><surname>Robinson</surname> <given-names>J. T.</given-names></name> <name><surname>Mesirov</surname> <given-names>J. P.</given-names></name></person-group> (<year>2013</year>). <article-title>Integrative genomics viewer (IGV): high-performance genomics data visualization and exploration.</article-title> <source><italic>Brief Bioinform.</italic></source> <volume>14</volume> <fpage>178</fpage>&#x2013;<lpage>192</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbs017</pub-id> <pub-id pub-id-type="pmid">22517427</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>He</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>B.</given-names></name></person-group> (<year>2007</year>). <source><italic>A Method to Construct Ogataea polymorpha Strains and Its Application. CN: 200410080517.2.</italic></source></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>J.</given-names></name></person-group> (<year>2011</year>). <source><italic>A Recombinant Hirudin Gene and Its Application. CN: 200810103154.8.</italic></source></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Yang</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <source><italic>A High-Dose Recombinant B Hepatitis Vaccine Expressed in Ogataea spp. CN: 201610178526.8.</italic></source></citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Guo</surname> <given-names>F.</given-names></name> <name><surname>Zhang</surname> <given-names>K.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>The complete genome of <italic>Brucella suis</italic> 019 provides insights on cross-species infection.</article-title> <source><italic>Genes</italic></source> <volume>7</volume> <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.3390/genes7020007</pub-id> <pub-id pub-id-type="pmid">26821047</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wickerham</surname> <given-names>L. J.</given-names></name></person-group> (<year>1951</year>). <article-title>Taxonomy of yeasts.</article-title> <source><italic>Tech. Bull. U. S. Dep. Agric.</italic></source> <volume>6</volume> <fpage>781</fpage>&#x2013;<lpage>782</lpage>.</citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiao</surname> <given-names>C. L.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Xie</surname> <given-names>S. Q.</given-names></name> <name><surname>Chen</surname> <given-names>K. N.</given-names></name> <name><surname>Xie</surname> <given-names>Z.</given-names></name></person-group> (<year>2016</year>). <article-title>MECAT: an ultra-fast mapping, error correction and de novo assembly tool for single-molecule sequencing reads.</article-title> <source><italic>bioRxiv</italic></source> [<comment>Preprint</comment>] <pub-id pub-id-type="doi">10.1101/089250</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>X.</given-names></name> <name><surname>Bei</surname> <given-names>J.</given-names></name> <name><surname>Xuan</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Chen</surname> <given-names>D.</given-names></name> <name><surname>Barker</surname> <given-names>S. C.</given-names></name><etal/></person-group> (<year>2020</year>). <article-title>Full-length genome sequence of segmented RNA virus from ticks was obtained using small RNA sequencing data.</article-title> <source><italic>BMC Genomics</italic></source> <volume>21</volume>:<issue>641</issue>. <pub-id pub-id-type="doi">10.1186/s12864-020-07060-5</pub-id> <pub-id pub-id-type="pmid">32938401</pub-id></citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>F.</given-names></name> <name><surname>Xu</surname> <given-names>T.</given-names></name> <name><surname>Mao</surname> <given-names>L.</given-names></name> <name><surname>Yan</surname> <given-names>S.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Wu</surname> <given-names>Z.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>Genome-wide analysis of Dongxiang wild rice (<italic>Oryza rufipogon</italic> Griff.) to investigate lost/acquired genes during rice domestication.</article-title> <source><italic>BMC Plant Biol.</italic></source> <volume>16</volume>:<issue>103</issue>. <pub-id pub-id-type="doi">10.1186/s12870-016-0788-2</pub-id> <pub-id pub-id-type="pmid">27118394</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>M.</given-names></name> <name><surname>Zhan</surname> <given-names>F.</given-names></name> <name><surname>Sun</surname> <given-names>H.</given-names></name> <name><surname>Gong</surname> <given-names>X.</given-names></name> <name><surname>Fei</surname> <given-names>Z.</given-names></name> <name><surname>Gao</surname> <given-names>S.</given-names></name></person-group> (<year>2014</year>). &#x201C;<article-title>Fastq_clean: an optimized pipeline to clean the Illumina sequencing data with quality control</article-title>,&#x201D; in <source><italic>Proceedings of the 2014 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</italic></source>, (<publisher-loc>Belfast</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation></ref>
</ref-list>
<glossary>
<title>Abbreviations</title>
<def-list id="DL1">
<def-item><term>TR</term><def><p>tandem repeat</p></def></def-item>
<def-item><term>STR</term><def><p>short tandem repeat</p></def></def-item>
<def-item><term>LTR</term><def><p>long terminal repeat</p></def></def-item>
<def-item><term>ORF</term><def><p>open reading frame</p></def></def-item>
<def-item><term>CDS</term><def><p>coding sequence</p></def></def-item>
<def-item><term>SV</term><def><p>structural variation</p></def></def-item>
<def-item><term>SNP</term><def><p>single nucleotide polymorphism</p></def></def-item>
<def-item><term>InDel</term><def><p>insertion and deletion</p></def></def-item>
<def-item><term>mt</term><def><p>mitochondrial</p></def></def-item>
<def-item><term>nt</term><def><p>nucleotide</p></def></def-item>
<def-item><term>aa</term><def><p>amino acid.</p></def></def-item>
</def-list>
</glossary>
<fn-group>
<fn id="footnote1">
<label>1</label>
<p><ext-link ext-link-type="uri" xlink:href="https://genomevolution.org/CoGe">https://genomevolution.org/CoGe</ext-link></p></fn>
</fn-group>
</back>
</article>
