<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1502681</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2025.1502681</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Chromosome-level draft genome assembly of <italic>Hypomesus nipponensis</italic> reveals transposable element expansion reshaping the genome structure</article-title>
<alt-title alt-title-type="left-running-head">Zhu et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2025.1502681">10.3389/fgene.2025.1502681</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zhu</surname>
<given-names>Chenzhao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2778814/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kuang</surname>
<given-names>Youyi</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/103963/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Zhe</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Tang</surname>
<given-names>Fujiang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>College of Fisheries and Life Science</institution>, <institution>Shanghai Ocean University</institution>, <addr-line>Shanghai</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Scientific Observing and Experimental Station of Fishery Resources and Environment in Heilongjiang River Basin</institution>, <institution>Ministry of Agriculture and Rural Affairs</institution>, <institution>Heilongjiang River Fisheries Research Institute of Chinese Academy of Fishery Sciences</institution>, <addr-line>Harbin</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/48202/overview">Shi-Yi Chen</ext-link>, Sichuan Agricultural University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/880643/overview">Tulio de Lima Campos</ext-link>, Aggeu Magalh&#xe3;es Institute (IAM), Brazil</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1436062/overview">Gabriel M. Yazbeck</ext-link>, Universidade Federal de S&#xe3;o Jo&#xe3;o del-Rei, Brazil</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/893999/overview">Duminda Senevirathna</ext-link>, Uva Wellassa University, Sri Lanka</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1575152/overview">Renxie Wu</ext-link>, Guangdong Ocean University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Fujiang Tang, <email>rivery2008@163.com</email>; Youyi Kuang, <email>kuangyouyi@hrfri.ac.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>29</day>
<month>04</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1502681</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>09</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>04</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Zhu, Kuang, Li and Tang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Zhu, Kuang, Li and Tang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<italic>Hypomesus nipponensis</italic> a commercially valuable fish within the Osmeriformes order, is naturally found in northeastern Asia and has been extensively introduced for commercial purposes across eastern Asia. To investigate the taxonomic status and evolutionary history of <italic>Hypomesus nipponensis</italic> within the Osmeridae family, we first performed a <italic>de novo</italic> genome assembly using PacBio HiFi reads and CLR (Continuous Long Read) reads. Subsequently, we leveraged synteny information from closely related species to further refine the assembly and construct a chromosome-level genome. The final assembly spans 507.8 Mb, with a scaffold N50 of 20&#xa0;Mb, achieving chromosome-level contiguity. It comprises 164&#xa0;Mb of repetitive sequences and encodes 27,876 protein-coding genes. Compared to previous assembly, the <italic>H. nipponensis</italic> genome is notably more contiguous and complete. Notably, it contains an unusually high proportion of tandem repeats, which likely contributed to the assembly challenges encountered in earlier efforts. We also observed the transposons of <italic>H. nipponensis</italic> have expanded significantly in recent times, and paralogous gene families have expanded during the same period. Our analysis estimates that <italic>H. nipponensis</italic>, <italic>Osmerus eperlanus</italic>, and <italic>Hypomesus transpacificus</italic> diverged from a common ancestor approximately 24.1 million years ago, with significant chromosomal segment recombination events occurring during their divergence. Additionally, we compared the genomes of <italic>O. eperlanus</italic> and <italic>Hypomesus</italic> and found that most of the genes in the Presence/Absence Variants (PAVs) of <italic>O. eperlanus</italic> were associated with immune response. Our efforts significantly enhance the genome&#x2019;s integrity and continuity for this ecologically and commercially important fish, providing a chromosome-level genome draft that supports fundamental biological research while offering insights into the evolutionary relationships and genomic diversity within the Osmeriformes order. This advancement has profound implications for understanding the evolutionary history and adaptive strategies of <italic>H. nipponensis</italic>.</p>
</abstract>
<kwd-group>
<kwd>
<italic>Hypomesus nipponensis</italic>
</kwd>
<kwd>genome</kwd>
<kwd>evolutionary</kwd>
<kwd>gene family</kwd>
<kwd>repeat sequence</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Livestock Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>
<italic>Hypomesus nipponensis</italic> McAllister is a small, cold-water fish species characterized by a short life cycle, high fecundity, and rapid population growth. These traits that enhance contribute to diverse environmental conditions. This species belongs to the genus <italic>Hypomesus</italic>, the most species-rich genus in the smelt family (Osmeridae), which currently comprises five recognized species. During the juvenile stage, most individuals migrate to the sea (sea migratory type), while others remain resident in freshwater lake (lake dwelling type) (<xref ref-type="bibr" rid="B5">Asami, 2004</xref>). This dual life history strategy not only facilitates colonization of divergent habitats but also promotes genetic divergence and local adaptation. Originally distributed across Japan, the Korean Peninsula and Primorsky Krai, Far East of Russia this species has been intentionally introduced to various water systems. It has subsequently become both an economically important aquaculture species and a high-quality food source for piscivorous fish (<xref ref-type="bibr" rid="B59">Swanson et al., 2000</xref>). Before the 1980s, <italic>Hypomesus nipponensis</italic> was introduced into northeastern China. In the 1980s and 1990s, it was introduced from northeastern China to a wide range of inland regions, including the highland lakes of southwest China (<xref ref-type="bibr" rid="B65">Xie et al., 1992</xref>). As a highly mobile species, <italic>H. nipponensis</italic> readily disperses between aquatic systems and is now widely distributed across northeastern Asia, including China, Japan, and the Korean Peninsula (<xref ref-type="bibr" rid="B69">Yin et al., 2021</xref>).</p>
<p>In recent years, the genome of <italic>H. nipponensis</italic> was assembled to investigate molecular response mechanisms to heat stress (<xref ref-type="bibr" rid="B67">Xuan et al., 2021</xref>). However, while the article reports a genome size of 486&#xa0;Mb, only 34.4&#xa0;Mb of scaffolded sequences are publicly available, indicating potential incompleteness in assembly. Incomplete genome assemblies can lead to the partial or complete omission of genes, resulting in artifacts of pseudo-gene loss that may bias downstream functional analyses (<xref ref-type="bibr" rid="B34">Kim et al., 2022</xref>). Integrating genomic architecture with ecological performance is therefore critical for understanding how <italic>H. nipponensis</italic> adapts to rapid environmental change. To date, molecular biology and genomic research on <italic>H. nipponensis</italic> have been limited due to the absence of a complete genome and comprehensive annotation. This lack of genomic information significantly restricts studies on the phylogeny and genetic differentiation of <italic>H. nipponensis</italic>. Furthermore, it hampers the exploration of the adaptation and reproductive strategies of <italic>H. nipponensis</italic> at the genomic level. In this study, we reassembled the <italic>H. nipponensis</italic> genome using PacBio HiFi sequencing and CLR data, achieving significantly improved continuity and completeness. This high-quality genome assembly enables more accurate identification and characterization of key genes, regulatory elements, and structural variations, which are essential for understanding the species&#x2019; unique biological traits and adaptive mechanisms. To better understand the evolutionary processes of <italic>H. nipponensis</italic>, we conducted a direct comparison with the genomes of <italic>Hypomesus</italic> and <italic>O. eperlanus</italic>. Additionally, we created chromosome collinearity maps for several species and reconstructed the ancestral chromosomes of Osmeriformes. These comprehensive analyses not only provide a robust foundation for comparative genomic studies with related species but also offer valuable insights into the ecological adaptations and evolutionary pressures shaping <italic>H. nipponensis</italic>&#x2019; genome.</p>
</sec>
<sec sec-type="results|discussion" id="s2">
<title>Results and discussion</title>
<sec id="s2-1">
<title>Chromosome-level genome assembly</title>
<p>We assembled the genome using a pipeline listed in <xref ref-type="sec" rid="s12">Supplementary Figure S1</xref>. First, we sequenced the genome using the PacBio platform and acquired 26.88&#xa0;Gb of PacBio HiFi reads at a depth of 50&#xd7;. The estimated genome size of <italic>H. nipponensis</italic> based on HiFi reads is approximately 463.49 Mb, with a predicted heterozygosity rate of 0.396% (<xref ref-type="sec" rid="s12">Supplementary Figure S2</xref>). We generated Draft Genome v0.1 with a scaffold N50 of 0.5&#xa0;Mb and Draft Genome v0.2 with a scaffold N50 of 2.3&#xa0;Mb. Subsequently, we merged the two scaffolds to obtain Draft Genome v0.3, resulting in a merged scaffold N50 of 7.9&#xa0;Mb. The merged sequence was polished to remove redundancy, yielding Draft Genome v0.4 with a scaffold N50 of 8.1&#xa0;Mb. We then scaffolded the purged genome with CLR reads to obtain a draft genome v0.5 with a scaffold N50 of 8.1&#xa0;Mb (<xref ref-type="sec" rid="s12">Supplementary Table S1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1</xref>). Finally, we constructed 28 pseudochromosomes (<xref ref-type="fig" rid="F1">Figure 1</xref>). The assembly achieved a contig N50 of 8&#xa0;Mb and scaffold N50 of 20&#xa0;Mb, with the longest contig spanning 21&#xa0;Mb and an average contig length of 1.3&#xa0;Mb (<xref ref-type="table" rid="T1">Table 1</xref>). The combination of PacBio HiFi and CLR reads leverages the strengths of both technologies: HiFi provides high accuracy (&#x223c;0.1% error rate) and moderate read lengths (&#x223c;15&#xa0;kb), while CLR offers longer reads, albeit with higher noise (<xref ref-type="bibr" rid="B40">Logsdon et al., 2020</xref>; <xref ref-type="bibr" rid="B48">Nurk et al., 2022</xref>; <xref ref-type="bibr" rid="B64">Wenger et al., 2019</xref>). However, HiFi-based assemblies often fragment at large and homogeneous repeats as well as known sequence-specific coverage dropouts. Our assembly strategy, integrating both methods, resulted in a highly contiguous and accurate genome and has been validated in the tomato genome assembly (<xref ref-type="bibr" rid="B1">Alonge et al., 2022</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Circos plot of the <italic>Hypomesus nipponensis</italic> genome, with lines of different colors in the diagram represent the colinearity within the genome. The rings from inside to outside indicate (a) pseudochromosome length of the <italic>Hypomesus nipponensis</italic> genome (b) GC density (c) TE density, and (d) gene density; b-d were drawn in 100-kb sliding windows.</p>
</caption>
<graphic xlink:href="fgene-16-1502681-g001.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Statistics of the <italic>Hypomesus nipponensis</italic> genome assembly and corresponding gene prediction and functional annotation.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Global statistics</th>
<th align="center">Genome</th>
<th align="center">Gene models with evidence</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td colspan="3" align="left">Genome assembly</td>
</tr>
<tr>
<td align="center">Number of contigs</td>
<td align="center">186</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Total contig length (pb)</td>
<td align="center">532,605,080</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Estimated genome size (pb)</td>
<td align="center">478,351,723</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Contig length N50 (pb)</td>
<td align="center">8,193,377</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Scaffold N50 length (pb)</td>
<td align="center">20,113,295</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Longest contig (pb)</td>
<td align="center">21,290,931</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Average contig length (pb)</td>
<td align="center">2,863,468</td>
<td align="left"/>
</tr>
<tr>
<td align="center">GC content (%)</td>
<td align="center">45.84</td>
<td align="left"/>
</tr>
<tr>
<td align="center">N&#x2019;s per 100&#xa0;kbp</td>
<td align="center">3.55</td>
<td align="left"/>
</tr>
<tr>
<td colspan="3" align="left">BUSCO statistics (%)</td>
</tr>
<tr>
<td align="center">BUSCO (Actinopterygii) complete</td>
<td align="center">96.7</td>
<td align="center">97.2</td>
</tr>
<tr>
<td align="center">Complete and single-copy</td>
<td align="center">94.6</td>
<td align="center">94.7</td>
</tr>
<tr>
<td align="center">Complete and duplicated</td>
<td align="center">2.1</td>
<td align="center">2.5</td>
</tr>
<tr>
<td align="center">Fragmented</td>
<td align="center">1.1</td>
<td align="center">1.1</td>
</tr>
<tr>
<td align="center">Missing</td>
<td align="center">2.2</td>
<td align="center">1.7</td>
</tr>
<tr>
<td colspan="3" align="left">Genome annotation</td>
</tr>
<tr>
<td align="center">Protein-coding gene number</td>
<td align="center">27,876</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Average gene length (bp)</td>
<td align="center">8,122.5</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Mean CDS length (bp)</td>
<td align="center">1,541.61</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Longest CDS (bp)</td>
<td align="center">36,194</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Mean protein length (aa)</td>
<td align="center">513.8</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Longest protein (aa)</td>
<td align="center">12,064</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Exon count per gene</td>
<td align="center">8.9</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Average exon length (bp)</td>
<td align="center">175.57</td>
<td align="left"/>
</tr>
<tr>
<td colspan="3" align="left">Functional annotation</td>
</tr>
<tr>
<td align="center">Swissprot</td>
<td align="center">23,764</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Gene Ontology terms</td>
<td align="center">17,635</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Kegg</td>
<td align="center">17,094</td>
<td align="left"/>
</tr>
<tr>
<td align="center">TrEMBL</td>
<td align="center">26,718</td>
<td align="left"/>
</tr>
<tr>
<td align="center">Interpro</td>
<td align="center">23,041</td>
<td align="left"/>
</tr>
</tbody>
</table>
</table-wrap>
<p>The assembled genome exhibited high completeness, with a BUSCO (actinopterygii_odb10) score of 96.7% (including 2.3% duplicated genes) (<xref ref-type="table" rid="T1">Table 1</xref>). To further assess accuracy, we calculated the quality value (QV) of the genome, resulting in a QV of 39.4088 with an error rate of 0.0168%. The QV value of its closely related species <italic>Hypomesus transpacificus</italic> was 36.5925 with an error rate of 0.0219%, but the QV value of <italic>O. eperlanus</italic> was unknown because the original data were not uploaded. We aligned the HiFi and CLR reads to the genome, revealing that 99.92% of the genome had coverage greater than 5x. Moreover, aligning the transcriptome data to the genome resulted in a sequence alignment of 98.22% (<xref ref-type="sec" rid="s12">Supplementary Table S2</xref>). These results demonstrate the high completeness and accuracy of our genome. When compared to other species of <italic>Hypomesus</italic>, our assembled genome exhibits superior completeness and continuity (<xref ref-type="sec" rid="s12">Supplementary Table S3</xref>). The high-quality fish genome serves as a transformative key into the intricate world of aquatic life, revealing the evolutionary history, environmental adaptations, and potential applications for aquaculture (<xref ref-type="bibr" rid="B25">Gui et al., 2022</xref>).</p>
</sec>
<sec id="s2-2">
<title>Genomic features and annotation quality</title>
<p>The assembled genome contains approximately 178.9&#xa0;Mb of repetitive sequences, accounting for 33.59% of the total genome length. This composition included 26.02% (46.55&#xa0;Mb) DNA transposons, 8.75% (15.67&#xa0;Mb) long interspersed nuclear elements (LINEs), 16.69% (29.86&#xa0;Mb) long terminal repeats (LTRs), and 8.75% (15.67&#xa0;Mb) short interspersed nuclear elements (SINEs) (<xref ref-type="fig" rid="F2">Figure 2A</xref>). The transposable elements in the genome of <italic>H. nipponensis</italic> are much higher than those of other fish with similar known genome sizes (<italic>Takifugu rubripes, Gasterosteus aculeatus</italic>), which are approximately 15% (<xref ref-type="bibr" rid="B32">Kasahara et al., 2007</xref>; <xref ref-type="bibr" rid="B52">Peichel et al., 2001</xref>; <xref ref-type="bibr" rid="B54">Shao et al., 2019</xref>). The high proportion of transposable elements in <italic>H. nipponensis</italic> suggests a dynamic evolutionary history, potentially contributing to genomic plasticity and adaptation. Similar patterns have been observed in other teleosts, where repetitive elements play a role in genome expansion and diversification (<xref ref-type="bibr" rid="B11">Chalopin et al., 2015</xref>). These findings highlight the importance of repetitive elements in shaping fish genomes and their functional implications.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>
<bold>(A)</bold> The ratio of different repeat sequences. <bold>(B)</bold> The top of the stacking strip diagram represents the integrity assessment, and the lower part of the strip diagram represents the consistency assessment. <bold>(C)</bold> The left, middle, and right panels illustrate the distributions of mRNA length, exon length, and CDS length, respectively, in comparison with those of other representative species.</p>
</caption>
<graphic xlink:href="fgene-16-1502681-g002.tif"/>
</fig>
<p>After masking repetitive regions, we predicted 46,271 genes using <italic>de novo</italic> methods, 11,869&#x2013;39,908 genes using homology prediction, and 22,485 genes using transcriptome prediction (<xref ref-type="sec" rid="s12">Supplementary Table S4</xref>). We annotated 27,876 protein-coding genes in the assembled genome. These 27,876 genes have an average length of 8&#xa0;kb and cumulatively account for 35.1% of the genome. Within the final gene set of <italic>H. nipponensis,</italic> 26,718 genes (95.8%) (<xref ref-type="table" rid="T1">Table 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S3</xref>) exhibited annotated functions with at least one hit from the searched databases.</p>
<p>The completeness, consistency, and accuracy of the gene structure annotation were evaluated using three different strategies. First, BUSCO analysis revealed that 97.2% of the 3,640 single-copy orthologs from the actinopterygii_odb10 database were successfully identified as complete, with 94.7% classified as single-copy genes and 2.5% as duplicated genes. In contrast, 1.1% were fragmented, and 1.7% were missing in the assembly (<xref ref-type="table" rid="T1">Table 1</xref>). The OMAK assessment yielded a completeness score of 14,066 (97.34%), with 13,131 (90.87%) identified as single-copy proteins and 935 (6.47%) as duplicated proteins out of a total of 27,876 proteins (<xref ref-type="fig" rid="F2">Figure 2B</xref>). Additionally, 25,310 (90.79%) proteins were consistently placed within their expected lineages. Furthermore, the length distributions of mRNA, CDS, and exons in <italic>Danio rerio</italic>, <italic>Esox lucius</italic>, <italic>T</italic>. <italic>rubripes</italic>, <italic>G. aculeatus</italic>, and <italic>H. transpacificus</italic> were found to be similar. (<xref ref-type="fig" rid="F2">Figure 2C</xref>). The high BUSCO completeness score (97.2%) and low fragmentation rate (1.1%) underscore the robustness of our gene annotation, which surpasses that of the previous assembly of this species (<xref ref-type="bibr" rid="B67">Xuan et al., 2021</xref>). The consistency in protein placement within expected lineages (90.79%) further supports the reliability of our annotation pipeline. The similarity in mRNA, CDS, and exon length distributions across multiple species suggests conserved gene structure characteristics within teleosts, consistent with findings from other studies (<xref ref-type="bibr" rid="B8">Braasch et al., 2016</xref>).</p>
</sec>
<sec id="s2-3">
<title>Phylogenetic placement of <italic>Hypomesus nipponensis</italic> and gene family analysis</title>
<p>
<italic>Hypomesus nipponensis</italic> and ten other representative species were subjected to evolutionary and protein family analyses (<xref ref-type="sec" rid="s12">Supplementary Table S6</xref>). Ultimately, we identified a total of 6,029 shared orthologous gene families, of which 586 were single-copy gene families (<xref ref-type="sec" rid="s12">Supplementary Figure S4</xref>). Using these single-copy orthologous gene families, we constructed a phylogenetic tree based on the maximum likelihood method (<xref ref-type="fig" rid="F3">Figure 3A</xref>). To estimate the accurate divergence time of <italic>H. nipponensis</italic>, we calculated the synonymous substitution rates of homoeologous genes to determine the divergence time between <italic>Hypomesus</italic> and <italic>Osmerus.</italic> The substitution rates for orthologous genes between <italic>H. nipponensis</italic> and various species were calculated as follows: 0.95 for <italic>E. lucius-H. nipponensis</italic>, 0.15 for <italic>O. eperlanus-H. nipponensis</italic>, 0.13 for <italic>H. transpacificus-H. nipponensis</italic>, and 0.27 for <italic>P. chinensis-H. nipponensis</italic> (<xref ref-type="fig" rid="F3">Figure 3B</xref>). In addition, we calculated the synonymous substitution rates of paralogous genes in the three Osmeridae species (<xref ref-type="sec" rid="s12">Supplementary Figure S5</xref>), with a peak value of Ks &#x3d; 1.1. We applied the previously determined molecular clock that Ks in teleost was &#x223c;3.51 &#xd7; 10<sup>&#x2212;9</sup> substitutions per synonymous site per year (<xref ref-type="bibr" rid="B17">David et al., 2003</xref>). Therefore, the divergence time of 21.4&#xa0;Mya given by synonymous substitution rates is similar to the result of the time-divergence tree, and we estimate that <italic>H. nipponensis, O. eperlanus</italic>, and <italic>H. transpacificus</italic> diverged from a common ancestor approximately 24.1 (21.4&#x2013;26.9) Mya ago. Moreover, the expansion time of these three Osmeridae paralogous genes was 150&#xa0;Mya, which coincides with the divergence time of Osmeridae. It is possible that the expansion of paralogous genes provided Osmeridae with a diverse gene repertoire, thereby promoting the divergence of species.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>
<bold>(A)</bold> Gene family analysis and divergence time of ten representative species, gray lines indicate confidence intervals. <bold>(B)</bold> The distribution of the synonymous substitution rates (Ks) of homologous genes between <italic>Esox lucius</italic> and <italic>Hypomesus nipponensis</italic>, <italic>O. eperlanus</italic> and <italic>Hypomesus nipponensis</italic>, <italic>Hypomesus transpacificus</italic> and <italic>Hypomesus nipponensis</italic>, <italic>P. chinensis</italic> and <italic>Hypomesus nipponensis</italic>. <bold>(C)</bold> Shows GO enrichment for expanded gene families. <bold>(D)</bold> shows GO enrichment for contracted.</p>
</caption>
<graphic xlink:href="fgene-16-1502681-g003.tif"/>
</fig>
<p>Furthermore, 231 expanded and 4,484 contracted gene families were identified in <italic>H. nipponensis</italic>. Notably, among the contracted gene families, those associated with sexual reproduction and gamete generation stand out (<xref ref-type="fig" rid="F3">Figure 3C</xref>). In contrast, we observed significant expansions in gene families related to organ or tissue-specific immune responses, antigen sampling in mucosa-associated lymphoid tissues, and pathways associated with kidney development and skeletal muscle contraction, specifically muscle filament sliding (<xref ref-type="fig" rid="F3">Figure 3D</xref>). These expansions suggest an enhancement in physical mobility and adaptability to diverse environmental challenges (<xref ref-type="bibr" rid="B69">Yin et al., 2021</xref>).</p>
</sec>
<sec id="s2-4">
<title>Repeat expansion reshapes chromosome structure</title>
<p>During the assembly of <italic>H. nipponensis</italic> genome, we observed relatively lower genome continuity. To investigate this, we compared the repeats of <italic>H. nipponensis</italic> with those of the <italic>G. aculeatus</italic> and <italic>T</italic>. <italic>rubripes</italic>. Our analysis revealed that the tandem repeat sequences constitute 3.26% of the <italic>G. aculeatus</italic> genome, 5.83% of the <italic>T</italic>. <italic>rubripes</italic> genome, and a significantly higher 14.61% of the <italic>H. nipponensis</italic> genome (<xref ref-type="fig" rid="F4">Figure 4A</xref>; <xref ref-type="sec" rid="s12">Supplementary Table S5</xref>). In order to explore the reasons for the increase in repeat sequences in <italic>H. nipponensis</italic>, we employed an analysis based on Kimura distance (<xref ref-type="bibr" rid="B35">Kimura, 1980</xref>), which revealed two major expansion events in the repetitive sequences of the genome (<xref ref-type="fig" rid="F4">Figure 4B</xref>). The more recent peak is caused by a significant expansion of all transposons (<xref ref-type="sec" rid="s12">Supplementary Figure S6</xref>), and the paralogous gene families of <italic>H. nipponensis</italic> were observed to have expanded recently (<xref ref-type="sec" rid="s12">Supplementary Figure S5</xref>). The most recently expanded genes were identified, and two paralogues with more pronounced amplification were extracted (<xref ref-type="sec" rid="s12">Supplementary Figure S7</xref>). The contigs containing these duplicated fragments were subsequently aligned, revealing aligned bases of 7.82% and 6.34%, respectively. Furthermore, the read depth of both HiFi and CLR data was consistent with genome-wide coverage, excluding alternative haplotypes. Assembly-induced duplication artifacts were also ruled out. The repetitive sequences in these recently amplified fragments were then extracted, and a Kimura distance map was constructed for them (<xref ref-type="sec" rid="s12">Supplementary Figure S8</xref>). It was found that the recent amplification of LINEs was very obvious, which was consistent with the amplification curve of the paralogous gene family in the genome, so the amplification of the repetitive sequences was likely to have led to the expansion of the gene family. Gene Ontology (GO) enrichment analysis of these genes indicated that the majority are associated with chromatin DNA binding functions (<xref ref-type="sec" rid="s12">Supplementary Figure S9</xref>). The DNA transposons and LINEs peaks are similar and show enrichment around 30&#xa0;Mya, which may be attributed to the differentiation of the Osmeridae family. The farther peak is caused by the expansion of LTRs, which occurred before 140&#xa0;Mya (<xref ref-type="sec" rid="s12">Supplementary Figure S6</xref>). This implies that <italic>H. nipponensis</italic> genome has experienced dynamic evolutionary changes due to the proliferation of diverse repetitive elements. The multiplication of these transposable elements likely contributed significantly to shaping the genome&#x2019;s current structure.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>
<bold>(A)</bold> The number of tandem repeat sequences of different lengths in <italic>H. nippomensis</italic>, <italic>Gasterosteus aculeatus</italic> and Tetraodontidae<italic>.</italic> <bold>(B)</bold> Transposable element (TE) accumulation history in the <italic>Hypomesus nipponensis</italic> genome, based on a Kimura distance-based copy divergence analysis of TEs, with Kimura substitution level (CpG adjusted) illustrated on the x-axis, and percentage of the genome represented by each repeat type on the y-axis. The bar color indicates the repeat type. <bold>(C)</bold> HiFi reads the coverage depth of three scattered sequences. <bold>(D)</bold> Heatmap of average nucleotide identities between pairwise combinations of genomic intervals from three sequences.</p>
</caption>
<graphic xlink:href="fgene-16-1502681-g004.tif"/>
</fig>
<p>Furthermore, we extracted some scattered contigs from the assembly results of HiFi reads, constructed a heat map of nucleotide similarity between pairwise combinations of genomic intervals (<xref ref-type="fig" rid="F4">Figure 4C</xref>), and found that the proportion of repeated sequences at one end was very high. Examining these contigs, we found that the depth at one end is very high (<xref ref-type="fig" rid="F4">Figure 4D</xref>), which may be the reason for the poor continuity of the genome. We then explored these duplications and found that they were all caused by SINE/5S amplification, likely associated with the most recent SINEs expansion (<xref ref-type="sec" rid="s12">Supplementary Figure S6</xref>).</p>
</sec>
<sec id="s2-5">
<title>Chromosomal structure and evolutionary patterns in osmeriformes</title>
<p>Reconstruction of the ancestral Osmeriformes karyotype identified 25 ancestral chromosomes (<xref ref-type="fig" rid="F5">Figure 5A</xref>), which aligns with prior estimates that the ancestral chromosomes of bony fish were 24 or 25 (<xref ref-type="bibr" rid="B43">Muffato et al., 2023</xref>; <xref ref-type="bibr" rid="B44">Nakatani et al., 2007</xref>). During the differentiation into <italic>O. eperlanus</italic>, chromosomes 7, 8, 13, and 14 experienced structural breakage; chromosomes 25 and 14 fused, resulting in the 28 chromosomes characteristic of Osmeridae. Additionally, chromosomes 5 and 6 experienced extensive recombination, as did chromosomes 23 and 14. In the lineage leading to <italic>H. nipponensis</italic>, beyond the aforementioned changes, chromosomes 8 and 3 recombined, and chromosome 23 further recombined with chromosome 18. <italic>Hypomesus transpacificus</italic> and <italic>H. nipponensis</italic> show similar recombination, but the reason why it has only 26 chromosomes may be due to its failure to assemble all 28 chromosomes, two of which are indistinguishable (<xref ref-type="bibr" rid="B37">Kitada et al., 1981</xref>). These chromosomal rearrangements&#x2014;including breakages, fusions, and recombination events&#x2014;appear to be key drivers in the speciation within Osmeriformes. The conserved ancestral karyotype of 25 chromosomes provides a stable basis, while lineage-specific structural modifications have likely facilitated ecological diversification and reproductive isolation. These mechanisms are consistent with observations in other teleosts, where chromosomal evolution has been linked to speciation processes (<xref ref-type="bibr" rid="B36">Kirkpatrick, 2010</xref>; <xref ref-type="bibr" rid="B50">Parey et al., 2022</xref>).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>
<bold>(A)</bold> Ancestral karyotype and chromosomal evolution of various Salmoninae species. <bold>(B)</bold> Collinearity analysis of <italic>Hypomesus nipponensis</italic>, <italic>Esox lucius</italic>, and <italic>O. eperlanus</italic> genomes.</p>
</caption>
<graphic xlink:href="fgene-16-1502681-g005.tif"/>
</fig>
<p>A collinearity map of <italic>E. lucius</italic>, <italic>O. eperlanus</italic>, <italic>H. nipponensis,</italic> and <italic>H. transpacificus</italic> revealed distinct chromosomal recombination patterns, with darker regions indicating higher recombination activity (<xref ref-type="fig" rid="F5">Figure 5B</xref>). Notably, chromosomes 6, 8, 9, 12, 13, 15, and 21 of <italic>H. nipponensis</italic> and the corresponding chromosomes of <italic>O. eperlanus</italic> have all undergone recombination events. Furthermore, genomic comparisons indicate that most chromosomes of <italic>H. nipponensis</italic> and <italic>H. transpacificus</italic> have undergone recombination in the past 20&#xa0;million years (<xref ref-type="sec" rid="s12">Supplementary Figure S10</xref>). Among them, chromosomes 21, 20, and 12 exhibit both inversions and translocations, and chromosome 23 has undergone multiple inversions and translocations. In addition, the density of repetitive sequences at both ends of the inversion fragment on chromosome 12 of <italic>H. nipponensis</italic> is very high, as is that on chromosome 23 (<xref ref-type="sec" rid="s12">Supplementary Figure S11</xref>), suggesting that these inversion events are likely mediated by repetitive sequences. Chromosomal recombination is a major driver of genomic evolution, often contributing to species diversification and adaptation. The observed chromosomal recombination events, including inversions and translocations, are likely more than mere structural variations; they may play a crucial role in the adaptive evolution of <italic>H. nipponensis</italic>. Enhanced recombination activity could generate novel gene combinations, thereby accelerating adaptive responses to environmental challenges such as rising water temperatures and habitat variability. Indeed, previous studies have linked chromosomal rearrangements to ecological diversification and the evolution of temperature tolerance in teleosts (<xref ref-type="bibr" rid="B18">Donelson et al., 2012</xref>; <xref ref-type="bibr" rid="B63">Wellenreuther and Bernatchez, 2018</xref>). In the case of <italic>H. nipponensis</italic>, the dynamic genomic architecture&#x2014;evidenced by high recombination and the association with repetitive elements&#x2014;may underlie its ability to thrive in diverse and changing environments. Furthermore, the correlation between regions of high repetitive sequence density and inversion breakpoints supports the hypothesis that transposable elements contribute to chromosomal instability, which in turn may facilitate rapid adaptation (<xref ref-type="bibr" rid="B11">Chalopin et al., 2015</xref>; <xref ref-type="bibr" rid="B54">Shao et al., 2019</xref>). These chromosomal modifications not only promote species diversification but also potentially enhance the capacity of <italic>H. nipponensis</italic> to adjust to environmental stressors, such as increasing temperatures and fluctuating ecological conditions.</p>
<p>We identified 42,174 segments in <italic>Hypomesus</italic>, with a total length of 96&#xa0;Mb, that are absent in <italic>O. eperlanus</italic>, and 48,296 segments in <italic>O. eperlanus</italic>, with a total length of 108&#xa0;Mb, that are absent in <italic>Hypomesus</italic> (<xref ref-type="sec" rid="s12">Supplementary Figure S12</xref>). Gene enrichment in the <italic>O. eperlanus</italic> PAVs regions revealed that they were significantly enriched in the immunological memory process pathway (GO:0090713) and the positive regulation of the protein secretion pathway (GO:0050714) (<xref ref-type="sec" rid="s12">Supplementary Figure S13</xref>). The comparative genomic analysis highlights the genomic diversity within Osmeriformes and provides insights into the adaptive evolution of these species. The enrichment of immune-related pathways in <italic>O. eperlanus</italic> PAVs suggests that its migratory lifestyle, which likely exposes the species to diverse pathogens and environmental stressors, has driven the expansion of immune-related genes. This finding is consistent with studies in migratory fish, where immune system adaptations are critical for survival in variable environments (<xref ref-type="bibr" rid="B27">Hopkins II and Warren, 2005</xref>). The identification of PAVs also underscores the importance of structural variations in shaping species-specific traits and ecological adaptations.</p>
</sec>
</sec>
<sec sec-type="materials|methods" id="s3">
<title>Materials and methods</title>
<sec id="s3-1">
<title>Sample collection, library construction, and genome sequencing</title>
<p>
<italic>Hypomesus nipponensis</italic> muscle sample was collected from the Yalu River in Dandong (N 40.51, E 124.97), China, for whole genome sequencing. High-quality genomic DNA was extracted using optimized Cetyl Trimethyl Ammonium Bromide protocol. A 15&#xa0;kb SMRTbell library was constructed using the SMRTbell Express Template Prep Kit 3.0 (Pacific Biosciences, CA, United States), following standard protocols for DNA shearing, damage repair, end repair, hairpin adapter ligation, size selection, and purification. The library was sequenced on the PacBio Revio platform (25 M SMRT Cell) in Circular Consensus Sequencing (CCS) mode to generate high-fidelity (HiFi) reads with greater than 99.9% accuracy.</p>
</sec>
<sec id="s3-2">
<title>Genome size estimation</title>
<p>KmerGenie version 1.7051 (<xref ref-type="bibr" rid="B15">Chikhi and Medvedev, 2014</xref>) was used to perform <italic>k</italic>-mer counting and determine the optimal <italic>k</italic>-mer size for downstream analysis. The optimal <italic>k</italic>-mer size was estimated to be 119&#xa0;bp, as it provided a balance between read coverage and specificity for accurate <italic>k</italic>-mer profiling. The <italic>k</italic>-mer frequency output from KmerGenie was then used as input to GenomeScope (<xref ref-type="bibr" rid="B61">Vurture et al., 2017</xref>) for genome size estimation.</p>
</sec>
<sec id="s3-3">
<title>Genome assembly</title>
<p>Before assembly, we conducted a quality control on the PacBio HiFi reads. Reads with a median quality below Q20 were filtered out, and residual adapter sequences were trimmed. Only high-quality reads that passed these filters were used for the <italic>de novo</italic> genome assembly. This ensured the accuracy and reliability of the final assembly.</p>
<p>First, we used hifiasm v0.19.8 (<xref ref-type="bibr" rid="B14">Cheng et al., 2022</xref>) to assemble the HiFi reads, which are high-fidelity long reads generated through CCS. To further improve assembly continuity, we integrated PacBio CLR data (PRJNA672783 from NCBI) followed by additional assembly using NextDenovo v2.5.2 (<xref ref-type="bibr" rid="B29">Hu et al., 2023</xref>). We integrated the two assemblies using QuickMerge v0.3 (<xref ref-type="bibr" rid="B10">Chakraborty et al., 2016</xref>) with the hifiasm assembly as the reference. The merged sequence was then polished with NextPolish v1.4.1 (<xref ref-type="bibr" rid="B28">Hu et al., 2020</xref>). Following this, we removed redundancy using Purge-Dups v1.2.6 (<xref ref-type="bibr" rid="B24">Guan et al., 2020</xref>) and minimap2 v2.26 (<xref ref-type="bibr" rid="B38">Li, 2021</xref>) and subsequently employed masurca v4.1.0 (<xref ref-type="bibr" rid="B70">Zimin et al., 2013</xref>) to construct scaffolds. Finally, we constructed pseudochromosomes by anchoring scaffolds to the reference genomes of <italic>H. transpacificus</italic> (GCF_021917145.1) and <italic>O. eperlanus</italic> (GCF_963692335.1) using Ragtag v2.1 (<xref ref-type="bibr" rid="B2">Alonge et al., 2019</xref>). We employed TGS-GapCloser v1.2.1 (<xref ref-type="bibr" rid="B66">Xu et al., 2020</xref>) to close assembly gaps using HiFi and CLR data (<xref ref-type="sec" rid="s12">Supplementary Figure S1</xref>). To assess genome completeness, we used BUSCO v5.4.6 (<xref ref-type="bibr" rid="B56">Sim&#xe3;o et al., 2015</xref>) with the actinopterygii_odb10 lineage database (<ext-link ext-link-type="uri" xlink:href="https://busco-data.ezlab.org/v5/data/lineages/actinopterygii_odb10.2024-01-08.tar.gz">https://busco-data.ezlab.org/v5/data/lineages/actinopterygii_odb10.2024-01-08.tar.gz</ext-link>) as a reference.</p>
</sec>
<sec id="s3-4">
<title>Repeat identification</title>
<p>We predicted repeat elements using both <italic>de novo</italic> and homology-based annotations. RepeatModeler v2.0.6 (<xref ref-type="bibr" rid="B22">Flynn et al., 2020</xref>) and EDTA v2.2.2 (<xref ref-type="bibr" rid="B49">Ou et al., 2019</xref>) were employed to perform <italic>de novo</italic> repeat prediction and construct a custom repeat library. Then, the two libraries were combined and used to annotate the assembled genome with RepeatMasker (<xref ref-type="bibr" rid="B13">Chen, 2004</xref>). For the homology-based prediction, the Repbase (<xref ref-type="bibr" rid="B30">Jurka et al., 2005</xref>) and Dfam (<xref ref-type="bibr" rid="B58">Storer et al., 2021</xref>) libraries were used with RepeatMasker to identify known repeat elements. Finally, data from both methods were integrated to produce a nonredundant repeat element set.</p>
<p>We calculated the Kimura substitution levels between repeat consensus sequences and their genomic copies using the calcDivergenceFromAlign.pl script, which is included in the RepeatMasker utility bundle. We generated repeat landscape plots with the R script Kimura_Distance_plot.R, leveraging the divsum output from calcDivergenceFromAlign.pl.</p>
</sec>
<sec id="s3-5">
<title>Gene annotation</title>
<p>To predict protein-coding genes, we employed a combination of homology-based, <italic>de novo</italic>, and transcriptome-based prediction methods. Protein sequences from nine representative teleost species&#x2014;<italic>D. rerio</italic> (zebrafish)<italic>, Tetraodon nigroviridis</italic> (pufferfish)<italic>, G. aculeatus</italic> (stickleback)<italic>, Oryzias latipes</italic> (medaka)<italic>, Salmo salar</italic> (salmon)<italic>, H. transpacificus</italic> (delta smelt)<italic>, T. rubripes</italic> (fugu)<italic>, Oreochromis niloticus</italic> (tilapia)<italic>, and E. lucius</italic> (northern pike)&#x2014;were retrieved from Ensembl (<xref ref-type="bibr" rid="B21">Flicek et al., 2014</xref>), Gene structures were subsequently inferred using miniprot v0.12 (<xref ref-type="bibr" rid="B39">Li, 2023</xref>). For <italic>de novo</italic> gene prediction, Augustus (<xref ref-type="bibr" rid="B57">Stanke et al., 2008</xref>) and BRAKER3 v3.0.8 (<xref ref-type="bibr" rid="B23">Gabriel et al., 2024</xref>) were used on the repeat-masked <italic>H. nipponensis</italic> genome. RNA-seq data (PRJNA672783) from NCBI (<xref ref-type="bibr" rid="B67">Xuan et al., 2021</xref>) were aligned to the <italic>H</italic>. <italic>nipponensis</italic> genome using Hisat2 v2.2.1 (D. <xref ref-type="bibr" rid="B33">Kim et al., 2019</xref>). Transcript assemblies and transcriptome-based prediction were using TransDecoder v5.7.1 (<xref ref-type="bibr" rid="B26">Haas et al., 2008</xref>), StringTie v2.2.1 (<xref ref-type="bibr" rid="B55">Shumate et al., 2022</xref>), and PASA v2.5.3. Finally, the gene models obtained from homology-based, <italic>de novo</italic>, and transcriptome-based predictions were integrated using EVidenceModeler v2.1.0 (<xref ref-type="bibr" rid="B26">Haas et al., 2008</xref>) to construct a unified and high-confidence gene set (<xref ref-type="sec" rid="s12">Supplementary Figure S1</xref>).</p>
<p>For functional annotation, BLASTp (<xref ref-type="bibr" rid="B3">Altschul et al., 1990</xref>) was used to align the predicted protein against five public databases, including SwissProt (<xref ref-type="bibr" rid="B7">Boeckmann et al., 2003</xref>), TrEMBL (<xref ref-type="bibr" rid="B7">Boeckmann et al., 2003</xref>), KEGG (<xref ref-type="bibr" rid="B31">Kanehisa and Goto, 2000</xref>), GO (<xref ref-type="bibr" rid="B16">Consortium, 2004</xref>) and InterPro (<xref ref-type="bibr" rid="B51">Paysan-Lafosse et al., 2023</xref>).</p>
<p>The completeness, consistency, and accuracy of the gene structure annotation were evaluated using three different strategies. First, BUSCO analysis was performed to assess the completeness of single-copy orthologs using the actinopterygii_odb10 database. Next, OMAK (<xref ref-type="bibr" rid="B45">Nevers et al., 2024</xref>) was applied to evaluate genome integrity and consistency by comparing the annotated protein sequences with those from the ancestors of Teleostei. Additionally, the length distributions of mRNA, CDS, and exons were compared among <italic>D. rerio</italic>, <italic>E</italic>. <italic>lucius</italic>, <italic>T</italic>. <italic>rubripes</italic>, <italic>G. aculeatus</italic>, and <italic>H. transpacificus</italic> to assess gene structural similarity and conservation.</p>
</sec>
<sec id="s3-6">
<title>Orthology and phylogenomics</title>
<p>To investigate evolutionary relationships, <italic>H</italic>. <italic>nipponensis</italic> and ten other fish species&#x2014;<italic>D</italic>. <italic>rerio, S</italic>. <italic>salar, E. lucius, T</italic>. <italic>rubripes, C</italic>. <italic>milii, O</italic>. <italic>sinensis, H. transpacificus, O. eperlanus, P</italic>. <italic>chinensis, N</italic>. <italic>taihuensis</italic>&#x2014;were selected for orthology analysis. To reduce redundancy, only the longest isoform per gene was retained in each species&#x2019; protein set. Orthology inference was performed using Orthofinder v2.5.4 (<xref ref-type="bibr" rid="B20">Emms and Kelly, 2019</xref>), which identified orthologs, paralogs, and co-orthologs across species. For phylogenetic tree construction, we first extracted 586 single-copy orthologs, aligned their corresponding CDS sequences individually using MUSCLE v5.1 (<xref ref-type="bibr" rid="B19">Edgar, 2022</xref>), and then concatenated them into a supergene matrix. Iqtree v2.3.3 (<xref ref-type="bibr" rid="B46">Nguyen et al., 2015</xref>) was applied to construct a phylogenetic tree with the maximum-likelihood method, with 100 bootstrap replicates. Species divergence times were estimated with MCMCTree (<xref ref-type="bibr" rid="B4">Alvarez-Carretero et al., 2022</xref>), a component of PAML (<xref ref-type="bibr" rid="B68">Yang, 2007</xref>), employing the parameters &#x2018;RootAge &#x2264;500, model &#x3d; F81, alpha &#x3d; 1, clock &#x3d; 3&#x2032;and calibration points for <italic>C</italic>. <italic>milii</italic> and <italic>D</italic>. <italic>rerio</italic> (440&#x2013;495&#xa0;Mya) (<xref ref-type="bibr" rid="B6">Betancur et al., 2013</xref>); <italic>D</italic>. <italic>rerio</italic> and <italic>S</italic>. <italic>salar</italic> (180Mya-251.5Mya) (<xref ref-type="bibr" rid="B42">Meynard et al., 2012</xref>)<italic>; S</italic>. <italic>salar</italic> and <italic>H</italic>. <italic>transpacificus</italic> (176Mya-264Mya) (<xref ref-type="bibr" rid="B53">Rabosky et al., 2018</xref>). To visualize the consistency between the genomes of <italic>H. nipponensis</italic> and its closely related species, <italic>O. eperlanus</italic> and <italic>E. lucius</italic>, the 28&#xa0;<italic>H. nipponensis</italic> chromosomes were aligned with <italic>O. eperlanus</italic> and <italic>E. lucius</italic> chromosomes by MCScanX (<xref ref-type="bibr" rid="B62">Wang et al., 2012</xref>).</p>
</sec>
<sec id="s3-7">
<title>Expansion and contraction of gene families</title>
<p>We assessed gene family expansion and contraction in <italic>H. nipponensis</italic> by comparing cluster size differences with ten other fish species using CAFE5 (<xref ref-type="bibr" rid="B41">Mendes et al., 2021</xref>). We applied a stochastic birth and death model to investigate changes in gene family size along each lineage of the phylogenetic tree. A gamma model was used to estimate the probability of transitions in gene family size between parent and child nodes. We calculated P-values for each lineage using conditional likelihoods as test statistics, with a P-value less than 0.05 indicating significant gene family expansion or contraction. We annotated the protein sequences of <italic>H. nipponensis</italic> with EggNOG mapper (<xref ref-type="bibr" rid="B9">Cantalapiedra et al., 2021</xref>) as background genes, extracted all significantly expanded or contracted gene families, performed GO enrichment analysis, and generated plots using TBtools (<xref ref-type="bibr" rid="B12">Chen et al., 2023</xref>).</p>
</sec>
<sec id="s3-8">
<title>Reconstruction of ancestral karyotype</title>
<p>A total of four species&#x2014;<italic>E. lucius, O. eperlanus, H. nipponensis,</italic> and <italic>H. transpacificus</italic>&#x2014;were selected for the reconstruction of the ancestral karyotype. <italic>Esox lucius</italic> was adopted as a reference genome and BLAST was used for pairwise interspecies comparisons and reciprocal best hits to obtain a set of genes homologous between species. The corrected posterior binomial test (q-value &#x3c;0.05, homologous genes count &#x2265;20) was applied to detect chromosomes that were homologous among species. The default parameters of MCScanX were used for the identification of inter-chromosomal colinear blocks. Finally, ANGeS v1.01 (<xref ref-type="bibr" rid="B62">Wang et al., 2012</xref>) was used to construct the ancestral karyotype. Interspecies collinearity was displayed using Circos v0.69 (<xref ref-type="bibr" rid="B47">Northcutt, 2009</xref>).</p>
</sec>
<sec id="s3-9">
<title>Identification of PAVs</title>
<p>We initially constructed the genomes of <italic>H. nipponensis</italic> and <italic>H. transpacificus</italic> using ppsPCP (<xref ref-type="bibr" rid="B60">Tahir Ul Qamar et al., 2019</xref>). Next, we identified putative PAVs by aligning the <italic>Hypomesus</italic> genomes with <italic>O. eperlanus</italic> using Mummer v4.0.1 and extracting unaligned regions from the &#x201c;show-diff&#x201d; command. These sequences were then filtered by discarding those overlapping with gap regions in the respective genome. To identify putatively unique presence regions, the remaining sequences were filtered by aligning them with the other genome using BLASTN (E-value &#x2264; 1e-5). Sequences exhibiting high similarity (&#x2265;95%) and coverage (&#x2265;90%) were removed from the dataset.</p>
</sec>
</sec>
<sec id="s4">
<title>Conclusion and limitations</title>
<p>In this study, we report an improved chromosome-level draft genome of <italic>H</italic>. <italic>nipponensis</italic>, totaling 507.8&#xa0;Mb with a scaffold N50 of 20&#xa0;Mb, of which 96.6% of the assembly was anchored to 28 pseudochromosomes. Our analysis revealed substantial expansions in gene families involved in tissue-specific immune responses, antigen sampling in mucosa-associated lymphoid tissues, kidney morphogenesis, and skeletal muscle function&#x2014;particularly muscle filament sliding&#x2014;thereby highlighting potential genomic bases for physiological adaptation. Notably, the recent expansion of LINEs appears to be closely associated with gene family proliferation, suggesting that transposable elements may play a central role in genome remodeling. Furthermore, ancestral karyotype reconstruction of Osmeriformes inferred 25 ancestral chromosomes and subsequent synteny analyses revealed that lineage-specific chromosomal rearrangements (including inversions and translocations) were likely facilitated by the accumulation of repetitive sequences. These findings highlight the critical role of repeat-driven structural variation in shaping the genome architecture of <italic>H. nipponensis</italic>, contributing to both its evolutionary divergence and ecological adaptation.</p>
<p>However, despite the valuable insights gained, several methodological and interpretative limitations should be acknowledged. Most notably, the chromosome-level assembly was not supported by Hi-C chromatin conformation data but was instead constructed using a synteny-based anchoring strategy with closely related species. Although this approach facilitates the inference of large-scale chromosomal architecture, it lacks the resolution to capture long-range chromatin interactions and accurately determine scaffold orientation. This limitation may introduce potential misassemblies, particularly in regions rich in repetitive elements or structural complexity, thereby affecting the accuracy of inferred chromosomal recombination, inversion, and translocation events. In summary, while this study establishes a valuable genomic resource for <italic>H. nipponensis</italic> and advances our understanding of its evolutionary dynamics, the absence of chromatin conformation data (Hi-C) and experimental validation introduces uncertainty regarding both genome assembly accuracy and functional interpretation. Future studies incorporating Hi-C sequencing, long-read haplotype phasing, and functional genomic analyses will be essential to validate and refine these findings, ultimately advancing our understanding of genome evolution and adaptive diversification in Osmeriformes.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The data presented in this study are deposited in the CNCB Sequence Archive of the China National Center for Bioinformation, under the BioProject accession number PRJCA024905. The chromosome-level genome assembly and annotation data are available in the Zenodo repository under accession number GWHETSC00000000.2, accessible at <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.14868385">https://doi.org/10.5281/zenodo.14868385</ext-link>. All other data generated or analyzed during this study are included in the manuscript and its <xref ref-type="sec" rid="s12">Supplementary Materials</xref>.</p>
</sec>
<sec sec-type="ethics-statement" id="s6">
<title>Ethics statement</title>
<p>The animal study was approved by Chinese Academy of Fishery Sciences (CAFS). The study was conducted in accordance with the local legislation and institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>CZ: Conceptualization, Data curation, Formal Analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review and editing. YK: Conceptualization, Funding acquisition, Methodology, Project administration, Resources, Supervision, Validation, Writing &#x2013; review and editing. ZL: Data curation, Investigation, Writing &#x2013; review and editing. FT: Data curation, Funding acquisition, Investigation, Project administration, Resources, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported by the Central level Nonprofit Scientific Research Institutes Special Fund of China (grant number: 2023TD07); and the National Key R&#x26;D Program of China (grant number: 2022YFD2400101).</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2025.1502681/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2025.1502681/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="DataSheet1.docx" id="SM2" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alonge</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lebeigle</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kirsche</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jenike</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ou</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Aganezov</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Automated assembly scaffolding using RagTag elevates a new tomato system for high-throughput genome editing</article-title>. <source>Genome Biol.</source> <volume>23</volume> (<issue>1</issue>), <fpage>258</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-022-02823-7</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alonge</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Soyk</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ramakrishnan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Goodwin</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sedlazeck</surname>
<given-names>F. J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>RaGOO: fast and accurate reference-guided scaffolding of draft genomes</article-title>. <source>Genome Biol.</source> <volume>20</volume> (<issue>1</issue>), <fpage>224</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1829-6</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altschul</surname>
<given-names>S. F.</given-names>
</name>
<name>
<surname>Gish</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Miller</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Myers</surname>
<given-names>E. W.</given-names>
</name>
<name>
<surname>Lipman</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>1990</year>). <article-title>Basic local alignment search tool</article-title>. <source>J. Mol. Biol.</source> <volume>215</volume> (<issue>3</issue>), <fpage>403</fpage>&#x2013;<lpage>410</lpage>. <pub-id pub-id-type="doi">10.1016/s0022-2836(05)80360-2</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alvarez-Carretero</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tamuri</surname>
<given-names>A. U.</given-names>
</name>
<name>
<surname>Battini</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nascimento</surname>
<given-names>F. F.</given-names>
</name>
<name>
<surname>Carlisle</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Asher</surname>
<given-names>R. J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>A species-level timeline of mammal evolution integrating phylogenomic data</article-title>. <source>Nature</source> <volume>602</volume>, <fpage>263</fpage>&#x2013;<lpage>267</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-021-04341-1</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Asami</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Early life ecology of Japanese smelt (Hypomesus nipponensis) in Lake Abashiri, a brackish water, eastern Hokkaido, Japan</article-title>. <source>Sci. Rep. Hokkaido Fish. Res. Inst.</source> (<issue>67</issue>).</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Betancur</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Broughton</surname>
<given-names>R. E.</given-names>
</name>
<name>
<surname>Wiley</surname>
<given-names>E. O.</given-names>
</name>
<name>
<surname>Carpenter</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lopez</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>The tree of life and a new classification of bony fishes</article-title>. <source>PLoS Curr.</source> <volume>5</volume>. <pub-id pub-id-type="doi">10.1371/currents.tol.53ba26640df0ccaee75bb165c8c26288</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boeckmann</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bairoch</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Apweiler</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Blatter</surname>
<given-names>M.-C.</given-names>
</name>
<name>
<surname>Estreicher</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gasteiger</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2003</year>). <article-title>The SWISS-PROT protein knowledgebase and its supplement TrEMBL in 2003</article-title>. <source>Nucleic Acids Res.</source> <volume>31</volume> (<issue>1</issue>), <fpage>365</fpage>&#x2013;<lpage>370</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkg095</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Braasch</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Gehrke</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Kawasaki</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Manousaki</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Pasquier</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>The spotted gar genome illuminates vertebrate evolution and facilitates human-teleost comparisons</article-title>. <source>Nat. Genet.</source> <volume>48</volume> (<issue>4</issue>), <fpage>427</fpage>&#x2013;<lpage>437</lpage>. <pub-id pub-id-type="doi">10.1038/ng.3526</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cantalapiedra</surname>
<given-names>C. P.</given-names>
</name>
<name>
<surname>Hernandez-Plaza</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Letunic</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Bork</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Huerta-Cepas</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>eggNOG-mapper v2: functional annotation, orthology assignments, and domain prediction at the metagenomic scale</article-title>. <source>Mol. Biol. Evol.</source> <volume>38</volume> (<issue>12</issue>), <fpage>5825</fpage>&#x2013;<lpage>5829</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msab293</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chakraborty</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Baldwin-Brown</surname>
<given-names>J. G.</given-names>
</name>
<name>
<surname>Long</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Emerson</surname>
<given-names>J. J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Contiguous and accurate <italic>de novo</italic> assembly of metazoan genomes with modest long read coverage</article-title>. <source>Nucleic Acids Res.</source> <volume>44</volume> (<issue>19</issue>), <fpage>e147</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw654</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chalopin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Naville</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Plard</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Galiana</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Volff</surname>
<given-names>J. N.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Comparative analysis of transposable elements highlights mobilome diversity and evolution in vertebrates</article-title>. <source>Genome Biol. Evol.</source> <volume>7</volume> (<issue>2</issue>), <fpage>567</fpage>&#x2013;<lpage>580</lpage>. <pub-id pub-id-type="doi">10.1093/gbe/evv005</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>TBtools-II: a &#x201c;one for all, all for one&#x201d; bioinformatics platform for biological big-data mining</article-title>. <source>Mol. Plant</source> <volume>16</volume> (<issue>11</issue>), <fpage>1733</fpage>&#x2013;<lpage>1742</lpage>. <pub-id pub-id-type="doi">10.1016/j.molp.2023.09.010</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Using repeat masker to identify repetitive elements in genomic sequences</article-title>. <source>Curr. Protoc. Bioinforma.</source> <volume>5</volume> (<issue>1</issue>), Unit 4.10. <pub-id pub-id-type="doi">10.1002/0471250953.bi0410s05</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jarvis</surname>
<given-names>E. D.</given-names>
</name>
<name>
<surname>Fedrigo</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Koepfli</surname>
<given-names>K.-P.</given-names>
</name>
<name>
<surname>Urban</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gemmell</surname>
<given-names>N. J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Haplotype-resolved assembly of diploid genomes without parental data</article-title>. <source>Nat. Biotechnol.</source> <volume>40</volume> (<issue>9</issue>), <fpage>1332</fpage>&#x2013;<lpage>1335</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-022-01261-x</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chikhi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Medvedev</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Informed and automated k-mer size selection for genome assembly</article-title>. <source>Bioinformatics</source> <volume>30</volume> (<issue>1</issue>), <fpage>31</fpage>&#x2013;<lpage>37</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btt310</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Consortium</surname>
<given-names>G. O.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>The Gene Ontology (GO) database and informatics resource</article-title>. <source>Nucleic Acids Res.</source> <volume>32</volume> (<issue>90001</issue>), <fpage>258D</fpage>&#x2013;<lpage>261D</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkh036</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>David</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Blum</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Feldman</surname>
<given-names>M. W.</given-names>
</name>
<name>
<surname>Lavi</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Hillel</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Recent duplication of the common carp (<italic>Cyprinus carpio</italic> L.) genome as revealed by analyses of microsatellite loci</article-title>. <source>Mol. Biol. Evol.</source> <volume>20</volume> (<issue>9</issue>), <fpage>1425</fpage>&#x2013;<lpage>1434</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msg173</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Donelson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Munday</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>McCormick</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pitcher</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Rapid transgenerational acclimation of a tropical reef fish to climate change</article-title>. <source>Nat. Clim. Chang.</source> <volume>2</volume> (<issue>1</issue>), <fpage>30</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1038/nclimate1323</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Edgar</surname>
<given-names>R. C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Muscle5: high-accuracy alignment ensembles enable unbiased assessments of sequence homology and phylogeny</article-title>. <source>Nat. Commun.</source> <volume>13</volume> (<issue>1</issue>), <fpage>6968</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-022-34630-w</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Emms</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Kelly</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>OrthoFinder: phylogenetic orthology inference for comparative genomics</article-title>. <source>Genome Biol.</source> <volume>20</volume> (<issue>1</issue>), <fpage>238</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1832-y</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Flicek</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Amode</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Barrell</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Beal</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Billis</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Brent</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Ensembl 2014</article-title>. <source>Nucleic Acids Res.</source> <volume>42</volume> (<issue>D1</issue>), <fpage>D749</fpage>&#x2013;<lpage>D755</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkt1196</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Flynn</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Hubley</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Goubert</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Rosen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Clark</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Feschotte</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>RepeatModeler2 for automated genomic discovery of transposable element families</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>117</volume> (<issue>17</issue>), <fpage>9451</fpage>&#x2013;<lpage>9457</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1921046117</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gabriel</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Br&#x16f;na</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Hoff</surname>
<given-names>K. J.</given-names>
</name>
<name>
<surname>Ebel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lomsadze</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Borodovsky</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>BRAKER3: fully automated genome annotation using RNA-seq and protein evidence with GeneMark-ETP, AUGUSTUS, and TSEBRA</article-title>. <source>Genome Res.</source> <volume>34</volume> (<issue>5</issue>), <fpage>769</fpage>&#x2013;<lpage>777</lpage>. <pub-id pub-id-type="doi">10.1101/gr.278090.123</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guan</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>McCarthy</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Wood</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Howe</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Durbin</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Identifying and removing haplotypic duplication in primary genome assemblies</article-title>. <source>Bioinformatics</source> <volume>36</volume> (<issue>9</issue>), <fpage>2896</fpage>&#x2013;<lpage>2898</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa025</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gui</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X. Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Rethinking fish biology and biotechnologies in the challenge era for burgeoning genome resources and strengthening food security</article-title>. <source>WBS</source> <volume>1</volume> (<issue>1</issue>), <fpage>100002</fpage>. <pub-id pub-id-type="doi">10.1016/j.watbs.2021.11.001</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haas</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Salzberg</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Pertea</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Allen</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Orvis</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Automated eukaryotic gene structure annotation using EVidenceModeler and the Program to Assemble Spliced Alignments</article-title>. <source>Genome Biol.</source> <volume>9</volume> (<issue>1</issue>), <fpage>R7</fpage>. <pub-id pub-id-type="doi">10.1186/gb-2008-9-1-r7</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hopkins</surname>
<given-names>R. L.</given-names>
<suffix>II</suffix>
</name>
<name>
<surname>Warren</surname>
<given-names>M. L. J. Y.</given-names>
<suffix>Jr</suffix>
</name>
</person-group> (<year>2005</year>). <source>Osmeridae: smelts</source>, <volume>400</volume>, <fpage>800</fpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>NextPolish: a fast and efficient genome polishing tool for long-read assembly</article-title>. <source>Bioinformatics</source> <volume>36</volume> (<issue>7</issue>), <fpage>2253</fpage>&#x2013;<lpage>2255</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz891</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ayoola</surname>
<given-names>A. O.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>An efficient error correction and accurate assembly tool for noisy long reads</article-title>. <source>Genome Biol.</source>, <fpage>2023</fpage>&#x2013;<lpage>2003</lpage>. <pub-id pub-id-type="doi">10.1186/s13059-024-03252-4</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jurka</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kapitonov</surname>
<given-names>V. V.</given-names>
</name>
<name>
<surname>Pavlicek</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Klonowski</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kohany</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Walichiewicz</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Repbase Update, a database of eukaryotic repetitive elements</article-title>. <source>Cytogenet. Genome Res.</source> <volume>110</volume> (<issue>1-4</issue>), <fpage>462</fpage>&#x2013;<lpage>467</lpage>. <pub-id pub-id-type="doi">10.1159/000084979</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kanehisa</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Goto</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>KEGG: kyoto encyclopedia of genes and genomes</article-title>. <source>Nucleic Acids Res.</source> <volume>28</volume> (<issue>1</issue>), <fpage>27</fpage>&#x2013;<lpage>30</lpage>. <pub-id pub-id-type="doi">10.1093/nar/28.1.27</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kasahara</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Naruse</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Sasaki</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nakatani</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ahsan</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>The medaka draft genome and insights into vertebrate genome evolution</article-title>. <source>Nature</source> <volume>447</volume>, <fpage>714</fpage>&#x2013;<lpage>719</lpage>. <pub-id pub-id-type="doi">10.1038/nature05846</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Paggi</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bennett</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Salzberg</surname>
<given-names>S. L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Graph-based genome alignment and genotyping with HISAT2 and HISAT-genotype</article-title>. <source>Nat. Biotechnol.</source> <volume>37</volume> (<issue>8</issue>), <fpage>907</fpage>&#x2013;<lpage>915</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-019-0201-4</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ko</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Yoo</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Won</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Phillippy</surname>
<given-names>A. M.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>False gene and chromosome losses in genome assemblies caused by GC content variation and repeats</article-title>. <source>Appl. Microbiol. Biot.</source> <volume>23</volume> (<issue>1</issue>), <fpage>204</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-022-02765-0</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kimura</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>1980</year>). <article-title>A simple method for estimating evolutionary rates of base substitutions through comparative studies of nucleotide sequences</article-title>. <source>J. Mol. Evol.</source> <volume>16</volume> (<issue>2</issue>), <fpage>111</fpage>&#x2013;<lpage>120</lpage>. <pub-id pub-id-type="doi">10.1007/BF01731581</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kirkpatrick</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>How and why chromosome inversions evolve</article-title>. <source>PLoS Biol.</source> <volume>8</volume> (<issue>9</issue>), <fpage>e1000501</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pbio.1000501</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kitada</surname>
<given-names>J.-I.</given-names>
</name>
<name>
<surname>Tatewaki</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Tagawa</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>1981</year>). <article-title>Chromosomes of the pond smelt, Hypomesus transpacificus nipponensis</article-title>.</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>New strategies to improve minimap2 alignment accuracy</article-title>. <source>Bioinformatics</source> <volume>37</volume>, <fpage>4572</fpage>&#x2013;<lpage>4574</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btab705</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Protein-to-genome alignment with miniprot</article-title>. <source>Bioinformatics</source> <volume>39</volume> (<issue>1</issue>), <fpage>btad014</fpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btad014</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Logsdon</surname>
<given-names>G. A.</given-names>
</name>
<name>
<surname>Vollger</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Eichler</surname>
<given-names>E. E.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Long-read human genome sequencing and its applications</article-title>. <source>Nat. Rev. Genet.</source> <volume>21</volume> (<issue>10</issue>), <fpage>597</fpage>&#x2013;<lpage>614</lpage>. <pub-id pub-id-type="doi">10.1038/s41576-020-0236-x</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mendes</surname>
<given-names>F. K.</given-names>
</name>
<name>
<surname>Vanderpool</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Fulton</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hahn</surname>
<given-names>M. W.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>CAFE 5 models variation in evolutionary rates among gene families</article-title>. <source>Bioinformatics</source> <volume>36</volume> (<issue>22-23</issue>), <fpage>5516</fpage>&#x2013;<lpage>5518</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa1022</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meynard</surname>
<given-names>C. N.</given-names>
</name>
<name>
<surname>Mouillot</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Mouquet</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Douzery</surname>
<given-names>E. J.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>A phylogenetic perspective on the evolution of Mediterranean teleost fishes</article-title>. <source>PLoS One</source> <volume>7</volume> (<issue>5</issue>), <fpage>e36443</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0036443</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Muffato</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Louis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>N. T. T.</given-names>
</name>
<name>
<surname>Lucas</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Berthelot</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Crollius</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Reconstruction of hundreds of reference ancestral genomes across the eukaryotic kingdom</article-title>. <source>Nat. Ecol. Evol.</source> <volume>7</volume> (<issue>3</issue>), <fpage>355</fpage>&#x2013;<lpage>366</lpage>. <pub-id pub-id-type="doi">10.1038/s41559-022-01956-z</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nakatani</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Takeda</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kohara</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Morishita</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Reconstruction of the vertebrate ancestral genome reveals dynamic genome reorganization in early vertebrates</article-title>. <source>Genome Res.</source> <volume>17</volume> (<issue>9</issue>), <fpage>1254</fpage>&#x2013;<lpage>1265</lpage>. <pub-id pub-id-type="doi">10.1101/gr.6316407</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nevers</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Warwick Vesztrocy</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rossier</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Train</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Altenhoff</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dessimoz</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Quality assessment of gene repertoire annotations with OMArk</article-title>. <source>Nat. Biotechnol.</source> <volume>43</volume>, <fpage>124</fpage>&#x2013;<lpage>133</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-024-02147-w</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>L. T.</given-names>
</name>
<name>
<surname>Schmidt</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>von Haeseler</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Minh</surname>
<given-names>B. Q.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>IQ-TREE: a fast and effective stochastic algorithm for estimating maximum-likelihood phylogenies</article-title>. <source>Mol. Biol. Evol.</source> <volume>32</volume> (<issue>1</issue>), <fpage>268</fpage>&#x2013;<lpage>274</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msu300</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Northcutt</surname>
<given-names>R. G.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Telencephalic organization in the spotted African Lungfish, <italic>Protopterus dolloi</italic>: a new cytological model</article-title>. <source>Brain Behav. Evol.</source> <volume>73</volume> (<issue>1</issue>), <fpage>59</fpage>&#x2013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1159/000204963</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nurk</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Koren</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rhie</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rautiainen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bzikadze</surname>
<given-names>A. V.</given-names>
</name>
<name>
<surname>Mikheenko</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>The complete sequence of a human genome</article-title>. <source>Science</source> <volume>376</volume> (<issue>6588</issue>), <fpage>44</fpage>&#x2013;<lpage>53</lpage>. <pub-id pub-id-type="doi">10.1126/science.abj6987</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ou</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chougule</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Agda</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Hellinga</surname>
<given-names>A. J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Benchmarking transposable element annotation methods for creation of a streamlined, comprehensive pipeline</article-title>. <source>Genome Biol.</source> <volume>20</volume>, <fpage>275</fpage>&#x2013;<lpage>318</lpage>. <pub-id pub-id-type="doi">10.1186/s13059-019-1905-y</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Parey</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Louis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Montfort</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guiguen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Crollius</surname>
<given-names>H. R.</given-names>
</name>
<name>
<surname>Berthelot</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>An atlas of fish genome evolution reveals delayed rediploidization following the teleost whole-genome duplication</article-title>. <source>Genome Res.</source> <volume>32</volume> (<issue>9</issue>), <fpage>1685</fpage>&#x2013;<lpage>1697</lpage>. <pub-id pub-id-type="doi">10.1101/gr.276953.122</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paysan-Lafosse</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Blum</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chuguransky</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Grego</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Pinto</surname>
<given-names>B. L.</given-names>
</name>
<name>
<surname>Salazar</surname>
<given-names>G. A.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>InterPro in 2022</article-title>. <source>Nucleic Acids Res.</source> <volume>51</volume> (<issue>D1</issue>), <fpage>D418</fpage>&#x2013;<lpage>D427</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkac993</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peichel</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Nereng</surname>
<given-names>K. S.</given-names>
</name>
<name>
<surname>Ohgi</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Cole</surname>
<given-names>B. L.</given-names>
</name>
<name>
<surname>Colosimo</surname>
<given-names>P. F.</given-names>
</name>
<name>
<surname>Buerkle</surname>
<given-names>C. A.</given-names>
</name>
<etal/>
</person-group> (<year>2001</year>). <article-title>The genetic architecture of divergence between threespine stickleback species</article-title>. <source>Nature</source> <volume>414</volume>, <fpage>901</fpage>&#x2013;<lpage>905</lpage>. <pub-id pub-id-type="doi">10.1038/414901a</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rabosky</surname>
<given-names>D. L.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Title</surname>
<given-names>P. O.</given-names>
</name>
<name>
<surname>Cowman</surname>
<given-names>P. F.</given-names>
</name>
<name>
<surname>Sallan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Friedman</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>An inverse latitudinal gradient in speciation rate for marine fishes</article-title>. <source>Nature</source> <volume>559</volume> (<issue>7714</issue>), <fpage>392</fpage>&#x2013;<lpage>395</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-018-0273-1</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shao</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Evolution and diversity of transposable elements in fish genomes</article-title>. <source>Sci. Rep.</source> <volume>9</volume> (<issue>1</issue>), <fpage>15399</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-019-51888-1</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shumate</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Pertea</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Pertea</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Improved transcriptome assembly using a hybrid of long and short reads with StringTie</article-title>. <source>PLoS Comput. Biol.</source> <volume>18</volume> (<issue>6</issue>), <fpage>e1009730</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1009730</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sim&#xe3;o</surname>
<given-names>F. A.</given-names>
</name>
<name>
<surname>Waterhouse</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Ioannidis</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kriventseva</surname>
<given-names>E. V.</given-names>
</name>
<name>
<surname>Zdobnov</surname>
<given-names>E. M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>BUSCO: assessing genome assembly and annotation completeness with single-copy orthologs</article-title>. <source>Bioinformatics</source> <volume>31</volume> (<issue>19</issue>), <fpage>3210</fpage>&#x2013;<lpage>3212</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv351</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stanke</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Diekhans</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Baertsch</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Haussler</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Using native and syntenically mapped cDNA alignments to improve <italic>de novo</italic> gene finding</article-title>. <source>Bioinformatics</source> <volume>24</volume> (<issue>5</issue>), <fpage>637</fpage>&#x2013;<lpage>644</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btn013</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Storer</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hubley</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Rosen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wheeler</surname>
<given-names>T. J.</given-names>
</name>
<name>
<surname>Smit</surname>
<given-names>A. F.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The Dfam community resource of transposable element families, sequence models, and genome annotations</article-title>. <source>Mob. DNA</source> <volume>12</volume> (<issue>1</issue>), <fpage>2</fpage>. <pub-id pub-id-type="doi">10.1186/s13100-020-00230-y</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Swanson</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Reid</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Young</surname>
<given-names>P. S.</given-names>
</name>
<name>
<surname>Cech</surname>
<given-names>J. J.</given-names>
<suffix>Jr.</suffix>
</name>
</person-group> (<year>2000</year>). <article-title>Comparative environmental tolerances of threatened delta smelt (Hypomesus transpacificus) and introduced wakasagi (H. nipponensis) in an altered California estuary</article-title>. <source>Oecologia</source> <volume>123</volume> (<issue>3</issue>), <fpage>384</fpage>&#x2013;<lpage>390</lpage>. <pub-id pub-id-type="doi">10.1007/s004420051025</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tahir Ul Qamar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L. L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>ppsPCP: a plant presence/absence variants scanner and pan-genome construction pipeline</article-title>. <source>Bioinformatics</source> <volume>35</volume> (<issue>20</issue>), <fpage>4156</fpage>&#x2013;<lpage>4158</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz168</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vurture</surname>
<given-names>G. W.</given-names>
</name>
<name>
<surname>Sedlazeck</surname>
<given-names>F. J.</given-names>
</name>
<name>
<surname>Nattestad</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Underwood</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gurtowski</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>GenomeScope: fast reference-free genome profiling from short reads</article-title>. <source>Bioinformatics</source> <volume>33</volume> (<issue>14</issue>), <fpage>2202</fpage>&#x2013;<lpage>2204</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx153</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Debarry</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>MCScanX: a toolkit for detection and evolutionary analysis of gene synteny and collinearity</article-title>. <source>Nucleic Acids Res.</source> <volume>40</volume> (<issue>7</issue>), <fpage>e49</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkr1293</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wellenreuther</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bernatchez</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Eco-evolutionary genomics of chromosomal inversions</article-title>. <source>TRENDS Ecol. Evol.</source> <volume>33</volume> (<issue>6</issue>), <fpage>427</fpage>&#x2013;<lpage>440</lpage>. <pub-id pub-id-type="doi">10.1016/j.tree.2018.04.002</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wenger</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Peluso</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Rowell</surname>
<given-names>W. J.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>P. C.</given-names>
</name>
<name>
<surname>Hall</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Concepcion</surname>
<given-names>G. T.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Accurate circular consensus long-read sequencing improves variant detection and assembly of a human genome</article-title>. <source>Nat. Biotechnol.</source> <volume>37</volume> (<issue>10</issue>), <fpage>1155</fpage>&#x2013;<lpage>1162</lpage>. <pub-id pub-id-type="doi">10.1038/s41587-019-0217-9</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X. J. L. S.</given-names>
</name>
<etal/>
</person-group> (<year>1992</year>). <source>The fishes of genus Hypomesus and utilization of its resource</source>. <publisher-loc>Shenyang</publisher-loc>: <publisher-name>Liaoning Science and Technology Press</publisher-name>.</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Peters</surname>
<given-names>B. A.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>TGS-GapCloser: a fast and accurate gap closer for large genomes with low coverage of error-prone long reads</article-title>. <source>Gigascience</source> <volume>9</volume> (<issue>9</issue>), <fpage>giaa094</fpage>. <pub-id pub-id-type="doi">10.1093/gigascience/giaa094</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xuan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>You</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Nam</surname>
<given-names>B. H.</given-names>
</name>
<name>
<surname>Noh</surname>
<given-names>E. S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Draft genome of the Korean smelt Hypomesus nipponensis and its transcriptomic responses to heat stress in the liver and muscle</article-title>. <source>G3</source> <volume>11</volume> (<issue>9</issue>), <fpage>jkab147</fpage>. <pub-id pub-id-type="doi">10.1093/g3journal/jkab147</pub-id>
</citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>PAML 4: phylogenetic analysis by maximum likelihood</article-title>. <source>Mol. Biol. Evol.</source> <volume>24</volume> (<issue>8</issue>), <fpage>1586</fpage>&#x2013;<lpage>1591</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msm088</pub-id>
</citation>
</ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yin</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ni</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Fish assemblage shift after Japanese smelt (Hypomesus nipponensis McAllister, 1963) invasion in Lake Erhai, a subtropical plateau lake in China</article-title>. <source>Water</source> <volume>13</volume> (<issue>13</issue>), <fpage>1800</fpage>. <pub-id pub-id-type="doi">10.3390/w13131800</pub-id>
</citation>
</ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zimin</surname>
<given-names>A. V.</given-names>
</name>
<name>
<surname>Mar&#xe7;ais</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Puiu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Roberts</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Salzberg</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Yorke</surname>
<given-names>J. A.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>The MaSuRCA genome assembler</article-title>. <source>Bioinformatics</source> <volume>29</volume> (<issue>21</issue>), <fpage>2669</fpage>&#x2013;<lpage>2677</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btt476</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>