<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="brief-report">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Mar. Sci.</journal-id>
<journal-title>Frontiers in Marine Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Mar. Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-7745</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmars.2022.841561</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Marine Science</subject>
<subj-group>
<subject>Brief Research Report</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>AquaGWAS: A Genome-Wide Association Study Pipeline for Aquatic Animals and Its Application to Reference-Required and Reference-Free Genome-Wide Association Study for Abalone</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Deng</surname> <given-names>Chao</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Peng</surname> <given-names>Wenzhu</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Ma</surname> <given-names>Zhi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Ke</surname> <given-names>Caihuan</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1524258/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>You</surname> <given-names>Weiwei</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/539972/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Wang</surname> <given-names>Ying</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/452253/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Automation, Xiamen University</institution>, <addr-line>Xiamen</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>State Key Laboratory of Marine Environmental Science, Xiamen University</institution>, <addr-line>Xiamen</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Fujian Key Laboratory of Genetics and Breeding of Marine Organisms</institution>, <addr-line>Xiamen</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>College of Ocean and Earth Sciences, Xiamen University</institution>, <addr-line>Xiamen</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>Xiamen Key Laboratory of Big Data Intelligent Analysis and Decision</institution>, <addr-line>Xiamen</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Yinghui Dong, Zhejiang Wanli University, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Xiaoting Huang, Ocean University of China, China; Zhiyi Bai, Shanghai Ocean University, China</p></fn>
<corresp id="c001">&#x002A;Correspondence: Weiwei You, <email>wwyou@xmu.edu.cn</email></corresp>
<corresp id="c002">Ying Wang, <email>wangying@xmu.edu.cn</email></corresp>
<fn fn-type="other" id="fn002"><p><sup>&#x2020;</sup>These authors share first authorship</p></fn>
<fn fn-type="other" id="fn004"><p>This article was submitted to Marine Fisheries, Aquaculture and Living Resources, a section of the journal Frontiers in Marine Science</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>02</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>9</volume>
<elocation-id>841561</elocation-id>
<history>
<date date-type="received">
<day>22</day>
<month>12</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>17</day>
<month>01</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2022 Deng, Peng, Ma, Ke, You and Wang.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Deng, Peng, Ma, Ke, You and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Aquaculture is a rapidly growing industry that brings huge economic benefits. Genome-wide association study (GWAS) is critical for aquaculture species&#x2019; productivity, sustainability, and product quality. The current integrated GWAS pipeline either includes only specific limited steps or requires a complex prerequisite environment and configurations. In this study, we developed AquaGWAS, a highly user-friendly graphical user interface (GUI) GWAS pipeline, by integrating four well-known GWAS models. AquaGWAS is a complete GWAS pipeline from preprocessing, multiple choice of GWAS models, postprocessing to visualizations. AquaGWAS offers GUI easy running on Linux and automatically generates running command lines for high-performance computing (HPC) or non-GUI servers. AquaGWAS is free from installation, configurations, and complicated augment inputs. It offers whole packages of required reference files for 27 common aquatic species. Furthermore, aiming at the issue that the availability of genomic reference sequences limits single-nucleotide polymorphism (SNP) detection, we attempted to detect SNPs in Pacific abalone using classical alignment-based reference-required strategy and <italic>k</italic>-mer-based reference-free strategy combined with downstream AquaGWAS. On 222 resequencing data of Pacific abalone, two strategies detected 221,061 and 230,213 variants, respectively, with 180,161 common variants. The two strategies emphasized different variant situations: capturing variants missed by incomplete or inaccurate reference genomic sequence (<italic>k</italic>-mer-based) and capturing the indel variants having the baseline of genomic sequence (alignment-based). Combining the two strategies offers a complementary framework to obtain the accurate and complete GWAS analysis for non-model organism species. AquaGWAS is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/Ying-Lab/AquaGWAS">https://github.com/Ying-Lab/AquaGWAS</ext-link>.</p>
</abstract>
<kwd-group>
<kwd>GWAS</kwd>
<kwd>aquatic animal</kwd>
<kwd>integrated pipeline</kwd>
<kwd>abalone</kwd>
<kwd><italic>k</italic>-mer-based GWAS</kwd>
</kwd-group>
<contract-num rid="cn001">62173282</contract-num>
<contract-num rid="cn001">U1605213</contract-num>
<contract-sponsor id="cn001">National Natural Science Foundation of China <named-content content-type="fundref-id">10.13039/501100001809</named-content></contract-sponsor>
<contract-sponsor id="cn002">National Key Research and Development Program of China <named-content content-type="fundref-id">10.13039/501100012166</named-content></contract-sponsor>
<counts>
<fig-count count="4"/>
<table-count count="0"/>
<equation-count count="0"/>
<ref-count count="19"/>
<page-count count="7"/>
<word-count count="4107"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>Introduction</title>
<p>Genome-wide association study (GWAS) effectively identifies genotypic variants associated with particular phenotypic traits. Lots of related tools have been developed to improve the efficiency of GWAS analysis. However, many tools were designed using a specific model or for a certain processing step, which led to fussy file format conversions between various interfaces and frequent switches among different running environments (C/C++, Python, R, etc.). In contrast, the integrated pipeline, such as iPat (<xref ref-type="bibr" rid="B1">Chen and Zhang, 2018</xref>), HAPPI GWAS (<xref ref-type="bibr" rid="B13">Slaten et al., 2020</xref>), either prerequire a JAVA environment or includes only a single GWAS model, and the complicated augments and options in command lines limit the feasibility for non-computational researchers.</p>
<p>Aquaculture is a rapidly growing industry that brings huge economic benefits (<xref ref-type="bibr" rid="B3">FAO, 2017</xref>). GWAS, which is used for exploring molecular markers and mechanisms related to traits, is critical to improving aquaculture species&#x2019; productivity, sustainability, and product quality. Meanwhile, species-specific single-nucleotide polymorphism (SNP) arrays are still not widely used for aquaculture species. In most cases, species were sequenced and aligned to reference genomic sequence and then genotyped using a <italic>vcf</italic> format file (<xref ref-type="bibr" rid="B4">Jiang et al., 2019</xref>; <xref ref-type="bibr" rid="B16">Wu et al., 2019</xref>). Three obstructions limit the feasibility of the GWAS study: (1) Most marine biologists have a limited background in modeling and bioinformatics. At the same time, it is hard for them to switch among different running environments, convert fussy file formats, and handle command lines with complex augments and different models. (2) Due to the diversity of aquaculture species, there is no common data format and there are a few available public databases. In that way, annotation and reference information requires extra manual operations. (3) A large amount of non-model aquaculture animals, novel species, or species are hard to detect SNPs because of lack of reference genomes or incomplete genome. Therefore, it is necessary to develop environment-friendly software to resolve these problems.</p>
<p>This study developed AquaGWAS, a highly integrated GWAS analysis pipeline with a full reference package, including 27 aquatic species. AquaGWAS offers an easy-to-use graphic user interface (GUI) for stand-alone Linux server running mode and an automatic command-generator on Windows for HPC job-submission running mode. Free from complex installation and configurations, AquaGWAS is compatible with various file formats and highly friendly to users with a non-computational background.</p>
<p>Furthermore, to implement GWAS analysis for aquaculture species without a reference genome or only with an incomplete genome, we attempted to detect SNPs for Pacific abalone <italic>Haliotis discus hannai</italic> using classical alignment-based reference-required strategy and <italic>k</italic>-mer-based reference-free strategy. The two strategies were implemented using AquaGWAS combined with upstream alignment-based [Bowtie2 (<xref ref-type="bibr" rid="B6">Langmead and Salzberg, 2012</xref>) SAMtools and BCFtools (<xref ref-type="bibr" rid="B2">Danecek et al., 2021</xref>)] and <italic>k</italic>-mer-based [DISCOSNP (<xref ref-type="bibr" rid="B14">Uricaru et al., 2015</xref>)] pipelines. Based on 222 genomic resequencing data of Pacific abalones, the two strategies detected 221,061 and 230,213 variants, respectively, with 180,161 common variants. The detected variants belonged to 2,048 and 2,068 genes, respectively, with 2,046 common gene resolutions. The two strategies were highly consistent in both gene and nucleotide resolutions. With MWR <inline-formula><mml:math id="INEQ1"><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mpadded width="+5pt"><mml:mi>Foot</mml:mi></mml:mpadded><mml:mo>&#x2062;</mml:mo><mml:mpadded width="+5pt"><mml:mi>muscle</mml:mi></mml:mpadded><mml:mo>&#x2062;</mml:mo><mml:mtext>weight</mml:mtext></mml:mrow><mml:mrow><mml:mpadded width="+5pt"><mml:mtext>Wet</mml:mtext></mml:mpadded><mml:mo>&#x2062;</mml:mo><mml:mtext>weight</mml:mtext></mml:mrow></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> and SLW <inline-formula><mml:math id="INEQ2"><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mpadded width="+5pt"><mml:mi>Shell</mml:mi></mml:mpadded><mml:mo>&#x2062;</mml:mo><mml:mi>length</mml:mi></mml:mrow><mml:mrow><mml:mpadded width="+5pt"><mml:mi>Shell</mml:mi></mml:mpadded><mml:mo>&#x2062;</mml:mo><mml:mi>width</mml:mi></mml:mrow></mml:mfrac><mml:mo rspace="7.5pt">)</mml:mo></mml:mrow></mml:math></inline-formula>as traits, the two strategies captured four and five common-associated SNPs using AquaGWAS. The two strategies have clear advantages and limitations due to their basic principle and have explicit complementation during the GWAS analysis.</p>
</sec>
<sec id="S2" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="S2.SS1">
<title>Data Acquisition of the Pacific Abalone</title>
<p>The Pacific abalone used in this study contained a total of 222 individuals randomly selected from 10 families (<xref ref-type="bibr" rid="B9">Peng et al., 2021</xref>). Two growth-related traits (the ratio of foot muscle weight divided by wet weight, MWR, and the ratio of shell length divided by shell width, SLW) were measured and calculated after 2.5 years of culture. The genotypic data were generated using whole-genome sequencing (WGS) technology. Sequence reads for each sample were quality controlled using an in-house Perl script with a total of 2,408.162 Gb clean data remaining. The average sequencing depth for each sample was 7.75&#x00D7;.</p>
</sec>
<sec id="S2.SS2">
<title>Framework of AquaGWAS</title>
<p>Currently, AquaGWAS integrates widely used GWAS tools within a GUI on Linux to perform a complete GWAS analysis pipeline for aquatic animals using various SNP files, <italic>vcf</italic>, <italic>ped/map</italic>, <italic>tped/tfam</italic>, and <italic>bed/bim/fam</italic> as inputs. AquaGWAS contains five functional modules: (1) GWAS; (2) linkage disequilibrium (LD) decay analysis; (3) visualization with principal component analysis (PCA), Manhattan plot, and quantile-quantile (Q-Q) plot; (4) gene structural annotation; and (5) gene functional annotation. After integrating the reference sequences, structural annotation, and function annotation information from 27 aquatic species, AquaGWAS can implement the complete GWAS analysis for these aquatic species without extra reference databases.</p>
<p>The framework of AquaGWAS is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>. The input to AquaGWAS was phenotype data and SNP information data. The SNP information data, generally containing <italic>genotype</italic> and <italic>map</italic>, were compatible with the following commonly used file formats: (1) <italic>vcf</italic>, (2) <italic>ped/map</italic>, (3) <italic>tped/tfam</italic>, and (4) <italic>bed/bim/fam</italic>. Quality control (QC) was performed for SNP input files. The optional files included covariate and kinship files, which would be the additional information for GWAS analysis models and postprocessing for gene structural or functional annotation, respectively. The output included <italic>p</italic>-values of SNPs, Manhattan plot, Q-Q plot, PCA plot, LD decay plot, and the following structural annotations and functional annotations.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>The framework of AquaGWAS. AquaGWAS is compatible with several SNP information file formats, <italic>vcf</italic>, <italic>ped</italic>/<italic>map</italic>, <italic>tped</italic>/<italic>tfam</italic>, and <italic>bed</italic>/<italic>bim</italic>/<italic>fam</italic>. Several widely used GWAS tools, preprocessing and postprocessing, visualizations, and structural and functional annotations are integrated seamlessly, and output textual and visual analyzing results.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-09-841561-g001.tif"/>
</fig>
</sec>
<sec id="S2.SS3">
<title>Implementation of AquaGWAS</title>
<p>The GWAS of AquaGWAS offered a linear model and a regression model from PLINK (<xref ref-type="bibr" rid="B11">Purcell et al., 2007</xref>), a linear mixed model (LMM, also called MLM) from GEMMA (<xref ref-type="bibr" rid="B19">Zhou and Stephens, 2012</xref>), and an efficient mixed-model association (EMMA) from EMMAX (<xref ref-type="bibr" rid="B5">Kang et al., 2010</xref>). Meanwhile, QC was achieved through PLINK. The PopLDdecay (<xref ref-type="bibr" rid="B18">Zhang et al., 2018</xref>) and GCTA (<xref ref-type="bibr" rid="B17">Yang et al., 2011</xref>) were used for LD decay analysis and PCA analysis of AquaGWAS, respectively. Regarding the structural annotation, it was achieved using Annovar (<xref ref-type="bibr" rid="B15">Wang et al., 2010</xref>). In addition, the gffread and gtfToGenePred (<xref ref-type="bibr" rid="B10">Pertea and Pertea, 2020</xref>) helped to automatically generate the input file of Annovar. Then, some custom functions were used for functional annotation to extract the response gene annotation information directly from the gene functional annotation database file by gene ID.</p>
<p>The tools integrated by AquaGWAS, such as PLINK and GEMMA, include multiple small functions. Additionally, some of them might not be required for GWAS analysis. In AquaGWAS, we only integrated the required function packages and offered the corresponding parameter options in GUI. The fixed parameters are predetermined in the code. For more flexible parameters and options, we welcome more voices, and we will always update and improve AquaGWAS according to the user&#x2019;s feedback.</p>
</sec>
<sec id="S2.SS4">
<title>Alignment-Based Reference-Required and <italic>k</italic>-mer-Based Reference-Free Genome-Wide Association Study</title>
<p>Different from the plant and livestock, many aquatic animals do not have genomic reference sequences or only have an incomplete genomic sequence, which limits the accurate and complete GWAS analysis for these species. Therefore, using Pacific abalone as an example, we implemented classical alignment-based reference-required GWAS (called alignment-based) and <italic>k</italic>-mer-based reference-free GWAS (called <italic>k</italic>-mer-based) and combined AquaGWAS with two different upstream strategies, as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. The alignment-based strategy refers to the classical GWAS processing pipeline. It aligns the genomic sequencing reads to reference genomic sequences using sequence alignment tools, such as Bowtie (<xref ref-type="bibr" rid="B7">Langmead et al., 2009</xref>) or BWA (<xref ref-type="bibr" rid="B8">Li and Durbin, 2009</xref>). The alignment results were then sorted and filtered using SAMtools and BCFtools to call the candidate genetic variants. The <italic>k</italic>-mer-based strategy was implemented using DISCOSNP, which calls the candidate variants free from the genomic reference sequence. DISCOSNP detects SNPs with a <italic>k</italic>-mer comparison based on a <italic>de Bruijn</italic> graph. It is a directed graph that contains all the <italic>k</italic>-mers present in the read dataset as vertices, and all the possible (<italic>k</italic> &#x2212; 1) overlaps between <italic>k</italic>-mers as edges. Then, for each such couple of <italic>k</italic>-mers starting with the same <italic>k</italic> &#x2212; 1 length prefix, if both paths cannot be right extended with the same nucleotide, the bubble was discarded, and only isolated SNPs can be detected. Then, the detected <italic>k</italic>-mers with SNPs were assembled into contigs, and the whole procedure was free from the genomic reference sequence. The detail of DISCOSNP can be found in the study by <xref ref-type="bibr" rid="B14">Uricaru et al. (2015)</xref>. The variants calling files from the above two strategies were input into AquaGWAS for GWAS, with MWR and SLW as traits.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>The two different upstream strategies, namely, alignment-based reference-required and <italic>k</italic>-mer-based reference-free, combined AquaGWAS for genome-wide association study (GWAS). The alignment-based strategy aligns the genomic sequencing reads to reference genomic sequences and then calls the candidate genetic variants. Additionally, the <italic>k</italic>-mer-based strategy calls the candidate variants free from reference genomic sequence with a <italic>k</italic>-mer comparison based on a de <italic>Bruijn</italic> graph. The variants calling files from the above two strategies are input into AquaGWAS for GWAS.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-09-841561-g002.tif"/>
</fig>
</sec>
</sec>
<sec id="S3" sec-type="results">
<title>Results</title>
<sec id="S3.SS1">
<title>Functions of AquaGWAS</title>
<p>There are several functions for AquaGWAS. First, AquaGWAS offers the thresholds to exclude individuals with too much missing genotype data, SNPs with too small minor allele frequency (MAF), and SNPs with too large missing genotype rate. Users are free to select none/some/all these parameters, and the parameters (window size, step length, and <italic>r</italic><sup>2</sup> threshold) for data filtering using linkage disequilibrium are also allowed to be selected and set.</p>
<p>Second, PLINK, GEMMA, and EMMAX are integrated by AquaGWAS for GWAS. Users can input single phenotype or multiple phenotype files for association analysis, and the kinship and covariates can be used as optional inputs to control false positive analysis results. The analysis results are visualized using the Manhattan plot and Q-Q plot. Users can also get the results of population genetic diversity of samples and linkage disequilibrium decay analysis, which are subsequently implemented using GCTA and PopLDdecay, and then visualized using PCA and LD plots.</p>
<p>Third, the process of annotation was divided into two steps, structure annotation and function annotation. Among these, the first step abstracted information of SNPs above the threshold [&#x2013;log (10&#x2013;5) by default] from the <italic>vcf</italic> file and then converted it into the file needed for gene structural and functional annotation. Structural annotations offer exon regions such as variant functions, types, amino acid changes, and all the genes and positions of the mutations. Additionally, the functional annotation is realized on the gene functional annotation database of corresponding species.</p>
</sec>
<sec id="S3.SS2">
<title>Features of AquaGWAS</title>
<p>AquaGWAS is an easy and user-friendly software for analyzing aquatic animals with non-commonly used reference genome databases and rare information in public databases, which makes it difficult for analyses by regular software and pipeline. Several features of the AquaGWAS are as follows.</p>
<p>First, AquaGWAS is free from any installation and environmental configuration, which offers the executable file to be run directly.</p>
<p>Second, AquaGWAS offers a user-friendly GUI. To begin with, encapsulating the entire tedious process, AquaGWAS offers an intuitive parameter setting interface. Furthermore, in AquaGWAS, users can switch among various tools smoothly and seamlessly during the analysis process, free from complicated file format conversions between interfaces, complicated command lines with tedious running arguments, and file folder switching.</p>
<p>Third, AquaGWAS automatically generate a command line in HPC. For HPC users, jobs are required to submit with command lines, where the direct running on the graphic interface is not available. Therefore, AquaGWAS also offers GUI automatic command-line generator on Windows so that users can submit the job with the generated command line on Windows conveniently.</p>
<p>Finally, AquaGWAS is compatible with various file formats. Because different GWAS tools support various file formats, AquaGWAS automatically converts the input file according to the selected tool. For example, although EMMAX only supports <italic>tped/tfam</italic> files, users can choose <italic>vcf</italic>, <italic>ped/map</italic>, or <italic>bed/bim/fam</italic> in AquaGWAS for analysis using EMMAX through automatic file format conversion. AquaGWAS also supports the automatic loading of multiple files with the same prefix filename under the same folder. For example, if the input file is of <italic>ped/map</italic> or <italic>tped/tfam</italic> format, users can choose to input only <italic>ped</italic> or <italic>tped</italic> file. AquaGWAS will automatically load the associated <italic>map</italic> or <italic>tfam</italic> file under the same path with the same prefix of the filename.</p>
<p>Meanwhile, because the AquaGWAS is a pipeline that integrates the current widely used GWAS analyzing tools, the upper limit of the amount of running data is determined by the selected tool during the AquaGWAS running and the hardware of the running server.</p>
</sec>
<sec id="S3.SS3">
<title>Genome-Wide Association Study for Pacific Abalone</title>
<p>The alignment-based and <italic>k</italic>-mer-based strategies were applied for GWAS analysis of Pacific abalone using 222 genomic resequencing data. Using Chromosome 1 as an example, the alignment-based and <italic>k</italic>-mer-based methods detected 221,061 and 230,213 variants, respectively. To compare the results from the two strategies, the contigs from the <italic>k</italic>-mer-based strategy were aligned to the reference genome of the Pacific abalone, and the corresponding position in Chromosome 1 was obtained. Compared with the genetic variants from the alignment-based method, 180,161 common variants were detected by the two strategies, as shown in the Venn diagram in <xref ref-type="fig" rid="F3">Figure 3A</xref>. The 50,052 variants detected by <italic>k</italic>-mer-based but missed from alignment-based were partly caused by incomplete or inaccurate genomic reference sequences. In contrast, the <italic>k</italic>-mer-based method also missed 40,900 variants found by the alignment-based method, which is probably caused by the structural insert and delete due to the <italic>de Bruijn</italic> assembly errors caused by low reads coverage, repeat sequences, and short overlaps. Considering the genes corresponding to SNP, the intersection number of genes detected by kmer-based (2048) and alignment-based (2068) is 2046, as shown in the Venn diagram in <xref ref-type="fig" rid="F3">Figure 3B</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>The comparison of the results from the alignment-based and the <italic>k</italic>-mer-based strategies. <bold>(A)</bold> The Venn diagram of candidate variants from <italic>k</italic>-mer-based and alignment-based strategies; <bold>(B)</bold> the Venn diagram of genes including variants from <italic>k</italic>-mer-based and alignment-based strategies; <bold>(C)</bold> the Manhattan plots for MWR traits from common, alignment-based, and <italic>k</italic>-mer-based SNPs; <bold>(D)</bold> the Manhattan plots for SLW traits from common, alignment-based, and <italic>k</italic>-mer-based SNPs.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-09-841561-g003.tif"/>
</fig>
<p>The variant calling files from the two strategies were input into AquaGWAS. The <italic>k</italic>-mer-based method used <italic>p</italic> &#x2212; value &#x2264; 10<sup>&#x2212;4</sup> and <italic>p</italic> &#x2212; value &#x2264; 10<sup>&#x2212;9</sup> to screen the significant SNPs of MWR and SLW and obtained 10 and 6 significant SNPs, respectively. For the alignment-based method, 11 and 8 significant SNPs were obtained by using a <italic>p</italic> &#x2212; value &#x2264; 10<sup>&#x2212;4</sup> and <italic>p</italic> &#x2212; value &#x2264; 10<sup>&#x2212;5</sup> of MWR and SLW. Using <italic>p</italic> &#x2212; value &#x2264; 10<sup>&#x2212;5</sup>, there are 4 and 5 common SNPs associated with MWR and SLW traits detected by both alignment-based and <italic>k</italic>-mer-based methods, which are shown in <xref ref-type="fig" rid="F3">Figures 3C,D</xref>. The detailed analysis results are presented in <xref ref-type="supplementary-material" rid="TS1">Supplementary Table 1</xref>.</p>
<p>To compare the variants detected by two strategies elaborately, the contigs, including variants detected by alignment-based strategy and the <italic>k</italic>-mers detected by variants from the <italic>k</italic>-mer-based strategy, are aligned to the genomic sequences of Pacific abalone. Because a reference sequence is required, we only gave the situations that both the <italic>k</italic>-mers and reads can be aligned to the reference genomic sequences shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. The genomic region from <italic>chromosome 1:21634620-21634760</italic> is selected and viewed on Integrative Genomics Viewer (IGV) (<xref ref-type="bibr" rid="B12">Robinson et al., 2011</xref>). The variant in the red border is detected by both strategies. Additionally, the variants in yellow borders are detected only by <italic>k</italic>-mer-based strategy. The variant and insertion in the green border are detected only by alignment-based strategy. The two strategies emphasized different variant situations. After tracing back to the original variant detection process, we found <italic>k</italic>-mer-based method can capture variants missed by incomplete or inaccurate reference genomic sequences. In contrast, an alignment-based strategy can capture the insertion or deletion having the baseline of genomic sequence. Therefore, the two strategies have explicit complementation for GWAS analysis.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>The detailed comparison of genetic variants detected by the alignment-based and <italic>k</italic>-mer-based strategies. Both strategies detect the variant in the red border. Furthermore, the variants in yellow borders are detected only by <italic>k</italic>-mer-based strategy. The variant and insertion in the green border are detected only by alignment-based strategy.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-09-841561-g004.tif"/>
</fig>
</sec>
</sec>
<sec id="S4" sec-type="conclusion|discussion">
<title>Conclusion and Discussion</title>
<p>The current integrated GWAS pipeline either includes only specific limited steps or requires a complex prerequisite environment and configurations. AquaGWAS offers GUI easy running for stand-alone Linux servers and an automatic command line generator for HPC or non-GUI servers. AquaGWAS integrated complete GWAS pipeline from preprocessing, multiple choice of GWAS models, postprocessing to visualizations. Furthermore, AquaGWAS is highly friendly to users of non-computational background, which is free from installation, configurations, and complicated augments input. AquaGWAS is compatible with various frequently used SNP file formats and can be applied to several species, especially aquatic animals. AquaGWAS offers whole packages of required reference files for 27 common aquatic species to implement the express aquaculture GWAS analysis without any extra reference inputs.</p>
<p>The experiment shows that the <italic>k</italic>-mer-based strategy can capture the variants, which are missed because of the incomplete or inaccurate reference genomic sequences. However, the <italic>k</italic>-mer-based reference-free strategy also missed SNPs, which are found by the classical reference-alignment strategy. The two strategies have clear advantages and limitations due to their basic principle and have explicit complementation during the GWAS analysis.</p>
<p>Our exploration not only offers an option to study the genetic variants for non-model organism species, novel species, or species without genomic reference sequence but also provides a complementary framework for accurate and complete GWAS analysis.</p>
</sec>
<sec id="S5" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>The data analyzed in this study is subject to the following licenses/restrictions: The dataset is on another ongoing manuscript, which will be public available after that manuscript is published. Requests to access these datasets should be directed to WY, <email>109852383@qq.com</email>.</p>
</sec>
<sec id="S6">
<title>Author Contributions</title>
<p>YW, WY, and CK planned the project. CD and ZM finished the programming. WP designed the processing steps of GWAS. WP, CD, and ZM tested the software. YW and CD designed the experiments. YW, CD, and WP wrote the main manuscript. All the authors read and approved the final manuscript.</p>
</sec>
<sec id="conf1" sec-type="COI-statement">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="pudiscl1" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<sec id="S7" sec-type="funding-information">
<title>Funding</title>
<p>This study was supported by the National Natural Science Foundation of China (62173282, U1605213, and 31872564), the National Key Research and Development Program of China (2018YFD0901401), Fujian Provincial S&#x0026;T Project (2019N0001 and 2017FJSCZY02), Open Fund of Engineering Research Center for Medical Data Mining and Application of Fujian Province (MDM2018002), and Natural Science Foundation of Fujian (2018J01097).</p>
</sec>
<ack>
<p>We thank W. He, S. Fang, and T. Wu from the Information and Network Center of Xiamen University for their help with GPU computing.</p>
</ack>
<sec id="S9" sec-type="supplementary-material">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmars.2022.841561/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmars.2022.841561/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.XLSX" id="TS1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>C. J.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name></person-group> (<year>2018</year>). <article-title>iPat: intelligent prediction and association tool for genomic research.</article-title> <source><italic>Bioinformatics</italic></source> <volume>34</volume> <fpage>1925</fpage>&#x2013;<lpage>1927</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty015</pub-id> <pub-id pub-id-type="pmid">29342241</pub-id></citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Danecek</surname> <given-names>P.</given-names></name> <name><surname>Bonfield</surname> <given-names>J. K.</given-names></name> <name><surname>Liddle</surname> <given-names>J.</given-names></name> <name><surname>Marshall</surname> <given-names>J.</given-names></name> <name><surname>Ohan</surname> <given-names>V.</given-names></name> <name><surname>Pollard</surname> <given-names>M. O.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Twelve years of SAMtools and BCFtools.</article-title> <source><italic>GigaScience</italic></source> <volume>10</volume>:<issue>8</issue>. <pub-id pub-id-type="doi">10.1093/gigascience/giab008</pub-id> <pub-id pub-id-type="pmid">33590861</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><collab>FAO</collab> (<year>2017</year>). <source><italic>Genome-Based Biotechnologies in Aquaculture.</italic></source> <publisher-loc>Rome</publisher-loc>: <publisher-name>Food and Agriculture Organization of the United Nations</publisher-name>, <fpage>2</fpage>&#x2013;<lpage>3</lpage>.</citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>D. L.</given-names></name> <name><surname>Gu</surname> <given-names>X. H.</given-names></name> <name><surname>Li</surname> <given-names>B. J.</given-names></name> <name><surname>Zhu</surname> <given-names>Z. X.</given-names></name> <name><surname>Qin</surname> <given-names>H.</given-names></name> <name><surname>Meng</surname> <given-names>Z. N.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Identifying a long QTL cluster across chrLG18 associated with salt tolerance in tilapia using GWAS and QTL-seq</article-title>. <source><italic>Mar. Biotechnol</italic></source>. <volume>21</volume>, <fpage>250</fpage>&#x2013;<lpage>261</lpage>. <pub-id pub-id-type="doi">10.1007/s10126-019-09877-y</pub-id> <pub-id pub-id-type="pmid">30737627</pub-id></citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kang</surname> <given-names>H. M.</given-names></name> <name><surname>Sul</surname> <given-names>J. H.</given-names></name> <name><surname>Service</surname> <given-names>S. K.</given-names></name> <name><surname>Zaitlen</surname> <given-names>N. A.</given-names></name> <name><surname>Kong</surname> <given-names>S.-Y.</given-names></name> <name><surname>Freimer</surname> <given-names>N. B.</given-names></name><etal/></person-group> (<year>2010</year>). <article-title>Variance component model to account for sample structure in genome-wide association studies</article-title>. <source><italic>Nat. Genet</italic></source>. <volume>42</volume>, <fpage>348</fpage>&#x2013;<lpage>354</lpage>. <pub-id pub-id-type="doi">10.1038/ng.548</pub-id> <pub-id pub-id-type="pmid">20208533</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Langmead</surname> <given-names>B.</given-names></name> <name><surname>Salzberg</surname> <given-names>S. L.</given-names></name></person-group> (<year>2012</year>). <article-title>Fast gapped-read alignment with Bowtie 2.</article-title> <source><italic>Nat. Methods</italic></source> <volume>9</volume> <fpage>357</fpage>&#x2013;<lpage>359</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.1923</pub-id> <pub-id pub-id-type="pmid">22388286</pub-id></citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Langmead</surname> <given-names>B.</given-names></name> <name><surname>Trapnell</surname> <given-names>C.</given-names></name> <name><surname>Pop</surname> <given-names>M.</given-names></name> <name><surname>Salzberg</surname> <given-names>S. L.</given-names></name></person-group> (<year>2009</year>). <article-title>Ultrafast and memory-efficient alignment of short DNA sequences to the human genome.</article-title> <source><italic>Genome Biol.</italic></source> <volume>10</volume>:<issue>R25</issue>. <pub-id pub-id-type="doi">10.1186/gb-2009-10-3-r25</pub-id> <pub-id pub-id-type="pmid">19261174</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Durbin</surname> <given-names>R.</given-names></name></person-group> (<year>2009</year>). <article-title>Fast and accurate short read alignment with Burrows&#x2013;Wheeler transform.</article-title> <source><italic>Bioinformatics</italic></source> <volume>25</volume> <fpage>1754</fpage>&#x2013;<lpage>1760</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp324</pub-id> <pub-id pub-id-type="pmid">19451168</pub-id></citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peng</surname> <given-names>W.</given-names></name> <name><surname>Yu</surname> <given-names>F.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Lu</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Identification of growth-related SNPs and genes in the genome of the Pacific abalone (<italic>Haliotis discus hannai</italic>) using GWAS.</article-title> <source><italic>Aquaculture</italic></source> <volume>541</volume>:<issue>736820</issue>. <pub-id pub-id-type="doi">10.1016/j.aquaculture.2021.736820</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pertea</surname> <given-names>G.</given-names></name> <name><surname>Pertea</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>GFF utilities: GffRead and GffCompare.</article-title> <source><italic>F1000Research</italic></source> <volume>9</volume>:<issue>23297</issue>. <pub-id pub-id-type="doi">10.12688/f1000research.23297.1</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Purcell</surname> <given-names>S.</given-names></name> <name><surname>Neale</surname> <given-names>B.</given-names></name> <name><surname>Todd-Brown</surname> <given-names>K.</given-names></name> <name><surname>Thomas</surname> <given-names>L.</given-names></name> <name><surname>Ferreira</surname> <given-names>M. A.</given-names></name> <name><surname>Bender</surname> <given-names>D.</given-names></name><etal/></person-group> (<year>2007</year>). <article-title>PLINK: a tool set for whole-genome association and population-based linkage analyses</article-title>. <source><italic>Am. J. Hum. Genet</italic></source>. <volume>81</volume>, <fpage>559</fpage>&#x2013;<lpage>575</lpage>. <pub-id pub-id-type="doi">10.1086/519795</pub-id> <pub-id pub-id-type="pmid">17701901</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Robinson</surname> <given-names>J. T.</given-names></name> <name><surname>Thorvaldsd&#x00F3;ttir</surname> <given-names>H.</given-names></name> <name><surname>Winckler</surname> <given-names>W.</given-names></name> <name><surname>Guttman</surname> <given-names>M.</given-names></name> <name><surname>Lander</surname> <given-names>E. S.</given-names></name> <name><surname>Getz</surname> <given-names>G.</given-names></name><etal/></person-group> (<year>2011</year>). <article-title>Integrative genomics viewer.</article-title> <source><italic>Nat. Biotechnol.</italic></source> <volume>29</volume> <fpage>24</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.1754</pub-id> <pub-id pub-id-type="pmid">21221095</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Slaten</surname> <given-names>M. L.</given-names></name> <name><surname>Chan</surname> <given-names>Y. O.</given-names></name> <name><surname>Shrestha</surname> <given-names>V.</given-names></name> <name><surname>Lipka</surname> <given-names>A. E.</given-names></name> <name><surname>Angelovici</surname> <given-names>R.</given-names></name></person-group> (<year>2020</year>). <article-title>HAPPI GWAS: holistic analysis with Pre- and Post-integration GWAS.</article-title> <source><italic>Bioinformatics</italic></source> <volume>36</volume> <fpage>4655</fpage>&#x2013;<lpage>4657</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa589</pub-id> <pub-id pub-id-type="pmid">32579187</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Uricaru</surname> <given-names>R.</given-names></name> <name><surname>Rizk</surname> <given-names>G.</given-names></name> <name><surname>Lacroix</surname> <given-names>V.</given-names></name> <name><surname>Quillery</surname> <given-names>E.</given-names></name> <name><surname>Plantard</surname> <given-names>O.</given-names></name> <name><surname>Chikhi</surname> <given-names>R.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>Reference-free detection of isolated SNPs.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>43</volume>:<issue>e11</issue>. <pub-id pub-id-type="doi">10.1093/nar/gku1187</pub-id> <pub-id pub-id-type="pmid">25404127</pub-id></citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>K.</given-names></name> <name><surname>Li</surname> <given-names>M.</given-names></name> <name><surname>Hakonarson</surname> <given-names>H.</given-names></name></person-group> (<year>2010</year>). <article-title>ANNOVAR: functional annotation of genetic variants from high-throughput sequencing data.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>38</volume>:<issue>e164</issue>. <pub-id pub-id-type="doi">10.1093/nar/gkq603</pub-id> <pub-id pub-id-type="pmid">20601685</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>L.</given-names></name> <name><surname>Yang</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>B.</given-names></name> <name><surname>Huang</surname> <given-names>W.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>First genome-wide association analysis for growth traits in the largest coral reef-dwelling bony fishes, the giant grouper (<italic>Epinephelus lanceolatus</italic>)</article-title>. <source><italic>Mar. Biotechnol</italic></source>. <volume>21</volume>, <fpage>707</fpage>&#x2013;<lpage>717</lpage>. <pub-id pub-id-type="doi">10.1007/s10126-019-09916-8</pub-id> <pub-id pub-id-type="pmid">31392592</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>J.</given-names></name> <name><surname>Lee</surname> <given-names>S. H.</given-names></name> <name><surname>Goddard</surname> <given-names>M. E.</given-names></name> <name><surname>Visscher</surname> <given-names>P. M.</given-names></name></person-group> (<year>2011</year>). <article-title>GCTA: a tool for genome-wide complex trait analysis.</article-title> <source><italic>Am. J. Hum. Genet.</italic></source> <volume>88</volume> <fpage>76</fpage>&#x2013;<lpage>82</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2010.11.011</pub-id> <pub-id pub-id-type="pmid">21167468</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>C.</given-names></name> <name><surname>Dong</surname> <given-names>S.</given-names></name> <name><surname>Xu</surname> <given-names>J.</given-names></name> <name><surname>He</surname> <given-names>W.</given-names></name> <name><surname>Yang</surname> <given-names>T.</given-names></name></person-group> (<year>2018</year>). <article-title>PopLDdecay: a fast and effective tool for linkage disequilibrium decay analysis based on variant call format files.</article-title> <source><italic>Bioinformatics</italic></source> <volume>35</volume> <fpage>1786</fpage>&#x2013;<lpage>1788</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty875</pub-id> <pub-id pub-id-type="pmid">30321304</pub-id></citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>X.</given-names></name> <name><surname>Stephens</surname> <given-names>M.</given-names></name></person-group> (<year>2012</year>). <article-title>Genome-wide efficient mixed-model analysis for association studies</article-title>. <source><italic>Nat. Genet</italic></source>. <volume>44</volume>, <fpage>821</fpage>&#x2013;<lpage>824</lpage>. <pub-id pub-id-type="doi">10.1038/ng.2310</pub-id> <pub-id pub-id-type="pmid">22706312</pub-id></citation></ref>
</ref-list>
</back>
</article>