<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1251382</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2023.1251382</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Technology and Code</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Genomic Variations Explorer (GenVarX): a toolset for annotating promoter and CNV regions using genotypic and phenotypic differences</article-title>
<alt-title alt-title-type="left-running-head">Chan et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2023.1251382">10.3389/fgene.2023.1251382</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Chan</surname>
<given-names>Yen On</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2362547/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Biov&#x00E1;</surname>
<given-names>Jana</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1385855/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mahmood</surname>
<given-names>Anser</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2402416/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dietz</surname>
<given-names>Nicholas</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1774901/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bilyeu</surname>
<given-names>Kristin</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1702020/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>&#x160;krabi&#x161;ov&#xe1;</surname>
<given-names>M&#xe1;ria</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/306661/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Joshi</surname>
<given-names>Trupti</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/723974/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>MU Institute for Data Science and Informatics</institution>, <institution>University of Missouri-Columbia</institution>, <addr-line>Columbia</addr-line>, <addr-line>MO</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Biochemistry</institution>, <institution>Faculty of Science</institution>, <institution>Palacky University in Olomouc</institution>, <addr-line>Olomouc</addr-line>, <country>Czechia</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Division of Plant Science and Technology</institution>, <institution>University of Missouri-Columbia</institution>, <addr-line>Columbia</addr-line>, <addr-line>MO</addr-line>, <country>United States</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Plant Genetics Research Unit</institution>, <institution>United States Department of Agriculture-Agricultural Research Service</institution>, <addr-line>Columbia</addr-line>, <addr-line>MO</addr-line>, <country>United States</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Christopher S. Bond Life Sciences Center</institution>, <institution>University of Missouri-Columbia</institution>, <addr-line>Columbia</addr-line>, <addr-line>MO</addr-line>, <country>United States</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Department of Electrical Engineering and Computer Science</institution>, <institution>University of Missouri-Columbia</institution>, <addr-line>Columbia</addr-line>, <addr-line>MO</addr-line>, <country>United States</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>Department of Biomedical Informatics, Biostatistics and Medical Epidemiology</institution>, <institution>University of Missouri-Columbia</institution>, <addr-line>Columbia</addr-line>, <addr-line>MO</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1358628/overview">Rui Yin</ext-link>, University of Florida, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/857127/overview">Dailu Guan</ext-link>, University of California, Davis, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/141362/overview">Xuewen Wang</ext-link>, University of North Texas Health Science Center, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: M&#xe1;ria &#x160;krabi&#x161;ov&#xe1;, <email>maria.skrabisova@upol.cz</email>; Trupti Joshi, <email>joshitr@missouri.edu</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>09</day>
<month>10</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1251382</elocation-id>
<history>
<date date-type="received">
<day>01</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>27</day>
<month>09</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Chan, Biov&#x00E1;, Mahmood, Dietz, Bilyeu, &#x160;krabi&#x161;ov&#xe1; and Joshi.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Chan, Biov&#x00E1;, Mahmood, Dietz, Bilyeu, &#x160;krabi&#x161;ov&#xe1; and Joshi</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The rapid growth of sequencing technology and its increasing popularity in biology-related research over the years has made whole genome re-sequencing (WGRS) data become widely available. A large amount of WGRS data can unlock the knowledge gap between genomics and phenomics through gaining an understanding of the genomic variations that can lead to phenotype changes. These genomic variations are usually comprised of allele and structural changes in DNA, and these changes can affect the regulatory mechanisms causing changes in gene expression and altering the phenotypes of organisms. In this research work, we created the GenVarX toolset, that is backed by transcription factor binding sequence data in promoter regions, the copy number variations data, SNPs and Indels data, and phenotypes data which can potentially provide insights about phenotypic differences and solve compelling questions in plant research. Analytics-wise, we have developed strategies to better utilize the WGRS data and mine the data using efficient data processing scripts, libraries, tools, and frameworks to create the interactive and visualization-enhanced GenVarX toolset that encompasses both promoter regions and copy number variation analysis components. The main capabilities of the GenVarX toolset are to provide easy-to-use interfaces for users to perform queries, visualize data, and interact with the data. Based on different input windows on the user interface, users can provide inputs corresponding to each field and submit the information as a query. The data returned on the results page is usually displayed in a tabular fashion. In addition, interactive figures are also included in the toolset to facilitate the visualization of statistical results or tool outputs. Currently, the GenVarX toolset supports soybean, rice, and <italic>Arabidopsis</italic>. The researchers can access the soybean GenVarX toolset from SoyKB via <ext-link ext-link-type="uri" xlink:href="https://soykb.org/SoybeanGenVarX/">https://soykb.org/SoybeanGenVarX/</ext-link>, rice GenVarX toolset, and <italic>Arabidopsis</italic> GenVarX toolset from KBCommons web portal with links <ext-link ext-link-type="uri" xlink:href="https://kbcommons.org/system/tools/GenVarX/Osativa">https://kbcommons.org/system/tools/GenVarX/Osativa</ext-link> and <ext-link ext-link-type="uri" xlink:href="https://kbcommons.org/system/tools/GenVarX/Athaliana">https://kbcommons.org/system/tools/GenVarX/Athaliana</ext-link>, respectively.</p>
</abstract>
<kwd-group>
<kwd>transcription factor</kwd>
<kwd>promoter</kwd>
<kwd>copy number variation</kwd>
<kwd>whole genome re-sequencing data</kwd>
<kwd>genomic variations</kwd>
<kwd>SNPs</kwd>
<kwd>Indels</kwd>
<kwd>phenotypes</kwd>
</kwd-group>
<contract-sponsor id="cn001">United Soybean Board<named-content content-type="fundref-id">10.13039/100012009</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>As the use of sequencing technology becomes popular in both industrial and academic sectors, a large amount of whole genome re-sequencing (WGRS) data has become publicly available for users to utilize for research activities or commercial purposes. From the WGRS data, researchers can understand the genomic variations and the potential effects on phenotypes of organisms (<xref ref-type="bibr" rid="B4">Bolger et al., 2017</xref>). The differences in phenotypes compared among accessions are the reflections of genomic variations and structural variations (<xref ref-type="bibr" rid="B15">Li et al., 2022</xref>). The genomic variations include changes of alleles in gene regions, upstream promoter regions, and intergenic regions, while the structural changes comprise insertion, deletion, and duplication of DNA segments. These allele changes can occur in the transcription factor binding domain which leads to regulatory mechanism dysfunction for some target genes. Ultimately, the issues cause alterations in gene expression patterns and lead to phenotypic diversity in organisms. Similarly, the larger structural variations seen in the form of copy number variations (CNVs) can cause gains or losses in DNA segments. These modifications in DNA can alter the copies of expressed genes and ultimately results in changes in phenotypes in organisms (<xref ref-type="bibr" rid="B36">&#x17b;mie&#x144;ko et al., 2014</xref>). Thus, gaining an in-depth understanding of the transcription factor (TF) binding sites and the CNVs through performing analysis with extensive data and open-source tools and packages that are available online is critical in plant research and development.</p>
<p>Currently, there are more than 3000 soybean, 1000 <italic>Arabidopsis</italic>, and 3000 rice WGRS accession datasets along with phenotype datasets, annotation datasets, and transcription factors datasets scattered across different resources (<xref ref-type="bibr" rid="B25">The 3000 rice genomes project, 2014</xref>; <xref ref-type="bibr" rid="B1">Alonso-Blanco et al., 2016</xref>; <xref ref-type="bibr" rid="B17">Liu et al., 2020</xref>). The datasets are usually available for researchers to download in the form of static files and are not pre-integrated with other omics datasets. The publicly accessible web portals and platforms such as Plant Transcription Factor Database (PlantTFDB) (<xref ref-type="bibr" rid="B9">Jin et al., 2016</xref>), Plant Transcriptional Regulatory Map (PlantRegMap) (<xref ref-type="bibr" rid="B27">Tian et al., 2019</xref>), Gene Transcription Regulation Database (GTRD) (<xref ref-type="bibr" rid="B32">Yevshin et al., 2017</xref>), and JASPAR (<xref ref-type="bibr" rid="B5">Castro-Mondragon et al., 2021</xref>) are the main providers of transcription factors related datasets. Moreover, the National Center for Biotechnology Information (NCBI), European Nucleotide Archive (ENA), the Genome Sequence Archive (GSA) of the National Genomics Data Center (NGDC), and the CyVerse data store (<xref ref-type="bibr" rid="B7">Goff et al., 2011</xref>; <xref ref-type="bibr" rid="B20">Merchant et al., 2016</xref>) are the main resources for publicly available WGRS datasets. Likewise, there are also many open-source tools and packages available that can be applied to the WGRS data and used to perform CNV analysis like cn.MOPS (<xref ref-type="bibr" rid="B14">Klambauer et al., 2012</xref>), CNV-seq (<xref ref-type="bibr" rid="B31">Xie and Tammi, 2009</xref>), cnvScan (<xref ref-type="bibr" rid="B23">Samarakoon et al., 2016</xref>), and more. Nevertheless, there is a lack of interactive and visualizable web applications to integrate and query TF binding sites in promoter regions and CNVs data with the genomic variability observed from large-scale studies involving thousands of accessions to gain insight about phenotypes, perform validations, and eventually roll out new discoveries. To solve this problem, we have dedicated effort towards building an interactive and visualization-enhanced toolset for supporting this analysis.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and methods</title>
<p>In the materials and methods section, we provide details about the datasets collected and utilized for this research work including the tools used to obtain the datasets, process the data, and the respective required inputs and outputs for these tools. The processed datasets are uploaded and stored in the MySQL database integrated into both SoyKB (<xref ref-type="bibr" rid="B11">Joshi et al., 2012</xref>; <xref ref-type="bibr" rid="B10">Joshi et al., 2013</xref>; <xref ref-type="bibr" rid="B12">Joshi et al., 2017</xref>) and KBCommons (<xref ref-type="bibr" rid="B33">Zeng et al., 2018</xref>; <xref ref-type="bibr" rid="B34">Zeng et al., 2019</xref>) web portals. The package used for uploading the datasets, indexing methods for the tables in the database, and technology used in the web development for the GenVarX toolset are also described.</p>
<sec id="s2-1">
<title>Datasets</title>
<p>The GenVarX toolset currently supports the soybean, rice, and <italic>Arabidopsis</italic> organisms. For each organism, the toolset is divided into two components mainly for the promoter regions and the CNV analysis. In order to facilitate the functionalities of each component in each organism, many datasets are required to be collected and processed together to power the GenVarX toolset.</p>
<p>For the promoter regions component, the transcription factors (TFs) datasets and binding TF position probability matrices from the Plant Transcription Factor Database (PlantTFDB v5.0) (<xref ref-type="bibr" rid="B9">Jin et al., 2016</xref>) were acquired for each organism. In addition, the predicted transcription factor binding sites datasets and motif-gene regulation datasets from the Plant Transcriptional Regulatory Map (PlantRegMap) website (<xref ref-type="bibr" rid="B27">Tian et al., 2019</xref>) were required for each organism.</p>
<p>For the CNV analysis component, we acquired WGRS datasets for each organism (in different formats depending on their availability in public resources) and generated the CNV results. The WGRS files mainly include sequencing data in FASTQ format and the reads mapped to genome sequences in Binary Alignment Map (BAM) format. The FASTQ format datasets are further processed and converted into BAM files as the BAM files are the required input format for our selected tool for generating CNV results.</p>
<p>For the CNV analysis component in soybean, we have acquired WGRS datasets of 1066 distinct soybean accessions from publicly available datasets including Zhou302v2 (<xref ref-type="bibr" rid="B35">Zhou et al., 2015</xref>), Liu304 (<xref ref-type="bibr" rid="B17">Liu et al., 2020</xref>), USB-15x (<xref ref-type="bibr" rid="B28">Valliyodan et al., 2021</xref>), USB-40x (<xref ref-type="bibr" rid="B28">Valliyodan et al., 2021</xref>), Soja (<xref ref-type="bibr" rid="B13">Kim et al., 2010</xref>), and MSMC (<xref ref-type="bibr" rid="B29">Valliyodan and Nguyen, 2006</xref>). The PGen pipeline (<xref ref-type="bibr" rid="B18">Liu et al., 2016</xref>) was used for mapping these reads to Williams 82 version 2 (Wm82.a2.v1) reference genome and for SNP and Indel calling to generate both BAM and Variant Call Format (VCF) files. The mapped soybean sequencing reads in BAM format for all datasets were collectively stored on the Cyverse Data Store. The soybean Wm82.a2.v1 reference genome was downloaded from the Phytozome website (<xref ref-type="bibr" rid="B8">Goodstein et al., 2011</xref>). We have also collected soybean phenotype datasets from the Germplasm Resources Information Network (GRIN) database.</p>
<p>Similarly, we have acquired 3000 BAM files and VCF files of rice from the 3000 Rice Genome on Amazon Web Services (AWS) (2014). The Nipponbare reference genome that was used to generate these BAM files was downloaded from the Rice Annotation Project Database (RAP-DB) (<xref ref-type="bibr" rid="B22">Sakai et al., 2013</xref>). Regarding the rice phenotypic datasets, our research group acquired the datasets from the International Rice Research Institute&#x2019;s official web portal (<ext-link ext-link-type="uri" xlink:href="https://snp-seek.irri.org/_download.zul">https://snp-seek.irri.org/_download.zul</ext-link>).</p>
<p>Furthermore, a selected set of 1043 <italic>Arabidopsis</italic> FASTQ files have also been acquired from the SRA Run Selector with project number PRJNA273563 using the SRA Toolkit 3.0.0 (<ext-link ext-link-type="uri" xlink:href="https://github.com/ncbi/sra-tools">https://github.com/ncbi/sra-tools</ext-link>). These <italic>Arabidopsis</italic> accessions are part of the 1135 <italic>Arabidopsis</italic> accessions presented in the 1001 Genomes Consortium (<xref ref-type="bibr" rid="B1">Alonso-Blanco et al., 2016</xref>). Since the collected files are in FASTQ format, we have also collected the <italic>Arabidopsis thaliana</italic> TAIR10 reference genome from the Phytozome website for mapping the FASTQ files to the reference genome and generating the BAM files. We also directly downloaded the <italic>Arabidopsis</italic> VCF files from the 1001 Genomes Consortium (<ext-link ext-link-type="uri" xlink:href="https://1001genomes.org/data/GMI-MPI/releases/v3.1/">https://1001genomes.org/data/GMI-MPI/releases/v3.1/</ext-link>).</p>
</sec>
<sec id="s2-2">
<title>Data processing</title>
<p>In data processing, the datasets for the promoter regions component and the CNV analysis component of the GenVarX toolset were processed separately using a different set of open-source and publicly available tools. There are four types of datasets that are utilized in the promoter regions component. The datasets are TF datasets, predicted transcription factor binding sites datasets, motif-gene regulation datasets, and binding TF position probability matrices. The first three types of datasets only need some additional processing for extracting the mandatory columns that are used in the promoter regions component. The binding TF position probability matrices, on the other hand, are processed further using the Ceqlogo tool in the Meme Suite (<xref ref-type="bibr" rid="B2">Bailey et al., 2015</xref>). The details about the binding TF position probability matrices and the sequence logo figures will be utilized in the promoter regions component to overlap with genomic variations (SNP and Indels) datasets and show the potential impacts of nucleotide variations on conserved positions in motifs.</p>
<p>For the CNV analysis component, there were two types of datasets involved, the FASTQ and BAM files. The FASTQ files were required to be processed and converted into BAM files so that they can be used as inputs to the cn. MOPS package. In the data processing of the FASTQ files, the files were aligned with their respective reference genome using the Burrows-Wheeler Aligner tool (BWA) version 0.7.17 (<xref ref-type="bibr" rid="B16">Li and Durbin, 2009</xref>) to create outputs in Sequence Alignment Map (SAM) format. Moreover, the SortSam, MarkDuplicates, and AddOrReplaceReadGroups commands in the Genome Analysis Toolkit (GATK) version 4.2.6.1 (<xref ref-type="bibr" rid="B19">McKenna et al., 2010</xref>) were used to sort the SAM files, mark duplications, assign reads to read groups, and output the final BAM files required for the next step.</p>
<p>In order to generate the CNV results, we utilized the cn. MOPS R package (<xref ref-type="bibr" rid="B14">Klambauer et al., 2012</xref>) written in C&#x2b;&#x2b; and R by <xref ref-type="bibr" rid="B14">Klambauer et al. (2012)</xref>. The cn. MOPS package can take in BAM files to calculate the coverage depth of each position across accessions and perform read variations decomposition based on the mixture components and Poisson distributions across accessions using a Bayesian method (<xref ref-type="bibr" rid="B14">Klambauer et al., 2012</xref>). Because of its implementation, this package can achieve a low false discovery rate (FDR) as high noise data is filtered out during the calculation process. The cn. MOPS package has demonstrated good performance and significance using metrics such as the precision-recall area-under-curve (PR AUC) and recall rate by comparing itself with other methods. The outputs of this package that were useful to our CNV analysis component are the CNV individual hits of each accession and the CNV consensus regions across all samples.</p>
<p>For the analysis, we have written R scripts to perform CNV calculations for each organism separately. Each R script that uses the cn. MOPS package takes in the file paths of the BAM files as inputs. The R script was designed to perform CNV analysis on the chromosomal sequences of an organism. Other non-chromosomal sequences like scaffold sequences, mitochondria sequences, and more were not included in the CNV analysis. Upon the completion of CNV analysis, CNV individual hits and consensus regions were collected and ready to be uploaded to the database.</p>
</sec>
<sec id="s2-3">
<title>Data storing</title>
<p>In the data storing section, we provide details about database, data upload methods, and data indexing methods. Uploading data into databases and storing the data with proper indexing can enhance the data query by speeding up the process and preserving the data for long-term usage. In our GenVarX toolset development, we adopted this common practice in order to provide good services to users.</p>
<p>In the GenVarX toolset development, the MySQL database integrated into the SoyKB and KBCommons web portals was utilized to store the datasets for both the promoter regions component and the CNV analysis component of different organisms. With regard to uploading the datasets to the MySQL database, the open-source SQLAlchemy package (<xref ref-type="bibr" rid="B3">Bayer, 2012</xref>) written in Python was used to assist the data upload process. For datasets that are very large in size such as the CNV consensus regions datasets and genotype datasets, the B&#x2b; tree indexing method was used to index the tables in the database. Having all necessary datasets in the database, queries from the web applications can be facilitated as the data in the database can be searched and returned to the users.</p>
</sec>
<sec id="s2-4">
<title>Web development</title>
<p>The GenVarX toolset is presented as a web-based application to the users. Therefore, the user-interactive parts of the GenVarX toolset were web-focused. In web development, several programming languages, libraries, and frameworks were utilized. Because the soybean GenVarX toolset and the GenVarX toolset of other organisms are being deployed on different platforms, the used frameworks are slightly different between the two GenVarX toolsets.</p>
<p>In the development of the soybean GenVarX toolset, HTML, CSS, JavaScript, PHP, and SQL programming languages were utilized in coding both the promoter and the CNV analysis components. Among the four programming languages, HTML, CSS, and JavaScript were focused on the front-end of the web application, while PHP and SQL were focused on the back-end development. In the front-end of the web application, the jQuery JavaScript library was used for enhancing the JavaScript functions and sending requests to the back-end for data retrieval. The PHP back-end of the web application was focused on rendering PHP code and communicating with the database to collect data from the database using SQL queries. The entire technological structure is closer to Linux, Apache, MySQL, and PHP (LAMP) stack.</p>
<p>Likewise, the GenVarX toolsets for other organisms also used the same programming languages, and the programming languages were focused on the front-end and back-end, respectively, similar to the soybean GenVarX toolset. The jQuery JavaScript library was also used for the same purposes in the GenVarX toolsets for the other organisms. Nevertheless, the framework used in the GenVarX toolsets for the other organisms was the Laravel framework which encompasses the back-end of the web application in classes that extends the based controller class and manages all the routes of the web application in one place. Using the Laravel framework, the GenVarX toolset can work for many organisms in the same code. Hence, using this framework simplifies the web development processes.</p>
<p>The deployment of the soybean GenVarX toolset and the other universal GenVarX toolsets are on different websites. The soybean GenVarX toolset was deployed on the SoyKB website while the rest of the toolsets were deployed on the KBCommons website. The links to access the GenVarX toolset are placed under the tool section of both websites. Users can click on the links to get redirected to the GenVarX toolset. For simplicity, below are the links to access the different GenVarX toolsets hosted on the SoyKB and KBCommons websites:<list list-type="simple">
<list-item>
<p>&#x2022; Soybean GenVarX toolset: <ext-link ext-link-type="uri" xlink:href="https://soykb.org/SoybeanGenVarX/">https://soykb.org/SoybeanGenVarX/</ext-link>
</p>
</list-item>
<list-item>
<p>&#x2022; Rice GenVarX toolset: <ext-link ext-link-type="uri" xlink:href="https://kbcommons.org/system/tools/GenVarX/Osativa">https://kbcommons.org/system/tools/GenVarX/Osativa</ext-link>
</p>
</list-item>
<list-item>
<p>&#x2022; <italic>Arabidopsis</italic> GenVarX toolset: <ext-link ext-link-type="uri" xlink:href="https://kbcommons.org/system/tools/GenVarX/Athaliana">https://kbcommons.org/system/tools/GenVarX/Athaliana</ext-link>
</p>
</list-item>
</list>
</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<p>In the results section, the promoter regions and the CNV analysis components of the GenVarX toolset are illustrated. The results focus on the functionalities, interfaces, inputs, and outputs of each component of the GenVarX toolset.</p>
<sec id="s3-1">
<title>Promoter regions component</title>
<p>The promoter regions component consists of a data search page, an independent promoter results page, and a phenotype data viewing page. The promoter search windows on the data search page allow users to search by gene IDs or search by binding TFs. Both windows take in user inputs to perform queries and render the results on the promoter results page and the phenotype data viewing page. Here, each search method is discussed separately.</p>
</sec>
<sec id="s3-2">
<title>Promoter regions component&#x2014;Search by Gene IDs</title>
<p>In the Search by Gene IDs of the promoter regions component, there is a window that has one gene identifiers input box, an upstream length input box, and a search button (<xref ref-type="fig" rid="F1">Figure 1A</xref>). The input box allows users to input multiple gene identifiers at one time, and each gene identifier must be separated into a new line. The upstream length input box takes an integer value from users to calculate upstream promoter regions of the inputted genes. When the search button is clicked, the query is done for each gene, that is, searchable in the database. At the same time, TF binding sites that are in the promoter region and all relevant information are also fetched from the database and displayed on the promoter results page.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>
<bold>(A)</bold> The Search by Gene IDs window in the promoter regions component. <bold>(B)</bold> The Search by Binding TFs window in the promoter regions component.</p>
</caption>
<graphic xlink:href="fgene-14-1251382-g001.tif"/>
</fig>
<p>On the promoter results page, the TF binding sites results are grouped by genes and shown independently for one section per gene (<xref ref-type="fig" rid="F2">Figure 2</xref>). Each section begins with the queried gene identifier along with the chromosome number, coordinates, and strand information. Below the gene identifier, a calculated upstream promoter region is shown. Underneath that, a TF binding sites table with information by row for each TF binding site in the promoter region displays information such as the TF binding site chromosome number, coordinates, strand, TF binding site identifiers, TF family type, and Williams 82 version 2 gene binding sequence. Each TF binding site identifier in the table is a clickable hyperlink. Users can click on the TF binding site identifier that they are interested in to retrieve more details about that TF binding site.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The promoter results page is rendered based on the query from the Search by Gene IDs section. This results page presents the TF binding sites of a gene in a table along with the sequence logo figure and position-nucleotide table.</p>
</caption>
<graphic xlink:href="fgene-14-1251382-g002.tif"/>
</fig>
<p>When users click on a TF binding site identifier (Binding_TF), a sequence logo figure and a position-nucleotide table will be loaded onto the respective section of the promoter results page. From the sequence logo figure, users can visualize the possible nucleotides in that TF binding site along with the information entropy of each nucleotide calculated using the Shannon entropy formula (<xref ref-type="bibr" rid="B24">Schneider and Stephens, 1990</xref>). The table below the sequence logo figure shows each position and each nucleotide of the Williams 82 version 2 gene binding sequence in a tabular form to assist in comparing the nucleotides in the Williams 82 version 2 gene binding sequence with the possible nucleotides of the TF binding sites shown in the sequence logo figures. The comparison provides insight into the sequence conservation of the nucleotides in the Williams 82 version 2 gene binding sequence and the TF binding sites.</p>
<p>Apart from that, the GenVarX toolset also shows SNPs and Indels in allele tables along with the counts of accessions having the corresponding alleles. The counts on the table are clickable and able to redirect users to the phenotype data viewing page (<xref ref-type="fig" rid="F3">Figure 3A</xref>). On the phenotype data viewing page, users not only can see accessions that have a particular allele but also able to connect the accessions with phenotype data. In the phenotype accordion drop-down menu, users can select the phenotypes in order to view or download the phenotype data. The phenotype headings on the table are also clickable to plot violin plots or bar plots depending on the data type (quantitative or qualitative) of that phenotype column (<xref ref-type="fig" rid="F3">Figures 3B, C</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>
<bold>(A)</bold> The phenotype data viewing page for users to select genotypes and phenotypes they are interested in and view or download the data. <bold>(B)</bold> The distribution of quantitative phenotype data plotted against genotypes in a violin plot figure. <bold>(C)</bold> The distribution of qualitative phenotype data plotted against genotypes in a bar plot.</p>
</caption>
<graphic xlink:href="fgene-14-1251382-g003.tif"/>
</fig>
<p>The purpose of generating violin plots and bar plots is to show the distributions of the phenotype data for different alleles. From a violin plot, users can understand the quantitative distribution of phenotype data based on the maximum, minimum, first quantile, third quantile, mean, and median values of that box. Likewise, users can uncover the qualitative distribution of phenotype data in a bar plot according to the counts of different qualitative categories. From the plots, users can also potentially understand the significance and linkage between alleles and phenotypes. Besides the plots, summary tables are also provided at the bottom of each figure to summarize the counts of accessions based on genotypes and either the existence of phenotype data for quantitative measurement or the categories of qualitative measurement. At the bottom of the page, users can also visualize the distribution of improvement status of accessions in different genotypes to understand the linkage between phenotype, improvement status, and genotypes.</p>
</sec>
<sec id="s3-3">
<title>Promoter regions component&#x2014;Search by Binding TFs</title>
<p>The Search by Binding TFs of the promoter regions component has a window, that is, composed of a TF binding site input box, a gene binding chromosome dropdown list, an upstream length input box, and a search button (<xref ref-type="fig" rid="F1">Figure 1B</xref>). In the TF binding site identifier input box, users can input multiple TF binding site identifiers with each in a new line. The gene-binding chromosome drop-down menu allows users to select one chromosome per search. The upstream length input box is for users to input an integer value for upstream promoter regions of genes&#x2019; calculations. When users click on the search button, the user input information is utilized in performing queries, and the results are returned to the promoter results page.</p>
<p>In the promoter results page, the information of each TF binding site identifier and its corresponding regulating genes are shown as an independent table (<xref ref-type="fig" rid="F4">Figure 4A</xref>). In each table, the TF binding site identifier is clickable to redirect to a new page to display a sequence logo figure, position-nucleotide table, as well as SNPs and Indels in allele tables (<xref ref-type="fig" rid="F4">Figure 4B</xref>). Similar to the results page in the Search by Gene IDs section, the alleles can also be linked with phenotype data in the phenotype data viewing page, and the functionalities such as data viewing, data downloading, and data plotting are also included as well (<xref ref-type="fig" rid="F3">Figures 3A&#x2013;C</xref>).</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>
<bold>(A)</bold> The promoter results page is rendered based on the query from the Search by Binding TFs section. It includes tables showing the information on TF binding sites and target genes. <bold>(B)</bold> This page is shown when users select a TF binding site identifier. This page has information related to the selected TF binding site, sequence logo figure, and position-nucleotide table.</p>
</caption>
<graphic xlink:href="fgene-14-1251382-g004.tif"/>
</fig>
</sec>
<sec id="s3-4">
<title>Copy number variation (CNV) component</title>
<p>The CNV analysis component consists of three different search sections which are the Search by Gene IDs section, Search By Accession and Copy Numbers section, and Search by Chromosome and Region section. Each section has a search window for users to input data for queries and the results of the queries are rendered on the corresponding results pages. Here, each section is discussed in more detail.</p>
</sec>
<sec id="s3-5">
<title>CNV analysis component&#x2014;Search by Gene IDs</title>
<p>In the Search by Gene IDs of the CNV analysis component, there is one gene ID input box, a data option dropdown menu, and a search button in a window (<xref ref-type="fig" rid="F5">Figure 5A</xref>). Users can input multiple genes of interest into the gene IDs input box, and each gene ID has to be separated into a new line. In the data option dropdown menu, users can select either consensus regions or individual hits. The consensus regions option is for CNV data summarized across accessions, and the individual hits option is for the CNV individual region of each accession. After the user completes the inputs, the user can click on the search button to perform queries and redirect to the results page.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>
<bold>(A)</bold> The Search by Gene IDs window in the CNV analysis component. <bold>(B)</bold> The Search By Accession and Copy Numbers window in the CNV analysis component. <bold>(C)</bold> The Search by Chromosome and Region window in the CNV analysis component.</p>
</caption>
<graphic xlink:href="fgene-14-1251382-g005.tif"/>
</fig>
<p>There are three sections on the Search by Gene IDs&#x2019; results page which are the queried genes section, the CNV regions and accession counts section, and the neighboring genes in different CNV regions section (<xref ref-type="fig" rid="F6">Figure 6</xref>). The queried genes section displays genes&#x2019; relevant information like chromosomes, coordinates, strands, identifiers, and descriptions. Based on the coordinates of the queried genes, CNV regions that enclosed the queried gene coordinates are displayed in the CNV regions and accession counts section as a table along with the counts of accessions within copy numbers (CN0&#x2013;CN8). According to the cn. MOPS tool, CN0 and CN1 represent loss, CN2 is normal, and CN3 to CN8 represent gain (<xref ref-type="bibr" rid="B14">Klambauer et al., 2012</xref>). Each CN region and the accession counts within copy numbers are organized in a row. At the end of each row, there are view details button and connect phenotypes button that can be clicked to redirect to the detail viewing page and phenotype data viewing page. In the neighboring genes in different CNV regions section, each CNV region and the genes within that CNV region are shown as an independent table.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>The results render on the results page when users use the Search by Gene IDs window in the CNV analysis component.</p>
</caption>
<graphic xlink:href="fgene-14-1251382-g006.tif"/>
</fig>
<p>In the detail viewing page, the distribution of the improvement status in different copy numbers is presented in a bar plot fashion for the selected CNV region (<xref ref-type="fig" rid="F7">Figure 7</xref>). From the bar plot, users can visualize the distributions and uncover significant improvement status that links with a particular copy number if possible. At the bottom of the figure, there is a summary table that summarizes the counts of accessions by improvement status and copy numbers. Within the table, the percentages of accession counts are also calculated out of total accessions. Apart from the figure and summary table, a full table with information such as CNV region, CNV region length, accessions, improvement status, and copy numbers is also provided.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>The detail viewing page shows the distribution of the improvement status of the selected CNV region along with a summary table of the figure and a full table of the whole CNV region, accessions, improvement status, and copy numbers data.</p>
</caption>
<graphic xlink:href="fgene-14-1251382-g007.tif"/>
</fig>
<p>If users would like to gain information about accessions and phenotypes, they can click the connect phenotypes button to redirect to the phenotype data viewing page (<xref ref-type="fig" rid="F8">Figure 8A</xref>). On the phenotype data viewing page, users can select the copy numbers of interest and click the view data button for seeing the data or the download data button to collect the data in a comma-separated values (CSV) file format. Additionally, users can also select copy numbers and phenotypes of interest to overlap and view the data. The data in a tabular fashion allows users to click on a phenotype heading for plotting the distributions of the corresponding phenotype data in a violin plot or a bar chart depending on the data type (quantitative or qualitative) (<xref ref-type="fig" rid="F8">Figures 8B, C</xref>).</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>
<bold>(A)</bold> The phenotype data viewing page for users to select copy numbers and phenotypes they are interested in and view or download the data. <bold>(B)</bold> The distribution of a quantitative trait is plotted against copy numbers in a violin plot figure. <bold>(C)</bold> The distribution of a qualitative trait is plotted against copy numbers in a bar plot.</p>
</caption>
<graphic xlink:href="fgene-14-1251382-g008.tif"/>
</fig>
<p>In a violin plot, users can visualize the quantitative distributions of the phenotype data by different copy numbers. If users hover the pointer on a box in the figure, they can visualize the maximum, minimum, first quantile, third quantile, mean, and median values of that box. Users can also show and hide copy numbers by toggling the elements in the legend. In a bar plot, users can visualize and compare between counts of qualitative categories in the phenotype data by copy numbers. Similarly, users can also toggle elements in the legend to show or hide categories. At the bottom of the violin plot or bar plot, there is a summary table that summarizes the count of accessions by copy numbers. The summarization for quantitative data is counts of accessions by the existence of phenotype data, whereas the summarization for qualitative data is counts and percentages of accessions of each qualitative category. The last figure on the page is a bar plot that shows the distribution of the improvement status of accessions based on selected copy numbers. The figure aims to provide a linkage between the phenotype, improvement status, and copy numbers so that users can understand the improvement status that leads to the phenotype patterns by copy numbers.</p>
</sec>
<sec id="s3-6">
<title>CNV analysis component&#x2014;Search by Accession and Copy Numbers</title>
<p>The Search By Accession and Copy Numbers section of the CNV analysis component has an accession input box, a copy number input box, a data option dropdown menu, and a search button in a window (<xref ref-type="fig" rid="F5">Figure 5B</xref>). Users can input only one accession into the accession input box, input multiple copy numbers into the copy numbers input box (each in a new line), and select either consensus regions or individual hits in the data option dropdown menu. Upon completing all the fields in the window, users can click on the search button so that queries can be performed and rendered on the results page.</p>
<p>On the results page, there is only one table that has the information related to the inputted accession and copy numbers (<xref ref-type="fig" rid="F9">Figure 9</xref>). The CNV regions on the table have details such as the region chromosome, region start, region end, width, strand, accession, and copy number. The purpose of this table is to show users all the CNV regions of a particular accession and copy numbers.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>The results page redirected from the Search By Accession and Copy Numbers window to show CNV regions related to the inputted accession and copy numbers.</p>
</caption>
<graphic xlink:href="fgene-14-1251382-g009.tif"/>
</fig>
</sec>
<sec id="s3-7">
<title>CNV analysis component&#x2014;Search by Chromosome and Region</title>
<p>The Search by Chromosome and Region section of the CNV analysis component has one window that consists of a chromosome input box, a starting position input box, an ending position input box, a data option dropdown menu, and a search button (<xref ref-type="fig" rid="F5">Figure 5C</xref>). Users can input the region of interest with chromosome, starting position, and ending position into the respective input boxes and select either consensus region or individual hits in the data option dropdown menu. When users are ready to perform queries, they can click the search button, and, simultaneously, the users are redirected to the results page.</p>
<p>On the results page, there are two sections which are the queried CNV regions and accession counts section, and the accessions and copy numbers within the queried CNV region section (<xref ref-type="fig" rid="F10">Figure 10</xref>). The queried CNV regions and accession counts section shows a table that contains CNV regions that are bounded between the region of interest input by users. In this section, the table also displays the accession counts in each copy number within each CNV region. At the end of the table, users can also access the view details page and phenotype data viewing page with buttons to understand the improvement status distribution in that CNV region or connect copy numbers and accessions with phenotype data. In order to show more details of each CNV region, the accessions and copy numbers within the queried CNV region section present each CNV region along with all the accessions and copy numbers of that CNV region in a table. If users are interested in knowing the copy number variations within a region of interest, the results presented on this results page will suit their needs.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>The results page redirected from the Search by Chromosome and Region window to display CNV regions between users&#x2019; region of interest.</p>
</caption>
<graphic xlink:href="fgene-14-1251382-g010.tif"/>
</fig>
</sec>
<sec id="s3-8">
<title>Case studies</title>
<p>Analysis of CNV distribution by improvement status offers insight into soybean domestication-related gene gain that impacts plant height.</p>
<p>Gibberellin acid oxidase 2 (<italic>GA2ox</italic>) is an enzyme that, besides other enzymes, converts bioactive phytohormone gibberellins (GA) into inactive forms (<xref ref-type="bibr" rid="B26">Thomas et al., 1999</xref>). Modulation of GAs metabolism genes played an important role in the green revolution of crop improvement. In soybean, reducing trailing growth and shoot length were observed in soybean <italic>GA2ox8</italic> overexpressing mutants. Furthermore, plant height of wild <italic>G. soja</italic> (<italic>Glycine soja</italic>) ancestor W05 was associated with less copy number of <italic>GA2ox8A</italic> (<italic>Glyma.13G287600</italic>) and <italic>GA2ox8B</italic> (<italic>Glyma.13G288000</italic>) in comparison to generally shorter cultivated <italic>G.</italic> max <italic>C08</italic> (<xref ref-type="bibr" rid="B30">Wang et al., 2021</xref>). Thus, this suggests an important role in the <italic>GA2ox8</italic> copy number increase during domestication. Here, we analyzed CNV in the <italic>GA2ox8</italic>-related cn. MOPS predicted CNV associated region, that is, shown in our data on chromosome 13 (Chr13:38,798,562-38,802,911). There are 513 accessions with normal CN, 212 accessions with CN loss, and 341 accessions with CN gain (<xref ref-type="fig" rid="F11">Figure 11A</xref>). <xref ref-type="fig" rid="F11">Figure 11B</xref> illustrates the distribution of CNVs by improvement status in the soybean 1066 accessions and demonstrates that all G. soja accessions possess either CN1 or CN2 that are considered as loss or normal CNV whereas accessions with the other improvement status are all <italic>G.</italic> max (<italic>Glycine max</italic>) and can potentially bear more <italic>GA2ox8</italic> copies. This result is in accordance with <xref ref-type="bibr" rid="B30">Wang et al. (2021)</xref> and thus, supports the hypothesis of the <italic>GA2ox8</italic> gain during soybean domestication (<xref ref-type="bibr" rid="B30">Wang et al., 2021</xref>). We further associated the observed CNV with soybean plant height phenotype (<xref ref-type="fig" rid="F12">Figure 12A</xref>). Since there is no phenotype data available for any G. soja accessions for plant height in the GRIN database, the conclusions made based on this analysis might be influenced by this fact. However, when comparing the three CNVs with the highest accession counts/known phenotype (CN2 - normal, CN1&#x2014;loss, and CN4&#x2014;gain, <xref ref-type="fig" rid="F12">Figure 12B</xref>) here we can see median shifts to increase plant height of accessions with CNV loss (CNV1) in comparison to slightly reduced plant height of normal CNV (CNV2) and CNV gain (CNV4). Most importantly, among 208 accessions with CN4, 141 are Elite accessions with 83 out of the 102-known phenotype accessions in this group. Thus, this result indicates that the <italic>GA2ox8</italic> CN gain might be responsible for plant height in cultivated soybean varieties as demonstrated by <xref ref-type="bibr" rid="B30">Wang et al. (2021)</xref>; <xref ref-type="bibr" rid="B30">Wang et al. (2021)</xref>.</p>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>
<bold>(A)</bold> The CNV region on chromosome 13 that is, associated with the GA2ox8 gene. <bold>(B)</bold> The soybean 1066 accessions&#x2019; improvement status distribution in different copy numbers.</p>
</caption>
<graphic xlink:href="fgene-14-1251382-g011.tif"/>
</fig>
<fig id="F12" position="float">
<label>FIGURE 12</label>
<caption>
<p>
<bold>(A)</bold> The violin plot illustrates the statistical distribution of plant height of soybean 1066 accessions for CNV in the GA2ox8 gene-associated region on chromosome 13. <bold>(B)</bold> A summary table associated with the violin plot documents the corresponding phenotype counts and missing information.</p>
</caption>
<graphic xlink:href="fgene-14-1251382-g012.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>The GenVarX toolset offers tools that enable online analyses of the associations of pre-defined phenotypes with their variations in promoter regions and CNV on soybean, rice, and <italic>Arabidopsis</italic>. The toolset provides query, association testing, visualization, and download capabilities for users to obtain new insight from the promoter and CNV data. Using the toolset, users can interact with the web interfaces to do their research without the need of spending long-time processing and running big data to gain promoter and CNV results. It also provides opportunities for users who do not have large-scale servers and large storage space to have access to the promoter and CNV data.</p>
<p>In the development of the GenVarX toolset, we faced some challenges in SNP data processing and phenotype data retrieval. SNP data in a VCF file usually consists of many positions and accessions in a single file. Processing the SNP data in a VCF file and uploading the processed SNP data into a database are usually time-consuming processes. A divide-and-conquer strategy is usually required in code implementation to achieve a certain speed-up in the processing and uploading. Furthermore, multi-processing and multi-threading can also be helpful when processing and uploading the data programmatically. Besides that, retrieving the SNP data from the database could also be a slow process. Therefore, a database indexing method is required to increase the data retrieval speed. Apart from the SNP data processing problem, we faced another challenge that was caused by the limited availability of phenotype data for rice and <italic>Arabidopsis</italic>. To solve this problem, online data explorations are required in order to find suitable data for our GenVarX toolset.</p>
<p>The GenVarX toolset is developed with the incorporation of extensive capabilities that are related to genotype and phenotype. The extensions of showing gene binding sequences, sequence logo figures, mutative variant positions, as well as the linkage of the genotype data to phenotype data are the strengths of this toolset. Although PlantTFDB and PlantRegMap are the main sources for the datasets of the GenVarX toolset, these new capabilities are not developed in their web portals. Another promoter database is the Eukaryotic Promoter Database (EPD) (<xref ref-type="bibr" rid="B21">P&#xe9;rier et al., 2000</xref>) which also allows users to search for promoters. However, important crop species like soybean and rice are not available in that database. In terms of CNV, there is a lack of plant and crop related CNV databases for users to perform queries, visualize CNV, and link CNV to phenotypes. Thus, our research group developed the GenVarX toolset to assist the research community to advance their research.</p>
<p>In future development, our research group will focus on expanding the GenVarX toolset to support more organisms. Besides that, we will also incorporate more phenotypic data into the database to allow users to visualize different phenotypes more easily. Furthermore, integration of CNV results from other tools can also be done to make the CNV component of the GenVarX toolset capable of a more enriched comparative analysis. As mentioned in Gabrielaite et al., which shows the comparisons of different CNV tools in terms of outcomes, metrics, and performance, there are several tools such as GATK gCNV, Lumpy, and DELLY, that were developed with different methodologies that outperform many other CNV tools (<xref ref-type="bibr" rid="B6">Gabrielaite et al., 2021</xref>). These CNV tools can be used to further analyze our data and integrate into the GenVarX toolset so that users can select CNVs from different methods to link with genotype and phenotype.</p>
</sec>
<sec sec-type="conclusion" id="s5">
<title>Conclusion</title>
<p>In the GenVarX toolset development, we have collected and processed publicly available data from various sources and platforms. Having the data, we have built the GenVarX toolset that has promoter regions and CNV analysis components for soybean, rice, and <italic>Arabidopsis</italic>. The soybean GenVarX toolset is deployed on the SoyKB website (<ext-link ext-link-type="uri" xlink:href="https://soykb.org/SoybeanGenVarX/">https://soykb.org/SoybeanGenVarX/</ext-link>) whereas the universal GenVarX toolset for other organisms is deployed on the KBCommons website (<ext-link ext-link-type="uri" xlink:href="https://kbcommons.org/system/tools/GenVarX/Osativa">https://kbcommons.org/system/tools/GenVarX/Osativa</ext-link> and <ext-link ext-link-type="uri" xlink:href="https://kbcommons.org/system/tools/GenVarX/Athaliana">https://kbcommons.org/system/tools/GenVarX/Athaliana</ext-link>). The broad plant research community can utilize the GenVarX toolset as a comprehensive source of information to gain insights into soybean, rice, and <italic>Arabidopsis</italic> promoter regions or CNV analysis outcomes. More specifically, a better understanding of variations in the TF binding sites and CNV can be predicted with the toolset. Hence, it serves as a valuable pre-experimental step for further gene transcription studies.</p>
</sec>
<sec id="s6">
<title>Available and requirements</title>
<p>Project Name: GenVarX</p>
<p>Project Homepage: <list list-type="simple">
<list-item>
<p>&#x2022; Soybean GenVarX Toolset: <ext-link ext-link-type="uri" xlink:href="https://soykb.org/SoybeanGenVarX/">https://soykb.org/SoybeanGenVarX/</ext-link>
</p>
</list-item>
<list-item>
<p>&#x2022; Rice GenVarX Toolset: <ext-link ext-link-type="uri" xlink:href="https://kbcommons.org/system/tools/GenVarX/Osativa">https://kbcommons.org/system/tools/GenVarX/Osativa</ext-link>
</p>
</list-item>
<list-item>
<p>&#x2022; <italic>Arabidopsis</italic> GenVarX Toolset: <ext-link ext-link-type="uri" xlink:href="https://kbcommons.org/system/tools/GenVarX/Athaliana">https://kbcommons.org/system/tools/GenVarX/Athaliana</ext-link>
</p>
</list-item>
</list>
</p>
<p>Programming Languages:<list list-type="simple">
<list-item>
<p>&#x2022; Data Analytics: Python and R</p>
</list-item>
<list-item>
<p>&#x2022; Web Development: PHP, HTML, CSS, and JavaScript</p>
</list-item>
</list>
</p>
<p>Other Requirements:<list list-type="simple">
<list-item>
<p>&#x2022; Data Analytics:</p>
<list list-type="simple">
<list-item>
<p>o Python 3.7.0 or higher</p>
</list-item>
<list-item>
<p>o R 3.6.0 or higher</p>
</list-item>
<list-item>
<p>o SQLAlchemy 1.4.41 or higher</p>
</list-item>
<list-item>
<p>o cn.MOPS 1.40.0 or higher</p>
</list-item>
<list-item>
<p>o Burrows-Wheeler Aligner (BWA) 0.7.17</p>
</list-item>
<list-item>
<p>o Genome Analysis Toolkit (GATK) 4.2.6.1</p>
</list-item>
</list>
</list-item>
<list-item>
<p>&#x2022; Web Development:</p>
<list list-type="simple">
<list-item>
<p>o PHP 8</p>
</list-item>
</list>
</list-item>
<list-item>
<p>&#x2022; Web Browsing:</p>
<list list-type="simple">
<list-item>
<p>o Google Chrome (Recommended), Firefox, or Microsoft Edge</p>
</list-item>
</list>
</list-item>
</list>
</p>
<p>Source Code:<list list-type="simple">
<list-item>
<p>&#x2022; GenVarX Data Processing Scripts: <ext-link ext-link-type="uri" xlink:href="https://github.com/yenon118/GenVarX_Data_Processing">https://github.com/yenon118/GenVarX_Data_Processing</ext-link>
</p>
</list-item>
<list-item>
<p>&#x2022; Soybean GenVarX Toolset Source Code: <ext-link ext-link-type="uri" xlink:href="https://github.com/yenon118/SoybeanGenVarX">https://github.com/yenon118/SoybeanGenVarX</ext-link>
</p>
</list-item>
<list-item>
<p>&#x2022; Rice and <italic>Arabidopsis</italic> Toolsets Source Code: <ext-link ext-link-type="uri" xlink:href="https://github.com/yenon118/GenVarX">https://github.com/yenon118/GenVarX</ext-link>
</p>
</list-item>
</list>
</p>
<p>License:<list list-type="simple">
<list-item>
<p>&#x2022; Soybean GenVarX Toolset: MIT License</p>
</list-item>
<list-item>
<p>&#x2022; Rice GenVarX Toolset: MIT License</p>
</list-item>
<list-item>
<p>&#x2022; <italic>Arabidopsis</italic> GenVarX Toolset: MIT License</p>
</list-item>
</list>
</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary Material, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s8">
<title>Author contributions</title>
<p>YC collected the transcription factor related data, sequencing data, annotation data, genotypic, and phenotypic data. YC created scripts to generate the copy number variation data and developed the GenVarX toolset. YC wrote the first draft of the manuscript. TJ, M&#x160;, and KB proposed the tool functionalities. YC, AM, ND, M&#x160;, KB, and TJ participated in the GenVarX toolset design process. M&#x160; and JB performed numerous tests in the tool development cycles, conducted analyses, and assembled presentable results. YC, KB, M&#x160;, and TJ revised and edited the manuscript. TJ provided expert guidance on the conceptualization and development process of the toolset in SoyKB and KBCommons. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s9">
<title>Funding</title>
<p>The research was supported using Missouri soybean farmers&#x2019; checkoff dollars provided by the United Soybean Board (USB). The principal investigators, project title, and grant ID information are as below: 1. Dr. Trupti Joshi and Dr. Kristin Bilyeu; Applied Genomics to Improve Soybean Seed Protein; &#x23;1920-152-0131-C. 2. Dr. Trupti Joshi and Dr. Kristin Bilyeu; Enhancing Soybean Applied Genomics Tools for Improving Soybean; &#x23;2220-152-0202. 3. Dr. Trupti Joshi and Dr. Kristin Bilyeu; Leveraging Genomics to Enhance The US Soybean Quality Reputation; &#x23;2332-201-0101.</p>
</sec>
<ack>
<p>Our research group would like to acknowledge Sai Preethi Induri for assisting us in setting up the RiceKB on the KBCommons website. We also would like to thank Ajay Kumar and Adama Tukuli for providing assistance in searching for rice phenotypic datasets.</p>
</ack>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12">
<title>Abbreviations</title>
<p>BAM, binary alignment map; BWA, burrows-wheeler aligner; ENA, european nucleotide archive; GATK, genome analysis toolkit; GSA, genome sequence archive; GRIN, resources information network; Indels, insertions and deletions; NCBI, national center for biotechnology information; SAM, sequence alignment map; SNP, single nucleotide polymorphism; VCF, variant call format; WGRS, whole genome re-sequencing.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alonso-Blanco</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Andrade</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Becker</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bemm</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Bergelson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Borgwardt</surname>
<given-names>K. M.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>1,135 Genomes Reveal the Global Pattern of Polymorphism in <italic>Arabidopsis thaliana</italic>
</article-title>. <source>Cell</source> <volume>166</volume>, <fpage>481</fpage>&#x2013;<lpage>491</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2016.05.063</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bailey</surname>
<given-names>T. L.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Grant</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Noble</surname>
<given-names>W. S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>The MEME Suite</article-title>. <source>Nucleic Acids Res.</source> <volume>43</volume>, <fpage>W39</fpage>&#x2013;<lpage>W49</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkv416</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bayer</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2012</year>). <source>SQLAlchemy. Mountain view: aosabook.org</source>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bolger</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schwacke</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gundlach</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Schmutzer</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Arend</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>From plant genomes to phenotypes</article-title>. <source>J. Biotechnol.</source> <volume>261</volume>, <fpage>46</fpage>&#x2013;<lpage>52</lpage>. <pub-id pub-id-type="doi">10.1016/j.jbiotec.2017.06.003</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Castro-Mondragon</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Riudavets-Puig</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Rauluseviciute</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Berhanu lemma</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Turchi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Blanc-Mathieu</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>JASPAR 2022: the 9th release of the open-access database of transcription factor binding profiles</article-title>. <source>Nucleic Acids Res.</source> <volume>50</volume>, <fpage>D165</fpage>&#x2013;<lpage>D173</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkab1113</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gabrielaite</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Torp</surname>
<given-names>M. H.</given-names>
</name>
<name>
<surname>Rasmussen</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Andreu-S&#xe1;nchez</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Vieira</surname>
<given-names>F. G.</given-names>
</name>
<name>
<surname>Pedersen</surname>
<given-names>C. B.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>A Comparison of Tools for Copy-Number Variation Detection in Germline Whole Exome and Whole Genome Sequencing Data</article-title>. <source>Cancers</source> <volume>13</volume>, <fpage>6283</fpage>. <pub-id pub-id-type="doi">10.3390/cancers13246283</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goff</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Vaughn</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mckay</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lyons</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Stapleton</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gessler</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>The iPlant Collaborative: cyberinfrastructure for plant biology</article-title>. <source>Front. Plant Sci.</source> <volume>2</volume>, <fpage>34</fpage>. <pub-id pub-id-type="doi">10.3389/fpls.2011.00034</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goodstein</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Shu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Howson</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Neupane</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hayes</surname>
<given-names>R. D.</given-names>
</name>
<name>
<surname>Fazo</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Phytozome: a comparative platform for green plant genomics</article-title>. <source>Nucleic Acids Res.</source> <volume>40</volume>, <fpage>D1178</fpage>&#x2013;<lpage>D1186</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkr944</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>D.-C.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>Y.-Q.</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>PlantTFDB 4.0: toward a central hub for transcription factors and regulatory interactions in plants</article-title>. <source>Nucleic Acids Res.</source> <volume>45</volume>, <fpage>D1040</fpage>&#x2013;<lpage>D1045</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw982</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Joshi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Fitzpatrick</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Endacott</surname>
<given-names>R. Z.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Soybean knowledge base (SoyKB): a web resource for integration of soybean translational genomics and molecular breeding</article-title>. <source>Nucleic Acids Res.</source> <volume>42</volume>, <fpage>D1245</fpage>&#x2013;<lpage>D1252</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkt905</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Joshi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Patil</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Fitzpatrick</surname>
<given-names>M. R.</given-names>
</name>
<name>
<surname>Franklin</surname>
<given-names>L. D.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Cook</surname>
<given-names>J. R.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>Soybean Knowledge Base (SoyKB): a web resource for soybean translational genomics</article-title>. <source>BMC Genomics</source> <volume>13</volume>, <fpage>S15</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2164-13-S1-S15</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Joshi</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). &#x201c;<article-title>The Evolution of Soybean Knowledge Base (SoyKB)</article-title>,&#x201d; in <source>Plant genomics databases: methods and protocols</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Van Dijk</surname>
<given-names>A. D. J.</given-names>
</name>
</person-group> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Springer New York</publisher-name>), <fpage>149</fpage>&#x2013;<lpage>159</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>M. Y.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Van</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>T.-H.</given-names>
</name>
<name>
<surname>Jeong</surname>
<given-names>S.-C.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>I.-Y.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Whole-genome sequencing and intensive analysis of the undomesticated soybean (Glycine soja Sieb. and Zucc.) genome</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>107</volume>, <fpage>22032</fpage>&#x2013;<lpage>22037</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1009526107</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Klambauer</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Schwarzbauer</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Mayr</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Clevert</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Mitterecker</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bodenhofer</surname>
<given-names>U.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>cn.MOPS: mixture of Poissons for discovering copy number variations in next-generation sequencing data with a low false discovery rate</article-title>. <source>Nucleic Acids Res.</source> <volume>40</volume>, <fpage>e69</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gks003</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Qu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Comparison of structural variants in the whole genome sequences of two Medicago truncatula ecotypes: jemalong a17 and r108</article-title>. <source>BMC Plant Biol.</source> <volume>22</volume>, <fpage>77</fpage>. <pub-id pub-id-type="doi">10.1186/s12870-022-03469-0</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Durbin</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Fast and accurate short read alignment with Burrows&#x2013;Wheeler transform</article-title>. <source>Bioinformatics</source> <volume>25</volume>, <fpage>1754</fpage>&#x2013;<lpage>1760</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp324</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Pan-Genome of Wild and Cultivated Soybeans</article-title>. <source>Cell</source> <volume>182</volume>, <fpage>162</fpage>&#x2013;<lpage>176</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2020.05.023</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rynge</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>PGen: large-scale genomic variations analysis workflow and browser in SoyKB</article-title>. <source>BMC Bioinforma.</source> <volume>17</volume>, <fpage>337</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-016-1227-y</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mckenna</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hanna</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Banks</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Sivachenko</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cibulskis</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kernytsky</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>The Genome Analysis Toolkit: a MapReduce framework for analyzing next-generation DNA sequencing data</article-title>. <source>Genome Res.</source> <volume>20</volume>, <fpage>1297</fpage>&#x2013;<lpage>1303</lpage>. <pub-id pub-id-type="doi">10.1101/gr.107524.110</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Merchant</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Lyons</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Goff</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Vaughn</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ware</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Micklos</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>The iPlant Collaborative: cyberinfrastructure for enabling data to discovery for the life sciences</article-title>. <source>PLoS Biol.</source> <volume>14</volume>, <fpage>e1002342</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pbio.1002342</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>P&#xe9;rier</surname>
<given-names>R. C.</given-names>
</name>
<name>
<surname>Praz</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Junier</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Bonnard</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bucher</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>The eukaryotic promoter database (EPD)</article-title>. <source>Nucleic Acids Res.</source> <volume>28</volume>, <fpage>302</fpage>&#x2013;<lpage>303</lpage>. <pub-id pub-id-type="doi">10.1093/nar/28.1.302</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sakai</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Tanaka</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Numa</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kawahara</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Rice Annotation Project Database (RAP-DB): an integrative and interactive database for rice genomics</article-title>. <source>Plant Cell Physiol.</source> <volume>54</volume>, <fpage>e6</fpage>. <pub-id pub-id-type="doi">10.1093/pcp/pcs183</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Samarakoon</surname>
<given-names>P. S.</given-names>
</name>
<name>
<surname>Sorte</surname>
<given-names>H. S.</given-names>
</name>
<name>
<surname>Stray-Pedersen</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>R&#xf8;dningen</surname>
<given-names>O. K.</given-names>
</name>
<name>
<surname>Rognes</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lyle</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>cnvScan: a CNV screening and annotation tool to improve the clinical utility of computational CNV prediction from exome sequencing data</article-title>. <source>BMC Genomics</source> <volume>17</volume>, <fpage>51</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-016-2374-2</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schneider</surname>
<given-names>T. D.</given-names>
</name>
<name>
<surname>Stephens</surname>
<given-names>R. M.</given-names>
</name>
</person-group> (<year>1990</year>). <article-title>Sequence logos: a new way to display consensus sequences</article-title>. <source>Nucleic Acids Res.</source> <volume>18</volume>, <fpage>6097</fpage>&#x2013;<lpage>6100</lpage>. <pub-id pub-id-type="doi">10.1093/nar/18.20.6097</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<collab>The 3,000 rice genomes project</collab> (<year>2014</year>). <article-title>The 3,000 rice genomes project</article-title>. <source>GigaScience</source> <volume>3</volume>, <fpage>7</fpage>. <pub-id pub-id-type="doi">10.1186/2047-217X-3-7</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thomas</surname>
<given-names>S. G.</given-names>
</name>
<name>
<surname>Phillips</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>Hedden</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Molecular cloning and functional expression of gibberellin 2- oxidases, multifunctional enzymes involved in gibberellin deactivation</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>96</volume>, <fpage>4698</fpage>&#x2013;<lpage>4703</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.96.8.4698</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tian</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>D.-C.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>Y.-Q.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>PlantRegMap: charting functional regulatory maps in plants</article-title>. <source>Nucleic Acids Res.</source> <volume>48</volume>, <fpage>D1104-D1113</fpage>&#x2013;<lpage>D1113</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkz1020</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Valliyodan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>A. V.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Patil</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Otyama</surname>
<given-names>P. I.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Genetic variation among 481 diverse soybean accessions, inferred from genomic re-sequencing</article-title>. <source>Sci. Data</source> <volume>8</volume>, <fpage>50</fpage>. <pub-id pub-id-type="doi">10.1038/s41597-021-00834-w</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Valliyodan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>H. T.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Understanding regulatory networks and engineering for enhanced drought tolerance in plants</article-title>. <source>Curr. Opin. Plant Biol.</source> <volume>9</volume>, <fpage>189</fpage>&#x2013;<lpage>195</lpage>. <pub-id pub-id-type="doi">10.1016/j.pbi.2006.01.019</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.-W.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>F.-L.</given-names>
</name>
<name>
<surname>Luk</surname>
<given-names>C.-Y.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>C.Y.-L.</given-names>
</name>
<name>
<surname>Yung</surname>
<given-names>W.-S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Increased copy number of gibberellin 2-oxidase 8 genes reduced trailing growth and shoot length during soybean domestication</article-title>. <source>Plant J.</source> <volume>107</volume>, <fpage>1739</fpage>&#x2013;<lpage>1755</lpage>. <pub-id pub-id-type="doi">10.1111/tpj.15414</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tammi</surname>
<given-names>M. T.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>CNV-seq, a new method to detect copy number variation using high-throughput sequencing</article-title>. <source>BMC Bioinforma.</source> <volume>10</volume>, <fpage>80</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-10-80</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yevshin</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Sharipov</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Valeev</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kolpakov</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>GTRD: a database of transcription factor binding sites identified by ChIP-seq experiments</article-title>. <source>Nucleic Acids Res.</source> <volume>45</volume>, <fpage>D61-D67</fpage>&#x2013;<lpage>d67</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkw951</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lyu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Narisetti</surname>
<given-names>S. R. K.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Joshi</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Knowledge Base Commons (KBCommons) v1.0: A multi OMICS&#x27; web-based data integration framework for biological discoveries</article-title>,&#x201d; in <conf-name>2018 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</conf-name>, <conf-loc>Madrid, Spain</conf-loc>, <conf-date>December 6 2018</conf-date>, <fpage>589</fpage>&#x2013;<lpage>594</lpage>.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lyu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Narisetti</surname>
<given-names>S. R. K.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Joshi</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Knowledge Base Commons (KBCommons) v1.1: a universal framework for multi-omics data integration and biological discoveries</article-title>. <source>BMC Genomics</source> <volume>20</volume>, <fpage>947</fpage>. <pub-id pub-id-type="doi">10.1186/s12864-019-6287-8</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gou</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lyu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Resequencing 302 wild and cultivated accessions identifies genes related to domestication and improvement in soybean</article-title>. <source>Nat. Biotechnol.</source> <volume>33</volume>, <fpage>408</fpage>&#x2013;<lpage>414</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.3096</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>&#x17b;mie&#x144;ko</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Samelak</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Koz&#x142;owski</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Figlerowicz</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Copy number polymorphism in plant genomes</article-title>. <source>Theor. Appl. Genet.</source> <volume>127</volume>, <fpage>1</fpage>&#x2013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1007/s00122-013-2177-7</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>