<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1645004</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>BioVizSeq: an R package for visualization the element on bio-sequences</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zhao</surname>
<given-names>Shiqi</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/3035213/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Runqi</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>He</surname>
<given-names>Maoqiu</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/2815980/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Shui</surname>
<given-names>Bonian</given-names>
</name>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhang</surname>
<given-names>Yu</given-names>
</name>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<institution>School of Fishery, Zhejiang Ocean University</institution>, <addr-line>Zhoushan</addr-line>,&#xa0;<country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1825103/overview">George V Popescu</ext-link>, Mississippi State University, United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/778029/overview">Zhibin Lv</ext-link>, Sichuan University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1630154/overview">Vitor Lima Coelho</ext-link>, Sistema FIRJAN RJ, Brazil</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Bonian Shui, <email xlink:href="mailto:shuibonian@163.com">shuibonian@163.com</email>; Yu Zhang, <email xlink:href="mailto:zhangy@zjou.edu.cn">zhangy@zjou.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>19</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1645004</elocation-id>
<history>
<date date-type="received">
<day>11</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>01</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Zhao, Zhang, He, Shui and Zhang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Zhao, Zhang, He, Shui and Zhang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The identification and visualization of functional elements within biological sequences offers visual presentation for biologists to integrate annotation, and also helps them to produce high-quality figures for publication. Although there are now some standalone tools that can perform this function, these tools generally lack flexibility and cannot meet personalized needs. Based on the advantages of R language in graphic display, we have developed an R package: BioVizSeq (CRAN: <ext-link ext-link-type="uri" xlink:href="https://cran.r-project.org/package=BioVizSeq">https://cran.r-project.org/package=BioVizSeq</ext-link>. Github: <ext-link ext-link-type="uri" xlink:href="https://github.com/zhaosq2022/BioVizSeq">https://github.com/zhaosq2022/BioVizSeq</ext-link>). It is designed for visualizing the types and distribution of elements within bio-sequences. These data could come from users or analysis programs, such as, GFF/GTF, MEME, SMART, Plantcare, PFAM, CDD and etc. BioVizSeq can be conducted locally or online, providing great convenience for researchers without coding training. Its user-friendly visualization function can simultaneously meet users&#x2019; general needs and personalized exploration.</p>
</abstract>
<kwd-group>
<kwd>BioVizSeq</kwd>
<kwd>biosequence</kwd>
<kwd>element</kwd>
<kwd>visualization</kwd>
<kwd>shinyApp</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="2"/>
<equation-count count="0"/>
<ref-count count="14"/>
<page-count count="9"/>
<word-count count="3165"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Plant Bioinformatics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Biological sequences typically refer to nucleotide sequences (DNA or RNA) or protein sequences, which encode the genetic information of an organism. Nucleotide sequences contain coding regions (open reading frames, ORFs), regulatory elements (such as promoters, enhancers, silencers), splice sites, transcription factor binding sites, and replication origins. Protein sequences, composed of amino acids, fold into specific three-dimensional structures or multi-subunit complexes to carry out various biological functions. Functional motifs or domains in proteins include signal peptides, enzymatic active sites, binding sites, phosphorylation sites, and transmembrane regions, all of which are critical for the protein&#x2019;s function (<xref ref-type="bibr" rid="B12">Savojardo et&#xa0;al., 2023</xref>). In summary, the functional elements and motifs within biological sequences interact to facilitate a wide range of physiological processes in living organisms.</p>
<p>The identification and visualization of functional elements within biological sequences offers visual presentation for biologists to integrate annotation, and also helps them to produce high-quality figures for publication. For nucleic acid sequence structures, GTF (Gene Transfer Format) and GFF (General Feature Format) files are commonly used for display (<xref ref-type="bibr" rid="B11">Pertea and Pertea, 2020</xref>). PlantCare is typically used to analyze and predict cis-regulatory elements on plant gene promoter sequences, but there is no similar tool for animal gene promoters at present (<xref ref-type="bibr" rid="B8">Lescot et&#xa0;al., 2002</xref>). For functional elements on protein sequences, tools such as PFAM (<xref ref-type="bibr" rid="B10">Mistry et&#xa0;al., 2021</xref>), SMART (<xref ref-type="bibr" rid="B9">Letunic et&#xa0;al., 2021</xref>), NCBI-CDD (<xref ref-type="bibr" rid="B14">Wang et&#xa0;al., 2023</xref>), or MEME (<xref ref-type="bibr" rid="B1">Bailey et&#xa0;al., 2015</xref>) are commonly used for analysis and prediction. To better observe the similarities and differences of these elements across different subfamilies and explore gene evolution, these structures are often compared with evolutionary trees. In terms of tools for displaying and combining these structures, there are standalone tools like TBtools (<xref ref-type="bibr" rid="B3">Chen et&#xa0;al., 2020</xref>, <xref ref-type="bibr" rid="B5">2023</xref>) and CFVisual (<xref ref-type="bibr" rid="B4">Chen et&#xa0;al., 2022</xref>). However, the disadvantages of standalone software tools are also present in these tools, such as limited flexibility, low automation, and platform restrictions.</p>
<p>R language has numerous advantages in data analysis and visualization. In terms of data analysis, it offers a wide range of third-party packages for tasks such as data cleaning and transformation. Additionally, R excels in data visualization capabilities; through plotting packages like ggplot2 (<xref ref-type="bibr" rid="B7">Klaus and Galensa, 2017</xref>), it can generate professional and complex graphics with strong customization options. Furthermore, tools like the shiny package enable the creation of interactive graphics, allowing users to interact with the visualizations. Based on these features, we developed a biosequence element visualization package in R: BioVizSeq. In this package, we have written multiple functions for data analysis and visualization. Based on these functions, we have also developed a Shiny app within the package. Theoretically, BioVizSeq is not limited to the visualization of biological sequences. This package not only meets the interactive needs of general users but also supports the personalized display needs of advanced users.</p>
</sec>
<sec id="s2" sec-type="results">
<label>2</label>
<title>Results and discussion</title>
<sec id="s2_1">
<label>2.1</label>
<title>Overview of BioVizSeq</title>
<p>The BioVizSeq package is designed for visualizing the types and distribution of elements within bio-sequences. These data could come from users or analysis programs, such as, GFF/GTF, MEME, SMART, Plantcare, Pfam, CDD and etc (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>). The BioVizSeq provides two modes for displaying: &#x201c;Step by Step&#x201d; and &#x201c;One Step&#x201d;. Since ggplot2 does not provide a geom for generating rounded rectangles, we have written a geom layer that can draw rounded rectangles: geom_rrect. There is a special function BioVizSeq to start our shinyapp interface system, specially customized for the BioVizSeq R package.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>The basic workflow of BioVizSeq package. Input file: result files of software or databases (e.g.: MEME, SMART, PFAM, etc.), or manually organized location file.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1645004-g001.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a data processing workflow for visualizing motif data. Two types of input files are used: sequences (*.fasta, *.gff/gtf) and location/length data. Functions like gff_to_loc(), meme_to_loc(), and others process these files into element location data and sequence length data. Both data types are used to generate visualizations through motif_plot() and other plotting functions.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>BioVizSeq resources</title>
<p>BioVizSeq is open source under Artistic-2.0 license, and all source codes of R package, API documentation, websites, and Shinyapp are stored in the GitHub repository <ext-link ext-link-type="uri" xlink:href="https://github.com/zhaosq2022/BioVizSeq">https://github.com/zhaosq2022/BioVizSeq</ext-link>. BioVizSeq provides a stable version for multi&#x2010;platform installation on the Comprehensive R Archive Network (CRAN) website (<ext-link ext-link-type="uri" xlink:href="https://cran.r-project.org/package=BioVizSeq">https://cran.r-project.org/package=BioVizSeq</ext-link>), and the latest unstable version is available on the GitHub repository. Furthermore, we have created comprehensive API documentation and tutorials, including text content (<ext-link ext-link-type="uri" xlink:href="https://zhaosq2022.github.io/BioVizSeq/">https://zhaosq2022.github.io/BioVizSeq/</ext-link>).</p>
<p>To elucidate the benefits of BioVizSeq, we compared its functionality and resources with several other commonly used software (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). Standalone visualization tools typically offer predefined graphical elements, whereas R enables programmatic customization through its scripting interface. This difference may influence their applicability in scenarios requiring highly tailored&#xa0;visual outputs. TBtools II and CFVisual are desktop applications compiled in Java and Python, respectively. Both have natural advantages in operating desktop software, but there are also some drawbacks or shortcomings, such as requiring local installation, opaque data processing, and inability to personalize display. In contrast, BioVizSeq effectively addresses these issues. In addition, TBtools II also has a similar Basic plot feature, but CFVisual does not. Meanwhile, neither of these supports SMART result display.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Performance comparison of benchmarked tools.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Comment Points</th>
<th valign="middle" align="left">TBtools-II</th>
<th valign="middle" align="left">CFVisual</th>
<th valign="middle" align="left">BioVizSeq</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Cross-platform (Linux, Mac OS, and Windows)</td>
<td valign="middle" align="left">yes</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">yes</td>
</tr>
<tr>
<td valign="middle" align="left">GUI (Graphical User Interface)</td>
<td valign="middle" align="left">yes</td>
<td valign="middle" align="left">yes</td>
<td valign="middle" align="left">yes</td>
</tr>
<tr>
<td valign="middle" align="left">Website</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">yes</td>
</tr>
<tr>
<td valign="middle" align="left">Code writing</td>
<td valign="middle" align="left">No need</td>
<td valign="middle" align="left">No need</td>
<td valign="middle" align="left">Optional</td>
</tr>
<tr>
<td valign="middle" align="left">Programming language</td>
<td valign="middle" align="left">Java</td>
<td valign="middle" align="left">Python</td>
<td valign="middle" align="left">R</td>
</tr>
<tr>
<td valign="middle" align="left">Basic biosequence view</td>
<td valign="middle" align="left">yes</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">yes</td>
</tr>
<tr>
<td valign="middle" align="left">Gene Structure (gff3/gtf)</td>
<td valign="middle" align="left">yes</td>
<td valign="middle" align="left">yes</td>
<td valign="middle" align="left">yes</td>
</tr>
<tr>
<td valign="middle" align="left">MEME</td>
<td valign="middle" align="left">yes</td>
<td valign="middle" align="left">yes</td>
<td valign="middle" align="left">yes</td>
</tr>
<tr>
<td valign="middle" align="left">PFAM</td>
<td valign="middle" align="left">yes</td>
<td valign="middle" align="left">yes</td>
<td valign="middle" align="left">yes</td>
</tr>
<tr>
<td valign="middle" align="left">NCBI CDD</td>
<td valign="middle" align="left">yes</td>
<td valign="middle" align="left">yes</td>
<td valign="middle" align="left">yes</td>
</tr>
<tr>
<td valign="middle" align="left">SMART</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">yes</td>
</tr>
<tr>
<td valign="middle" align="left">Plantcare</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">yes</td>
<td valign="middle" align="left">yes</td>
</tr>
<tr>
<td valign="middle" align="left">Plantcare advance plot</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">yes</td>
</tr>
<tr>
<td valign="middle" align="left">Free combination</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">yes</td>
</tr>
<tr>
<td valign="middle" align="left">Data visualization</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">yes</td>
</tr>
<tr>
<td valign="middle" align="left">Long&#x2010;term maintenance</td>
<td valign="middle" align="left">yes</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">yes</td>
</tr>
<tr>
<td valign="middle" align="left">Link address (source code)</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">no</td>
<td valign="middle" align="left">
<ext-link ext-link-type="uri" xlink:href="https://github.com/zhaosq2022/BioVizSeq/">https://github.com/zhaosq2022/BioVizSeq/</ext-link>
</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>BioVizSeq shinyApp practicality</title>
<p>To make BioVizSeq more accessible to researchers without coding experience, we have developed an intuitive interactive analysis and visualization platform based on Shiny v1.7.5 (<ext-link ext-link-type="uri" xlink:href="https://github.com/rstudio/shiny/">https://github.com/rstudio/shiny/</ext-link>) and related packages (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). The BioVizSeq package allows users to install and run the tool locally, with all computations utilizing the local computer&#x2019;s resources, including memory, CPU, and storage. Users can launch BioVizSeq&#x2019;s shinyapp by executing the command &#x2018;biovizsq()&#x2019;. For added convenience, we also offer free online services and computational resources, enabling researchers to perform analysis tasks anytime and anywhere (<ext-link ext-link-type="uri" xlink:href="https://myshiny.cpolar.io/BioVizSeq/">https://myshiny.cpolar.io/BioVizSeq/</ext-link> and <ext-link ext-link-type="uri" xlink:href="https://mybase.vip.cpolar.cn/BioVizSeq/">https://mybase.vip.cpolar.cn/BioVizSeq/</ext-link>).</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Overview of the shinyapp of BioVizSeq. <bold>(A)</bold> The Home. <bold>(B)</bold> The About.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1645004-g002.tif">
<alt-text content-type="machine-generated">A screenshot of a Shiny app interface. The top section shows the &#x201c;Shinyapp Home&#x201d; tab open with a sidebar navigating different sections like &#x201c;4.2 MEME&#x201d; and a colorful bar chart displaying sequence data. The bottom section shows the &#x201c;Shinyapp About&#x201d; tab with project information for BioVizSeq, including GitHub repository links and author details for Shiqi Zhao of Zhejiang Ocean University.</alt-text>
</graphic>
</fig>
<p>The BioVizSeq Shinyapp user interface includes the following sections: Home, One Step Plot, Pre-processing, Basic Plot, Advanced Plot, and About (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). The Home and About pages in the application menu display the BioVizSeq API documentation and project information, respectively. The other four collapsible menus contain applications that are organized by function categories, with each menu containing multiple modules. The parameter operation panel is divided into two main parts: the left side is used for data upload, analysis, visualization, and parameter download. The final result is displayed in a dedicated panel, where you can choose to download graphical and tabular results. In conclusion, the rapid deployment and stable performance of the BioVizSeq Shinyapp are supported by the out-of-the-box functionalities of BioVizSeq, which work together to facilitate long-term maintenance and further development.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>User cases of BioVizSeq</title>
<sec id="s2_4_1">
<label>2.4.1</label>
<title>Basic plot</title>
<p>The core of BioVizSeq is to use the coordinate information of elements on the sequence and the length information of the sequence as input files, and use ggplot2 for graphic drawing. Based on this, we have developed a basic plot function motif_plot to facilitate users in graphic drawing (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>). In theory, motif_plot can be graphically drawn using these two files for all sequences, whether they are proteins, DNA, or RNA sequences.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>The Basic Plot function. <bold>(A)</bold> The Basic Plot operation and display interface. <bold>(B)</bold> Data preprocessing generates input data for the Basic Plot.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1645004-g003.tif">
<alt-text content-type="machine-generated">Panel A displays a data input interface for creating element plots, with options to upload element location and sequence length files, and adjust settings like element shape and size. The adjacent graph visualizes elements by gene with color coding for CDS and UTR. Panel B shows a pre-processing interface for uploading annotation and gene list files, with download options for feature and gene length files.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_4_2">
<label>2.4.2</label>
<title>Gene structure</title>
<p>GTF and GFF are popular file formats used by bioinformatics programs to represent and exchange information about various genomic features, such as gene and transcript locations and structure. There are currently multiple software or tools based on GFF3/GTF files for gene structure display, such as GSDS 2.0 (<xref ref-type="bibr" rid="B6">Hu et&#xa0;al., 2015</xref>), TBtools, etc. But there are no similar tools based on ggplot2 of R language yet. The BioVizSeq provides two modes for displaying gene structures through GFF3/GTF files. One approach is to divide the process into two steps. Firstly, the gene length information, UTR, and CDS information of the gene are obtained through gff_to_loc, and then the gene structure is freely drawn using motif_plot. Another approach is to directly use gff_plot to plot gene structure in one step, but with relatively limited freedom. (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>).</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>The one step plot function for GFF/GTF, MEME, PFAM, CDD, SMART, Plantcare.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1645004-g004.tif">
<alt-text content-type="machine-generated">Interface of the &#x201c;One Step Plot&#x201d; tool showing options to upload annotation and order files, set parameters for element shape, and adjust legend size. A gene structure plot visualizes gene elements with colored segments representing CDS and UTR across various genes. Download options for the graph are available in PNG and PDF formats.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_4_3">
<label>2.4.3</label>
<title>MEME motif</title>
<p>The MEME Suite web server (<ext-link ext-link-type="uri" xlink:href="https://meme-suite.org/meme/tools/meme">https://meme-suite.org/meme/tools/meme</ext-link>) is commonly used in bioinformatics to identify and display conserved patterns in DNA, RNA, or protein sequences. Its advantage lies in the ability to efficiently identify potential functional motifs and provide reliable significance evaluations for these motifs through statistical methods, helping researchers reveal important biological features in the sequence. However, MEME motifs also have certain drawbacks, especially in terms of display effectiveness. Firstly, the motif images generated by MEME cannot be downloaded. Secondly, the images generated by MEME tools are usually relatively single and difficult to flexibly combine or modify with other graphics or data, which limits their application in diversified graphic displays. In addition, the visual representation of motifs is relatively fixed and may not meet the expectations of aesthetic or customization needs in some specific studies. Fortunately, MEME Suite web server provides two result files: meme.xml and mast.xml. BioVizSeq first parses the result file, and then uses the motif plot function to plot the motif.</p>
</sec>
<sec id="s2_4_4">
<label>2.4.4</label>
<title>PFAM domain</title>
<p>Pfam is a widely used protein family database that provides a wealth of information on protein domains and families. It is currently hosted by InterPro (<ext-link ext-link-type="uri" xlink:href="https://www.ebi.ac.uk/interpro/">https://www.ebi.ac.uk/interpro/</ext-link>), an integrated database that integrates resources from multiple protein sequences and domains (<xref ref-type="bibr" rid="B2">Blum et&#xa0;al., 2025</xref>). The &#x201c;by sequence&#x201d; feature in Search InterPro allows users to search for domains, and functional sites by submitting sequences. However, its display results are in the form of a table, which is not conducive to direct comparative observation. BioVizSeq can organize this result file and display the domain.</p>
</sec>
<sec id="s2_4_5">
<label>2.4.5</label>
<title>CDD domain</title>
<p>NCBI&#x2019;s Batch CD Search (<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/Structure/bwrpsb/bwrpsb.cgi">https://www.ncbi.nlm.nih.gov/Structure/bwrpsb/bwrpsb.cgi</ext-link>) is a powerful tool that allows users to submit multiple protein sequences in batches and perform conservative domain searches and annotations on these sequences using the Conserved Domain Database (CDD). But its display results are also in tables. Therefore, BioVizSeq has developed the ability to showcase its results.</p>
</sec>
<sec id="s2_4_6">
<label>2.4.6</label>
<title>SMART domain</title>
<p>SMART (a Simple Modular Architecture Research Tool) (<ext-link ext-link-type="uri" xlink:href="https://smart.embl.de/">https://smart.embl.de/</ext-link>) allows the identification and annotation of genetically mobile domains and the analysis of domain architectures. It adopts a modular architecture that can quickly and accurately identify structural domains in protein sequences, and has been widely used in bioinformatics research. However, SMART usually requires a certain foundation of Linux operating system when analyzing protein sequences in batches. The results returned in bulk are text files, which are very unfavorable for displaying the results. BioVizSeq can automatically upload sequences in bulk and organize and plot the returned results, greatly reducing the operational threshold.</p>
</sec>
<sec id="s2_4_7">
<label>2.4.7</label>
<title>Plantcare <italic>cis</italic>-acting regulatory elements</title>
<p>PlantCARE is a database of plant <italic>cis</italic>-acting regulatory elements, enhancers and repressors, and is widely used to study the mechanisms of plant gene expression regulation (<ext-link ext-link-type="uri" xlink:href="https://bioinformatics.psb.ugent.be/webtools/plantcare/html/">https://bioinformatics.psb.ugent.be/webtools/plantcare/html/</ext-link>). After the submission sequence runs, the user will receive an email containing the result file. From this perspective, it is quite convenient. However, the file size limit for uploading sequences is 100kb. In addition, the information in the result file cannot be directly used to draw images and display them. Firstly, the result of each promoter subsequence is a compressed file, and secondly, the information contained in the result file is relatively large. These data need to be filtered and integrated before they can be used for image display. These are very difficult for ordinary researchers. BioVizSeq can automatically split and upload sequence files larger than 100kb, filter the resulting files directly merged by users, and then draw publication level figures.</p>
</sec>
<sec id="s2_4_8">
<label>2.4.8</label>
<title>Advance plot</title>
<p>Researchers often combine multiple types of images together when analyzing gene families, such as evolutionary tree + gene structure + MEME motif + SMART domain. Therefore, we have specifically developed a function: combi_p. The result of combi_p is a list containing multiple graphic files, and users can freely combine the graphics within it. At the same time, based on this function and patchwork package (<ext-link ext-link-type="uri" xlink:href="https://cran.r-project.org/package=patchwork">https://cran.r-project.org/package=patchwork</ext-link>), we have written a module in shinyapp specifically designed to facilitate users in combining graphics (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>).</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>The Advance Plot function. Researchers can combine multiple types of results together when analyzing gene families. <bold>(A)</bold> The phylogenetic tree. <bold>(B)</bold> GFF/GTF structure. <bold>(C)</bold> MEME result. <bold>(D)</bold> PFAM result.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1645004-g005.tif">
<alt-text content-type="machine-generated">Phylogenetic analysis and structural features of genes. Panel A shows a phylogenetic tree classifying genes into three groups: G1 (red), G2 (green), and G3 (blue). Panel B presents the DNA structure with coding sequences (CDS, pink) and untranslated regions (UTR, teal). Panel C illustrates protein motifs using various colors for each type. Panel D depicts protein domains, highlighting the AP2 domain in pink. Each panel provides detailed information about lengths in nucleotides or amino acids.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_4_9">
<label>2.4.9</label>
<title>Others</title>
<p>In addition, we have also developed some other features, such as Advance Plantcare, to display the types and quantity distribution of cis acting elements on the promoter (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>). The reason for developing this feature is that there are many types and quantities of cis acting elements on the promoter, and if we use Basic plot, it is easy to overlap and difficult to observe. For genes, we often conduct statistical analysis on their coordinates, sequence length, number of exons/introns, and analyze a series of physicochemical properties of the encoded protein sequence. Therefore, we have also added these two functions in Others, where users can analyze and download the results.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Combination diagram of types and numbers of <italic>cis</italic> regulatory elements on gene promoters. <bold>(A)</bold> The phylogenetic tree. <bold>(B)</bold> The diferent intensity colors and numbers of the grid indicate the numbers of different promoter elements in genes. <bold>(C)</bold> The different colored histogram represents the sum or percent of the <italic>cis</italic>-acting elements in each category.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1645004-g006.tif">
<alt-text content-type="machine-generated">Clustered heatmap and data matrix representing gene expression related to growth, development, light responsiveness, phytohormone responsiveness, and stress responsiveness. Panel A shows a dendrogram clustering 50 genes into groups G1, G2, and G3. Panel B is a grid with numerical values indicating motif occurrences across genes. Panel C displays a stacked bar chart quantifying motif group contributions, color-coded for different responses.</alt-text>
</graphic>
</fig>
</sec>
</sec>
</sec>
<sec id="s3" sec-type="conclusion">
<label>3</label>
<title>Conclusion</title>
<p>The BioVizSeq package provides a range of functions for analyzing and displaying components on biological sequences, such as gene structure, motif, structural domains and <italic>cis</italic> regulatory elements. At the same time, a local shinyApp with a user-friendly interface and online analysis services are provided, which greatly facilitates researchers without the need for coding capabilities.</p>
</sec>
<sec id="s4">
<label>4</label>
<title>Methods</title>
<p>The BioVizSeq package is developed using the R v4.3.1 (<xref ref-type="bibr" rid="B13">Team R. C, 2025</xref>) and utilizes the roxygen2 v7.2.3 package for generating and updating API documentation. The correctness of functions and the integrity of the package are verified using the devtools v2.4.5 package. The compilation is performed based on the DESCRIPTION and NAMESPACE files. The API documentation and website are built using pkgdown v2.0.7 with the _pkgdown.yml configuration file, which provides online help documentation (<ext-link ext-link-type="uri" xlink:href="https://zhaosq2022.github.io/BioVizSeq/">https://zhaosq2022.github.io/BioVizSeq/</ext-link>). For data manipulation, the seqinr v4.2&#x2013;36 and stringr v1.5.0 are used to process fasta files and strings, respectively. The dplyr v1.1.4, and tidyr v1.3.0 packages from the tidyverse ecosystem are used to transform data structures. BioVizSeq prioritizes the use of ggplot2 v3.5.0, a widely used package that offers customizable visualization capabilities. In the end, we developed a series of functions for data analysis and graphical presentation (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>).</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Major functions of BioVizSeq.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Function</th>
<th valign="middle" align="left">Description</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Biovizseq</td>
<td valign="middle" align="left">BioVizSeq shiny app start function</td>
</tr>
<tr>
<td valign="middle" align="left">geom_rrect</td>
<td valign="middle" align="left">A geom layer that can generate rounded rectangles</td>
</tr>
<tr>
<td valign="middle" align="left">motif_plot</td>
<td valign="middle" align="left">Basic plot to visualize the elements within bio-sequences</td>
</tr>
<tr>
<td valign="middle" align="left">gff_to_loc</td>
<td valign="middle" align="left">Extract the information of element from gff/gtf file</td>
</tr>
<tr>
<td valign="middle" align="left">gff_plot</td>
<td valign="middle" align="left">Visualization of element in gff/gtf file</td>
</tr>
<tr>
<td valign="middle" align="left">meme_to_loc</td>
<td valign="middle" align="left">Extract the information of elements from meme/mast file</td>
</tr>
<tr>
<td valign="middle" align="left">meme_plot</td>
<td valign="middle" align="left">Visualization of element in meme/mast file</td>
</tr>
<tr>
<td valign="middle" align="left">meme_seq</td>
<td valign="middle" align="left">Extract the motif sequence from meme/mast file</td>
</tr>
<tr>
<td valign="middle" align="left">pfam_to_loc</td>
<td valign="middle" align="left">Extract the information of domain from Pfam file</td>
</tr>
<tr>
<td valign="middle" align="left">pfam_plot</td>
<td valign="middle" align="left">Visualization of element in Pfam file</td>
</tr>
<tr>
<td valign="middle" align="left">cdd_to_loc</td>
<td valign="middle" align="left">Extract the information of domain from CDD file</td>
</tr>
<tr>
<td valign="middle" align="left">cdd_plot</td>
<td valign="middle" align="left">Visualization of element in CDD file</td>
</tr>
<tr>
<td valign="middle" align="left">smart_to_loc</td>
<td valign="middle" align="left">Extract the information of domain from SMART file</td>
</tr>
<tr>
<td valign="middle" align="left">smart_plot</td>
<td valign="middle" align="left">Visualization of element in SMART file</td>
</tr>
<tr>
<td valign="middle" align="left">upload_fa_to_plantcare</td>
<td valign="middle" align="left">Upload the fasta file to Plantcare and return the result</td>
</tr>
<tr>
<td valign="middle" align="left">plantcare_classify</td>
<td valign="middle" align="left">Classify the elements in the plantcare results</td>
</tr>
<tr>
<td valign="middle" align="left">plantcare_to_loc</td>
<td valign="middle" align="left">Extract the information of element from Plantcare file</td>
</tr>
<tr>
<td valign="middle" align="left">plantcare_plot</td>
<td valign="middle" align="left">Visualization of element in Plantcare file</td>
</tr>
<tr>
<td valign="middle" align="left">combi_p</td>
<td valign="middle" align="left">Get ggplot2 files to facilitate free combination</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material. Further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>SZ: Methodology, Conceptualization, Software, Visualization, Data curation, Writing &#x2013; original draft, Funding acquisition. RZ: Visualization, Writing &#x2013; original draft, Validation. HM: Writing &#x2013; review &amp; editing, Formal Analysis, Visualization. BS: Supervision, Writing &#x2013; review &amp; editing, Funding acquisition, Resources. YZ: Project administration, Writing &#x2013; review &amp; editing, Supervision.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research and/or publication of this article. This research was supported by Zhejiang Provincial Natural Science Foundation of China (ZCLMS25D0601), the General Projects of Zhejiang Provincial Department of Education (Y202353943), Zhejiang Provincial Discipline Construction Project - Degree Point Construction (Aquaculture)(110340602211).</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bailey</surname> <given-names>T. L.</given-names>
</name>
<name>
<surname>Johnson</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Grant</surname> <given-names>C. E.</given-names>
</name>
<name>
<surname>Noble</surname> <given-names>W. S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>The MEME suite</article-title>. <source>Nucleic Acids Res.</source> <volume>43</volume>, <fpage>W39</fpage>&#x2013;<lpage>W49</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkv416</pub-id>, PMID: <pub-id pub-id-type="pmid">25953851</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Blum</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Andreeva</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Florentino</surname> <given-names>L. C.</given-names>
</name>
<name>
<surname>Chuguransky</surname> <given-names>S. R.</given-names>
</name>
<name>
<surname>Grego</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Hobbs</surname> <given-names>E.</given-names>
</name>
<etal/>
</person-group>. (<year>2025</year>). <article-title>InterPro: the protein sequence classification resource in 2025</article-title>. <source>Nucleic Acids Res.</source> <volume>53</volume>, <fpage>D444</fpage>&#x2013;<lpage>D456</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkae1082</pub-id>, PMID: <pub-id pub-id-type="pmid">39565202</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>C. J.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Thomas</surname> <given-names>H. R.</given-names>
</name>
<name>
<surname>Frank</surname> <given-names>M. H.</given-names>
</name>
<name>
<surname>He</surname> <given-names>Y. H.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>TBtools: an integrative toolkit developed for interactive analyses of big biological data</article-title>. <source>Mol. Plant</source> <volume>13</volume>, <fpage>1194</fpage>&#x2013;<lpage>1202</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.molp.2020.06.009</pub-id>, PMID: <pub-id pub-id-type="pmid">32585190</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Shang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ge</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>CFVisual: an interactive desktop platform for drawing gene structure and protein architecture</article-title>. <source>BMC Bioinf.</source> <volume>23</volume>, <fpage>178</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12859-022-04707-w</pub-id>, PMID: <pub-id pub-id-type="pmid">35562653</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>C. J.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J. W.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zeng</surname> <given-names>Z. H.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>TBtools-II: A &#x201c;one for all, all for one&#x201d;bioinformatics platform for biological big-data mining</article-title>. <source>Mol. Plant</source> <volume>16</volume>, <fpage>1733</fpage>&#x2013;<lpage>1742</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.molp.2023.09.010</pub-id>, PMID: <pub-id pub-id-type="pmid">37740491</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>A. Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>GSDS 2.0: an upgraded gene feature visualization server</article-title>. <source>Bioinformatics</source> <volume>31</volume>, <fpage>1296</fpage>&#x2013;<lpage>1297</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btu817</pub-id>, PMID: <pub-id pub-id-type="pmid">25504850</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Klaus</surname>
</name>
<name>
<surname>Galensa</surname>
</name>
</person-group> (<year>2017</year>). <article-title>ggplot2: elegant graphics for data analysis (2nd ed.)</article-title>. <source>Computing Rev</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-319-24277-4</pub-id>
</citation></ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lescot</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Dehais</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Thijs</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Marchal</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Moreau</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Van de Peer</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2002</year>). <article-title>PlantCARE, a database of plant cis-acting regulatory elements and a portal to tools for in silico analysis of promoter sequences</article-title>. <source>Nucleic Acids Res.</source> <volume>30</volume>, <fpage>325</fpage>&#x2013;<lpage>327</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/30.1.325</pub-id>, PMID: <pub-id pub-id-type="pmid">11752327</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Letunic</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Khedkar</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Bork</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>SMART: recent updates, new developments and status in 2020</article-title>. <source>Nucleic Acids Res.</source> <volume>49</volume>, <fpage>D458</fpage>&#x2013;<lpage>D460</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkaa937</pub-id>, PMID: <pub-id pub-id-type="pmid">33104802</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mistry</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Chuguransky</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Williams</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Qureshi</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Salazar</surname> <given-names>G. A.</given-names>
</name>
<name>
<surname>Sonnhammer</surname> <given-names>E. L. L.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Pfam: The protein families database in 2021</article-title>. <source>Nucleic Acids Res.</source> <volume>49</volume>, <fpage>D412</fpage>&#x2013;<lpage>D419</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkaa913</pub-id>, PMID: <pub-id pub-id-type="pmid">33125078</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pertea</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Pertea</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>GFF utilities: gffRead and gffcompare</article-title>. <source>F1000Res</source> <volume>9</volume>, <fpage>304</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.12688/f1000research.23297.2</pub-id>, PMID: <pub-id pub-id-type="pmid">32489650</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Savojardo</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Martelli</surname> <given-names>P. L.</given-names>
</name>
<name>
<surname>Casadio</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Finding functional motifs in protein sequences with deep learning and natural language models</article-title>. <source>Curr. Opin. Struct. Biol.</source> <volume>81</volume>, <elocation-id>102641</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.sbi.2023.102641</pub-id>, PMID: <pub-id pub-id-type="pmid">37385080</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="book">
<person-group person-group-type="author">
<collab>Team R. C</collab>
</person-group>. (<year>2025</year>). <source>R: A Language and Environment for Statistical Computing</source> (<publisher-name>R Foundation for Statistical Computing</publisher-name>). Available online at: <uri xlink:href="https://www.r-project.org/">https://www.r-project.org/</uri>.</citation></ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Chitsaz</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Derbyshire</surname> <given-names>M. K.</given-names>
</name>
<name>
<surname>Gonzales</surname> <given-names>N. R.</given-names>
</name>
<name>
<surname>Gwadz</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>The conserved domain database in 2023</article-title>. <source>Nucleic Acids Res.</source> <volume>51</volume>, <fpage>D384</fpage>&#x2013;<lpage>D388</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkac1096</pub-id>, PMID: <pub-id pub-id-type="pmid">36477806</pub-id></citation></ref>
</ref-list>
</back>
</article>