<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2024.1367632</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Using ddRADseq to assess the genetic diversity of in-farm and gene bank cacao resources in the Baracoa region, eastern Cuba, for use and conservation purposes</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Ramirez-Ramirez</surname><given-names>Angel Rafael</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>*</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2625022"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mirzaei</surname><given-names>Khaled</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Men&#xe9;ndez-Grenot</surname><given-names>Miguel</given-names>
</name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Clap&#xe9;-Borges</surname><given-names>Pablo</given-names>
</name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Espinosa-Lop&#xe9;z</surname><given-names>Georgina</given-names>
</name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1840532"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bidot-Mart&#xed;nez</surname><given-names>Igor</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bertin</surname><given-names>Pierre</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Faculty of Agroforestry, University of Guant&#xe1;namo</institution>, <addr-line>Guant&#xe1;namo</addr-line>, <country>Cuba</country></aff>
<aff id="aff2"><sup>2</sup><institution>Earth and Life Institute, Universit&#xe9; catholique de Louvain (UCLouvain)</institution>, <addr-line>Louvain-la-neuve</addr-line>, <country>Belgium</country></aff>
<aff id="aff3"><sup>3</sup><institution>Unidad de Ciencia y T&#xe9;cnica de Base-Baracoa / Instituto de Investigaciones Agroforestales (UCTBBaracoa / INAF)</institution>, <addr-line>Baracoa</addr-line>, <country>Cuba</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Biochemistry, Faculty of Biology, University of Havana</institution>, <addr-line>La Habana</addr-line>, <country>Cuba</country></aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: David C. Haak, Virginia Tech, United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Haidong Yan, University of Georgia, United States</p>
<p>Alexander Sandercock, Cornell University, United States</p>
<p>James Friel, University of Milan, Italy</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Angel Rafael Ramirez-Ramirez, <email xlink:href="mailto:angelosopupi@gmail.com">angelosopupi@gmail.com</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>05</day>
<month>03</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1367632</elocation-id>
<history>
<date date-type="received">
<day>09</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>12</day>
<month>02</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Ramirez-Ramirez, Mirzaei, Men&#xe9;ndez-Grenot, Clap&#xe9;-Borges, Espinosa-Lop&#xe9;z, Bidot-Mart&#xed;nez and Bertin</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Ramirez-Ramirez, Mirzaei, Men&#xe9;ndez-Grenot, Clap&#xe9;-Borges, Espinosa-Lop&#xe9;z, Bidot-Mart&#xed;nez and Bertin</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The Baracoa region, eastern Cuba, hosts around 80 % of the country cacao (<italic>Theobroma cacao</italic> L.) plantations. Cacao plants in farms are diverse in origin and propagation, with grafted and hybrid plants being the more common ones. Less frequent are plants from cuttings, TSH progeny, and traditional Cuban cacao. A national cacao gene bank is also present in Baracoa, with 282 accessions either prospected in Cuba or introduced from other countries. A breeding program associated with the gene bank started in the 1990s based on agro-morphological descriptors. The genetic diversity of cacao resources in Baracoa has been poorly described, except for traditional Cuban cacao, affecting the proper development of the breeding program and the cacao planting policies in the region. To assess the population structure and genetic diversity of cacao resources in Baracoa region, we genotyped plants from both cacao gene bank (CG) and cacao farms (CF) applying a new ddRADseq protocol for cacao. After data processing, two SNPs datasets containing 11,425 and 6,481 high-quality SNPs were generated with 238 CG and 135 CF plants, respectively. SNPs were unevenly distributed along the 10 cacao chromosomes and laid mainly in noncoding regions of the genome. Population structure analysis with these SNP datasets identified seven and four genetic groups in CG and CF samples, respectively. Clustering using UPGMA and principal component analysis mostly agree with population structure results. Amelonado was the predominant cacao ancestry, accounting for 49.22 % (CG) and 57.73 % (CF) of the total. Criollo, Contamana, Iquitos, and Nanay ancestries were detected in both CG and CF samples, while Nacional and Mara&#xf1;on backgrounds were only identified in CG. Genetic differentiation among CG (<italic>F<sub>ST</sub>
</italic> ranging from 0.071 to 0.407) was higher than among CF genetic groups (<italic>F<sub>ST</sub>
</italic>: 0.093&#x2013;0.282). Genetic diversity parameters showed similar values for CG and CF samples. The CG and CF genetic groups with the lowest genetic diversity parameters had the highest proportion of Amelonado ancestry. These results should contribute to reinforcing the ongoing breeding program and updating the planting policies on cacao farms, with an impact on the social and economic life of the region.</p>
</abstract>
<kwd-group>
<kwd>ddRADSeq</kwd>
<kwd>snps</kwd>
<kwd>theobroma cacao</kwd>
<kwd>genetic diversity</kwd>
<kwd>Cuban cacao resources</kwd>
<kwd>gene bank</kwd>
<kwd>cacao farms</kwd>
</kwd-group>
<counts>
<fig-count count="7"/>
<table-count count="6"/>
<equation-count count="0"/>
<ref-count count="96"/>
<page-count count="20"/>
<word-count count="12838"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Functional and Applied Plant Genomics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<title>Introduction</title>
<p>The cacao species <italic>Theobroma cacao</italic> L is the center of the chocolate industry, involving millions of people around the world, from small farmers in remote areas of developing countries to chocolate shops in big cities of the industrial world (<xref ref-type="bibr" rid="B14">CacaoNet, 2022</xref>). The origin of <italic>T. cacao</italic> has been set in Upper Amazon, South America, in the current borders between Brazil, Colombia, Ecuador, and Per&#xfa;, from where it would have been extended first to Mesoamerica during pre-Columbian times and later to other tropical and subtropical regions of Latin America, Africa, and Asia (<xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al., 2008</xref>; <xref ref-type="bibr" rid="B96">Zhang and Motilal, 2016</xref>). Cacao bean global production was estimated at 5.76 million tons in 2020, with C&#xf4;te d&#x2019;Ivoire, Ghana, and Indonesia as the biggest producing countries (<xref ref-type="bibr" rid="B29">FAOSTAT, 2023</xref>).</p>
<p>In the past decade, cacao-producing areas have increased worldwide from 9.6 to 12.3 million hectares (ha) using cacao types with low cocoa quality in most cases. Despite this expansion, yield growth was minimal, with an increase from 450 kg/ha to 467 kg/ha recorded (<xref ref-type="bibr" rid="B29">FAOSTAT, 2023</xref>). Cacao genetic studies have a role to play in improving the yield and cocoa quality, especially when there is evidence of a narrow genetic base used in cacao farming (<xref ref-type="bibr" rid="B96">Zhang and Motilal, 2016</xref>; <xref ref-type="bibr" rid="B20">Cornejo et&#xa0;al., 2018</xref>). A very wide range of molecular markers has been used in cacao genetic studies (<xref ref-type="bibr" rid="B69">Motilal et&#xa0;al., 2017</xref>), which, combined with morphological data, have served for genetic diversity assessment, clone classification, QTL identification, association studies, linkage map, etc. However, more efforts are required to face future challenges of the cacao and chocolate industries, which comprise increasing demand for chocolate, including those with superior qualities, the spreading of diseases and pests, changing environmental conditions, and the need for sustainable production (<xref ref-type="bibr" rid="B96">Zhang and Motilal, 2016</xref>; <xref ref-type="bibr" rid="B92">Wickramasuriya and Dunwell, 2018</xref>).</p>
<p>Three traditional cacao groups have first been recognized based on plant morphological profiles: Criollo, Forastero, and Trinitario. The productivity of Criollo is low but of high quality, while Forastero is highly productive but of lower quality. Trinitario appeared as the result of the crossing of both former groups, carrying intermediate characteristics (<xref ref-type="bibr" rid="B30">Figueira et&#xa0;al., 1994</xref>; <xref ref-type="bibr" rid="B9">Badrie et&#xa0;al., 2015</xref>). More recently, a new classification system was established based on Simple Sequence Repeat (SSR) markers, and 10 cacao ancestry genetic groups have been recognized: Amelonado, Contamana, Criollo, Curaray, Guiana, Iquitos, Mara&#xf1;&#xf3;n, Nanay, Nacional, and Pur&#xfa;s (<xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al., 2008</xref>). Both classification systems are currently in use, and cacao genetic studies are challenging when it comes to the proper classification of cacao clones, especially those plants resulting from the crossing of already admixed parents (<xref ref-type="bibr" rid="B68">Motilal, 2018</xref>; <xref ref-type="bibr" rid="B23">dos Santos Menezes et&#xa0;al., 2022</xref>).</p>
<p>The use of single nucleotide polymorphism (SNP) markers to assess genetic diversity in <italic>T. cacao</italic> has sharply increased in the last few years. Most of the studies are based on SNP datasets derived from the study of <xref ref-type="bibr" rid="B7">Argout et&#xa0;al. (2008)</xref>. These SNPs have been successfully used to describe cacao ancestry genetic group classification and genetic diversity evaluation (<xref ref-type="bibr" rid="B38">Ji et&#xa0;al., 2013</xref>; <xref ref-type="bibr" rid="B28">Fang et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B21">Cosme et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B51">Lindo et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B6">Arevalo-Gardini et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B90">Wang et&#xa0;al., 2020</xref> and <xref ref-type="bibr" rid="B50">Li et&#xa0;al., 2021</xref>), but difficulties have been reported in the proper separation of some cacao ancestry genetic groups (<xref ref-type="bibr" rid="B54">Lukman et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B86">Takrama et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B74">Osorio-Guarin et&#xa0;al., 2017</xref>), driving the search for other SNP sets suitable for cacao classification (<xref ref-type="bibr" rid="B55">Mahabir et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B37">Guti&#xe9;rrez et&#xa0;al., 2021</xref>).</p>
<p>Few studies describe the use of large SNP datasets derived from next-generation sequencing (NGS) technologies in cacao genetic diversity evaluation. <xref ref-type="bibr" rid="B20">Cornejo et&#xa0;al. (2018)</xref> sequenced 200 cacao genomes to explore cacao domestication history, and <xref ref-type="bibr" rid="B76">Osorio-Guarin et&#xa0;al. (2018)</xref> reported a modified GBS approach with 30 samples and the same references as <xref ref-type="bibr" rid="B20">Cornejo et&#xa0;al. (2018)</xref> for the identification of 7,009 SNPs carrying cacao ancestry information. Recently, GBS experiments based on genomic digestion with two enzymes were used to perform genetic studies in cacao from French Guiana, Martinique, and Colombia (<xref ref-type="bibr" rid="B46">Lachenaud et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B2">Adenet et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B75">Osorio-Guarin et&#xa0;al., 2020</xref>).</p>
<p>Double-digest Restriction-Associated DNA sequencing (ddRADseq) is a RADseq-derived technique that uses NGS technology to uncover hundreds or thousands of polymorphic genetic markers across the genome. This is a reduced-representation genome sequencing method that also combines two restriction enzymes to digest DNA and has become a frequently used approach for SNP marker discovery and genotyping of nonmodel organisms (<xref ref-type="bibr" rid="B77">Peterson et&#xa0;al., 2012</xref>; <xref ref-type="bibr" rid="B5">Andrews et&#xa0;al., 2016</xref>). Several ddRADseq protocols have been exploited for crop genetics studies, differing mainly in the enzyme combination and size of the selected DNA fragments. ddRADseq outperforms other GBS protocols used for cacao genetic analysis in higher average coverage and lower missing data, though it is acknowledged that the success of ddRADseq protocols requires high-quality DNA preparations (<xref ref-type="bibr" rid="B83">Scheben et&#xa0;al., 2017</xref>).</p>
<p>Cuba is a small cacao producer with 1,577 tons of cacao beans obtained in 2020 (<xref ref-type="bibr" rid="B73">ONEI, Oficina Nacional de Estad&#xed;stica e Informaci&#xf3;n, 2021</xref>; <xref ref-type="bibr" rid="B29">FAOSTAT, 2023</xref>). There is a debate about the date and place of cacao introduction in Cuba, pointed either to the central region in <italic>Mi Cuba</italic> near Cabaigu&#xe1;n in 1540 from M&#xe9;xico or by French colonists running from the Haitian Revolution in late eighteenth century who settled down in the region of <italic>Ti Arriba</italic> in eastern Cuba (<xref ref-type="bibr" rid="B71">Nu&#xf1;ez-Gonz&#xe1;lez, 2010</xref>). During the nineteenth century, cacao plantations and plants for self-consumption could be found in several regions of the country, including locations near Havana in the western part of the country. Reports attest to the exportation of 1,500 tons of cacao in the early 90s of the nineteenth century and 2,000 tons of production at the beginning of the 1900s, after a devastating war period during 1895&#x2013;1898. Little information about the origin of the plants is available, though the harvesting of good quality cacao beans of the types &#x201c;Criollo&#x201d; or &#x201c;Cubano&#x201d;, &#x201c;Guayaquil&#x201d; (Ecuador), and &#x201c;Caracas&#x201d; (Venezuela) is recognized.</p>
<p>The expansion of the sugar industry during the late nineteenth and early twentieth centuries pushed cacao planting areas to places not suitable for sugar cane cropping. The mountainous regions in the eastern and central part of the country were the most appropriate, including the current provinces of Sancti Spiritus, Cienfuegos, and Villa Clara in the center and Guant&#xe1;namo, Santiago de Cuba, and Granma in the east. This distribution has remained almost the same until the present, where Baracoa municipality, in Guant&#xe1;namo province, excels in the favorable climate conditions for cacao farming. Baracoa hosts around 80% of Cuban cacao plantations, and more than 20% of the cultivated land in the region is used for cacao cropping, which was responsible for 74.6% of Cuban cacao production in 2020. In this region, the cacao and chocolate agroindustry is part of a strong tradition lasting decades with a great impact on both the social and economic lives of the local inhabitants (<xref ref-type="bibr" rid="B71">Nu&#xf1;ez-Gonz&#xe1;lez, 2010</xref>; <xref ref-type="bibr" rid="B73">ONEI, Oficina Nacional de Estad&#xed;stica e Informaci&#xf3;n, 2021</xref>).</p>
<p>In Cuba, cacao is cultivated organically, and farms use an agroforestry, multispecies, and multilayer cultivation system with shade trees and various associated perennial and annual crops. Plantations contain cacao plants of diverse origins and reproduction modes. The most abundant are grafted plants obtained from specific clones, mainly from the United Fruit Company (UF) introduced in the country around 1955. Another type is hybrid plants grown from seeds, which could be either certified seeds produced by hand pollination of certain cacao clones at Unidad de Ciencia y T&#xe9;cnica de Base-Baracoa, Instituto de Investigaciones Agroforestales (UCTB-Baracoa/INAF) or seeds produced on farms by farmer-selected plants under open pollination conditions. Other less common sources of cacao plants include cuttings from highly productive plants or cacao clones and progeny from Trinidad Selected Hybrids (TSH) imported as seed during the 1970s (<xref ref-type="bibr" rid="B56">M&#xe1;rquez-Rivero and Aguirre-G&#xf3;mez, 2008</xref>; <xref ref-type="bibr" rid="B71">Nu&#xf1;ez-Gonz&#xe1;lez, 2010</xref>; <xref ref-type="bibr" rid="B58">Mart&#xed;nez-Su&#xe1;rez et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B62">Men&#xe9;ndez-Grenot et&#xa0;al., 2016</xref>).</p>
<p>
<xref ref-type="bibr" rid="B10">Bidot Mart&#xed;nez et&#xa0;al. (2015)</xref> analyzed another class of plants found in Cuban cacao farms, known as traditional Cuban cacao, which represents around 6% of the Cuban cacao. These are very old plants remaining in cacao plantations whose propagation has relied exclusively on farmers and are supposed to be the closest ones to the cacao primarily introduced in Cuba. These cacao plants from central and eastern Cuba were sampled, and the population structure and genetic diversity were analyzed with SSR markers and morphological descriptors. Two groups of plants were identified, mostly corresponding with the geographical regions of collection: central and eastern. The cacao ancestry of these plants was mainly divided into Amelonado, Criollo, Mara&#xf1;on, and Contamana. Morphological profiles revealed the Trinitario type as the most abundant among the studied plants. The persistence of these plants proves their ability to resist local environmental conditions, including diseases and pests, and in some cases, they contain seeds with white cotyledons (<xref ref-type="bibr" rid="B11">Bidot Mart&#xed;nez et&#xa0;al., 2017</xref>). The real productive potential of these plants is yet to be studied; however, some of them, mainly with white cotyledons, were selected for conservation purposes. A deeper knowledge of the genetic diversity of commercial cacao farms in Baracoa is crucial to facing the current and future challenges of the local cacao agroindustry and requires more in-depth studies.</p>
<p>A national cacao gene bank (CG) started to grow in the 1980s under the supervision of UCTB-Baracoa/INAF for conservation and research purposes. Currently, the gene bank hosts 282 accessions: 194 prospected in Cuba in the provinces of Guant&#xe1;namo (163), Santiago de Cuba (24), and Mayabeque (seven), and 88 introduced from different geographical regions (South America (41), Central America (18), Caribbean (17), North America (11), and Africa (one)). The 194 prospected accessions were plants collected in field expeditions, including the traditional Cuban cacao plants aforementioned, and hybrid plants selected from breeding experiments using hand pollination between clones of interest. The 88 introduced accessions comprise cacao clones of the series UF, Pound, SCA, EET, ICS, TSH, GS, SIAL, and SIC&#x2014;among others&#x2014;imported from other countries throughout the second half of the last century.</p>
<p>Several of these accessions have been partially characterized with morphological and agronomical descriptors, including resistance to <italic>Phytophthora palmivora</italic> and commercial quality, as part of a breeding and selection program launched by UCTB-Baracoa/INAF during the 1990s (<xref ref-type="bibr" rid="B61">Men&#xe9;ndez-Grenot et&#xa0;al., 2012</xref>, <xref ref-type="bibr" rid="B60">2014</xref>; <xref ref-type="bibr" rid="B58">Mart&#xed;nez-Su&#xe1;rez et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B59">Matos-Cueto et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B62">Men&#xe9;ndez-Grenot et&#xa0;al., 2016</xref>). This program has allowed the identification of clones with high productive potential and the establishment of procedures for certified hybrid seed production from selected clones. Unfortunately, genetic characterization of the gene bank is pending, and the program goal for higher quality cacao in terms of cocoa and chocolate quality remains evasive. Reversing such a scenario is mandatory in the efforts to get access to more demanding cacao markets.</p>
<p>The genetic characterization of plants of the Cuban National cacao gene bank (CG) and cacao farms (CF) from Baracoa region, including the identification of ancestry genetic groups, represents a significant step forward for the future development of the cacao and chocolate industries in the region. To this end, the goals of our study were (1) to apply a new ddRADseq protocol for cacao SNP identification and use these SNPs and (2) use these SNPs to assess the population structure and genetic diversity of conserved and in-use cacao resources in the Baracoa region. The results derived from this research provide the much-needed information about the genetic diversity of cacao resources in Baracoa. This information could allow a better planning of breeding experiments as part of the ongoing breeding and selection program and the revalorization of previously obtained cacao clones, especially when association studies using phenotypic data already collected will be possible. Additionally, an evaluation of the cacao ancestry genetic groups to be introduced in both the national gene bank and cacao farms will be possible, resulting in an improvement in the genetic diversity of conserved and in-used cacao resources.</p>
</sec>
<sec id="s2" sec-type="material|methods">
<title>Material and methods</title>
<sec id="s2_1">
<title>Plant material</title>
<p>Mature leaves were collected from clone accessions of <italic>Theobroma cacao</italic> of the CG. In order to select the samples from CF, a survey was applied to cover the diversity of cacao farms currently in production (commercial cacao farms) present in Baracoa with the help and experience of cacao specialists from the UCTB-Baracoa/INAF. The farms were located in the three major productive poles: Jamal, San Luis, and Paso de Cuba/Sabanilla, and the survey covered production (yield of cacao and other side products), soil properties (fertility, humidity, drainage, erosion), topography, slope orientation, canopy diversity, and more importantly, cacao plant origin according to farmers (grafted, hybrid, traditional).</p>
<p>Four farm types were identified: type 1 farms consisted of flat, wet valleys with high humidity and variable soil drainage, with mostly hybrid and grafted cacao plants; type 2 contained mostly flat and wet valleys with favorable soil drainage but only grafted cacao plants; type 3 contained mountainside farms with favorable drainage and eroded soil with hybrid and traditional cacao plants; and type 4 was a mixture of flat and mountainside topography with favorable soil drainage and combined the three cacao plants&#x2019; origins: traditional, grafted, and hybrids.</p>
<p>Seven cacao farms, comprising all farm types identified, were selected as representative of the surveyed ones (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table S1</bold></xref>). On these seven farms, 32 plots with 25 plants each were randomly raised and established as sampling units. All combinations of cacao plant origins and topographic conditions detected on each farm were covered by the plots. The total number of plots included replica plots that were done whenever the farm size allowed it. Plants were numbered from 1 to 800, and 160 of them were randomly taken (five per plot) with a random number generation procedure. Mature leaves of selected plants were collected for analysis.</p>
<p>Leaves of the sampled plants were placed in a closed container with an air dehumidifier to ensure fast drying; the temperature was regularly monitored and always maintained below 40&#xb0;C. Once dried, the leaves were kept at &#x2212;70&#xb0;C until use.</p>
<p>Cacao plants (65) belonging to the 10 cacao ancestry genetic groups, according to <xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al. (2008)</xref> were taken as reference plants. <xref ref-type="bibr" rid="B20">Cornejo et&#xa0;al. (2018)</xref> determined the cacao ancestry of these plants using whole genome sequencing experiments and classified them as Amelonado (10), Contamana (seven), Criollo (four), Curaray (five), Guiana (seven), Iquitos (six), Mara&#xf1;on (10), Nacional (four), Nanay (eight), and Pur&#xfa;s (four) (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table S2</bold></xref>).</p>
</sec>
<sec id="s2_2">
<title>DNA extraction and purification</title>
<p>DNA was purified following a previously described protocol with some modifications (<xref ref-type="bibr" rid="B85">Souza et&#xa0;al., 2012</xref>). A sample of 25 mg of dry cacao leaves was frozen with liquid nitrogen and reduced to a fine powder using a Tissue Lyser (QIAGEN, Germany). The powder was washed twice with 1.5 mL of cold sorbitol buffer (0.35 M sorbitol, 100 mM Tris-HCl, 5 mM EDTA, 1% PVP-40, 1% 2-mercaptoethanol, pH = 8.0) and centrifuged for 10 min at 4,500&#xd7;<italic>g</italic> and 4&#xb0;C. The pellet obtained was resuspended in 1 mL of prewarmed (65&#xb0;C) extraction buffer (3% CTAB, 100 mM Tris-HCl, 20 mM EDTA, 3 M NaCl, pH = 8.0); additionally, 20 &#xb5;L of proteinase K (10 mg/mL), 35 &#xb5;L of 30% sarkosyl, and 30 mg of PVPP were added to each tube. The homogenate was incubated at 65&#xb0;C for 1 h and mixed by inversion every 15 min. After cooling at room temperature, 800 &#xb5;L of chloroform:isoamyl alcohol (24:1) was added and mixed by inversion for 15 min, followed by a centrifugation step at 13,000&#xd7;<italic>g</italic> for 10 min at room temperature. The upper phase was transferred to a fresh clean tube and volumes equal to 0.1 times of 3 M of NaAc at pH 5.2 and 2/3 times of cold isopropanol (&#x2212;20&#xb0;C) were added to the homogenates. Tubes were mixed by inversion and kept at &#x2212;20&#xb0;C overnight. The DNA pellet was collected by centrifugation at 15,500&#xd7;<italic>g</italic> for 30 min at 4&#xb0;C, washed by the addition of 700 &#xb5;L of 70% ethanol, and centrifuged again for 5 min, as previously. The supernatants were carefully removed to avoid losing the nucleic acid pellet, and the tubes were left open to dry at room temperature. Pellets were resuspended in 100 &#x3bc;L of TE buffer with 2 &#xb5;L of DNase-free RNase A (10 mg/mL) by incubation at 37&#xb0;C until complete dissolution and avoiding pipetting. DNA preparations were stored at &#x2212;20&#xb0;C until required.</p>
<p>A DNA cleaning step was implemented using the silica columns provided with the DNeasy Plant Pro Kit from <xref ref-type="bibr" rid="B78">QIAgen (2019)</xref> and following the manufacturer&#x2019;s instructions with some modifications. Shortly, 200 &#xb5;L of Milli-Q distilled water was added to 100 &#xb5;L of the purified DNA solution. The mix was placed in a water bath at 37&#xb0;C until a homogeneous solution was achieved; next, 375 &#xb5;L of the APP buffer was added, and the homogenate was applied to the silica columns. Column washing steps were completed according to the manufacturer. DNA was eluted in 65 &#xb5;L of TE buffer and kept at &#x2212;20&#xb0;C until use.</p>
<p>DNA integrity was checked by agarose gel electrophoresis at 1.5%. DNA yield was estimated with a fluorimeter and fluorescent DNA-binding dye (Qubit&#x2122; dsDNA BR Assay Kit, Thermo Fisher Scientific, USA), according to the manufacturer&#x2019;s instructions. A total of 406 samples were successfully purified: 264 from CG and 142 from CF, which were used for library preparation.</p>
</sec>
<sec id="s2_3">
<title>ddRADseq library preparation and sequencing</title>
<p>Reagents used in library preparation were obtained from New England Biolabs (NEB), USA unless specified. ddRADseq libraries were prepared as described (<xref ref-type="bibr" rid="B77">Peterson et&#xa0;al., 2012</xref>) with some modifications. Briefly, 1,000 ng of DNA were digested with 10 U of EcoRI HF and 5 U of NlaIII using the Cut Smart Buffer in a final volume of 30 &#xb5;L. The digestion reactions were left to occur at 37&#xb0;C overnight. A volume of 15 &#xb5;L of digested DNA was put to ligation with adaptors designed for ddRADseq sequencing libraries (<xref ref-type="bibr" rid="B77">Peterson et&#xa0;al., 2012</xref>) using T4 ligase (0.1 U per reaction). Ligation reactions occurred for 8 h at 16&#xb0;C in a final volume of 20 &#xb5;L.</p>
<p>After ligation, samples were combined to form pools containing between 44 and 48 samples (sublibraries). The sublibraries were purified with magnetic beads (Promega, USA) and fragments between 300-500 bp were selected using a BluePippin instrument (Sage Science Inc., USA). Size-selected sublibraries were polymerase chain reaction (PCR)-enriched using the enzyme Phusion<sup>&#xae;</sup> High-Fidelity DNA Polymerase (NEB) with producer recommendations. Reactions occurred for 12 cycles, and sublibraries indexes were added according to <xref ref-type="bibr" rid="B77">Peterson et&#xa0;al. (2012)</xref>. PCR products were magnetic bead-purified and combined to conform three ddRADseq libraries. Libraries were sequenced on a HiSeq2500 instrument (Illumina, San Diego, CA, USA) following a pair-end strategy with a read length of 150 bp.</p>
</sec>
<sec id="s2_4">
<title>Data processing and SNP calling</title>
<p>DNA sequence quality was checked with FastQC v0.11.9. Demultiplexing was done with process_radtags from Stacks v2.5 (<xref ref-type="bibr" rid="B17">Catchen et&#xa0;al., 2013</xref>) following recommended options (<xref ref-type="bibr" rid="B81">Rochette and Catchen, 2017</xref>). Sequences with an average base quality (<italic>Q</italic>) score lower than 25 in a single 15-nt window, following a sliding window algorithm, were discarded. After that, TrimGalore/cutadapt was employed to remove 8 nt and 15 nt from the 5&#x2032; and 3&#x2032; ends, respectively, along with a 3&#x2032; quality trimming to remove bases with <italic>Q</italic> &lt; 25 (<xref ref-type="bibr" rid="B44">Krueger, 2017</xref>). Only paired reads longer than 75 nt were kept.</p>
<p>Sequences were aligned to the Matina 1&#x2013;6 cacao reference genome (<xref ref-type="bibr" rid="B67">Motamayor et&#xa0;al., 2013</xref>) using the BWA MEM algorithm (BWA v0.7.17) with the default settings (<xref ref-type="bibr" rid="B48">Li and Durbin, 2009</xref>). <italic>sam</italic> files were converted into <italic>bam</italic> files with Samtools v1.10 (<xref ref-type="bibr" rid="B49">Li et&#xa0;al., 2009</xref>), and output files were cleaned, fixed, sorted, and the RG group was added with Picard tools v2.18.25 (<xref ref-type="bibr" rid="B13">Broad Institute, 2016</xref>).</p>
<p>For SNP calling, samples from the cacao CG and CF were analyzed independently from each other, and only the read mapping in cacao chromosomes was used. This approach was applied to properly assess the genetic diversity in both CG and CF scenarios, especially when plant origins were different. Indeed, gene bank included both Cuban prospected and worldwide imported clones, while farm plants included locally propagated plants by different methods. SNPs were identified with GATK v4.2.0.0 (<xref ref-type="bibr" rid="B89">Van der Auwera and O&#x2019;Connor, 2020</xref>), combining the following tools: BaseRecalibrator, HaplotypeCaller, CombineGVCFs, and GenotypeGVCFs. Raw SNPs were filtered following GATK hard filtering recommendations: QD &lt; 2.0, SOR &gt; 3.0, MQ &lt; 50.0, FS &gt; 50.0, MQSumRank &lt; &#x2212;12.5 and ReadPosSumRank &lt; &#x2212;8.0 (<xref ref-type="bibr" rid="B15">Caetano-Anolles, 2022</xref>). An additional filtering for representativeness was applied using VCFtools v0.1.16 (maf &gt; 0.05, site missing &lt; 5%, biallelic, SNP depth coverage: 10&#x2013;80, SNP spacing &gt; = 1,000 nt) (<xref ref-type="bibr" rid="B22">Danecek et&#xa0;al., 2011</xref>).</p>
<p>Final SNP datasets were obtained by intersecting filtered SNPs from CG and CF samples with another SNP dataset built&#x2014;as described (<xref ref-type="bibr" rid="B20">Cornejo et&#xa0;al., 2018</xref>)&#x2014;from available sequence data of 65 cacao reference plants of the 10 ancestry genetic groups described by <xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al. (2008)</xref> (reference SNP dataset, <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table S2</bold></xref>), keeping only the intersected SNPs (coincident). These final SNP datasets (henceforth SNP datasets) from CG and CF samples contained 11,425 and 6,481 variants, respectively, and were employed for further analysis. SNP datasets, Transition/Transversion ratios, missing data, and depth of coverage on individual bases were estimated with VCFtools. The SNP distribution along the cacao chromosomes was analyzed using a 1 Mb window size with the function CMplot from the R package with the same name (v4.5.0) (<xref ref-type="bibr" rid="B93">Yin et&#xa0;al., 2021</xref>).</p>
</sec>
<sec id="s2_5">
<title>SNP annotation and gene ontology analysis</title>
<p>SNPs from CG and CF were annotated separately with SnpEff software v5.1d (<xref ref-type="bibr" rid="B19">Cingolani et&#xa0;al., 2012</xref>) using the available annotation for the <italic>Theobroma caca</italic>o Matina 1&#x2013;6 genome. The potential effect of the SNPs on gene expression and function, considering SNP position with respect to coding regions, was analyzed.</p>
<p>PANTHER classification system version 17.0 (released 22 February 2022) (<ext-link ext-link-type="uri" xlink:href="http://pantherdb.org/">http://pantherdb.org/</ext-link>) was used for gene ontology analysis as described (<xref ref-type="bibr" rid="B63">Mi et&#xa0;al., 2019</xref>). For that purpose, genes containing SNPs with moderate and high impact, according to the SnpEff tool, were selected. Gene lists were analyzed against the following databases: GO-Slim Molecular Function, GO-Slim Biological Process, and PANTHER Protein Class (<xref ref-type="bibr" rid="B32">GO Consortium C, 2017</xref>; <xref ref-type="bibr" rid="B63">Mi et&#xa0;al., 2019</xref>). Overrepresentation analyses of the identified genes with the databases GO molecular function complete and GO biological process complete were executed based on Fisher&#x2019;s exact test (<italic>p</italic> &lt; 0.05) with false discovery rate correction (FDR &lt; 0.05).</p>
</sec>
<sec id="s2_6">
<title>Population analysis</title>
<sec id="s2_6_1">
<title>Population structure</title>
<p>Two types of population analysis were undertaken: the first one without reference plants in order to detect genetic groups, both among CG and CF plants, independently from each other; the second one with reference plants to identify the membership of both CG and CF plants to genetic groups of cacao according to <xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al. (2008)</xref>, here referred to as ancestry genetic groups.</p>
<p>Firstly, CG and CF genetic groups were detected by ADMIXTURE v1.3.0 software (<xref ref-type="bibr" rid="B4">Alexander et&#xa0;al., 2009</xref>) with a fivefold cross-validation procedure under penalized (&#x2212;<italic>l</italic> 500, &#x2212;<italic>e</italic> 0.2) and random seedling (&#x2212;s time) conditions (<xref ref-type="bibr" rid="B3">Alexander and Lange, 2011</xref>; <xref ref-type="bibr" rid="B20">Cornejo et&#xa0;al., 2018</xref>). Twenty replicas for <italic>K</italic> values ranging from 1 to 18 (CG) and from 1 to 12 (CF) were performed following the recommendations of <xref ref-type="bibr" rid="B52">Liu et&#xa0;al. (2020)</xref>. The best <italic>K</italic> value identification was guided by the premise that all identified genetic groups must contain individuals with a high membership (<italic>Q</italic> &gt; 0.90). To this end, <italic>Q</italic>-matrices were analyzed both individually and by the online platform CLUMPAK (<xref ref-type="bibr" rid="B43">Kopelman et&#xa0;al., 2015</xref>), along with the cross-validation error (CV error) and the number of iterations to convergence of ADMIXTURE runs (<xref ref-type="bibr" rid="B4">Alexander et&#xa0;al., 2009</xref>; <xref ref-type="bibr" rid="B3">Alexander and Lange, 2011</xref>). Secondly, the kinship to the cacao ancestry genetic groups of <xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al. (2008)</xref> of each individual was calculated by running ADMIXTURE in supervised mode with the aforementioned penalized options. Shortened versions of the reference SNP dataset containing the same SNP positions as CG and CF SNP datasets were used for training purposes of ADMIXTURE runs under supervised mode. The two vcf files used for cacao ancestry estimation with ADMIXTURE contained: 1&#xb0; CG plants and the cacao reference plants and 2&#xb0; CF plants and the cacao reference plants, and were obtained by properly merging CG and CF SNP datasets and the shortened reference SNP datasets abovementioned. The capacity of SNPs included in CG and CF SNP datasets to properly separate the 65 reference plants into the expected cacao ancestry genetic group was assessed before vcf files merged, and shortened reference SNP datasets were used for that purpose (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Tables S2&#x2013;S6</bold></xref>; <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figures S1&#x2013;S6</bold></xref>).</p>
</sec>
<sec id="s2_6_2">
<title>Clustering and PCA</title>
<p>Both CG and CF samples were studied alone and in combination with the reference clones of the cacao ancestry genetic groups, leading to four different clustering and PCA analyses. Clustering analyses were performed by unweighted pair-group method with arithmetic averages (UPGMA) from Hamming distance matrices. Trees were built using a bootstrapping procedure with 1,000 replicas with aboot function of poppr package v2.9.4 (<xref ref-type="bibr" rid="B42">Kamvar et&#xa0;al., 2014</xref>, <xref ref-type="bibr" rid="B41">2015</xref>) from R statistical language version 4.2 (<xref ref-type="bibr" rid="B79">R Core Team, 2020</xref>) and visualized with ggtree package v3.10.0 (<xref ref-type="bibr" rid="B94">Yu et&#xa0;al., 2017</xref>). For principal component analysis (PCA), the glPca function from adegenet package v2.1.10 (<xref ref-type="bibr" rid="B39">Jombart, 2008</xref>; <xref ref-type="bibr" rid="B40">Jombart and Ahmed, 2011</xref>) was used with the number of alleles scaled and the alleles assumed as a unit.</p>
</sec>
<sec id="s2_6_3">
<title>Differentiation between groups and genetic diversity</title>
<p>An admixed group (Adm) of plants was formed in ADMIXTURE (see Results) and was excluded from these analyses. AMOVA test was used to detect differences among the genetic groups defined by ADMIXTURE, according to <xref ref-type="bibr" rid="B27">Excoffier et&#xa0;al. (1992)</xref>, with the function poppr.amova from poppr package v2.9.4. For the CF samples, additional putative levels of variability were also considered, i.e., farms and productive poles of the plants under study. The significance of the test was estimated by the randtest function of ade4 package v1.7-22 (<xref ref-type="bibr" rid="B24">Dray and Dufour, 2007</xref>; <xref ref-type="bibr" rid="B88">Thioulouse et&#xa0;al., 2018</xref>) with 999 permutations (<xref ref-type="bibr" rid="B79">R Core Team, 2020</xref>).</p>
<p><italic>F</italic><sub>ST</sub> pairwise coefficients for the ADMIXTURE-defined groups were estimated according to <xref ref-type="bibr" rid="B91">Weir and Cockerham (1984)</xref> with the gl.fst.pop function of dartR package v2.9.7 (<xref ref-type="bibr" rid="B35">Gruber et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B64">Mijangos et&#xa0;al., 2022</xref>); 10,000 bootstrappings were performed for confidence intervals (95%) and <italic>p</italic>-value estimation. Genetic diversity parameters, i.e., observed (<italic>H</italic><sub>obs</sub>) and expected (<italic>H</italic><sub>exp</sub>) heterozygosity, and the polymorphic information content (PIC) were estimated with adegenet and poppr packages from the R program.</p>
</sec>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>Results</title>
<sec id="s3_1">
<title>SNP calling and SNP dataset characterization</title>
<p>The three ddRADseq libraries prepared contained 406 different cacao plants, and 1,806,293,684 reads were generated during DNA sequencing. After data cleaning, 1,244,420,354 DNA sequences were retained, making an average of 2,941,892 reads per sample. However, 33 samples were removed as their read counts dropped below 1 million reads. Thus, in total, 373 cacao plants were properly sequenced with the described protocol: 238 from the CG and 135 from the CF.</p>
<p>The reads were aligned to the Matina 1&#x2013;6 cacao reference genome (<xref ref-type="bibr" rid="B67">Motamayor et&#xa0;al., 2013</xref>). CG and CF plants were treated independently from each other for the SNP calling process to properly assess the genetic diversity of gene bank and in-farm cacao resources. Raw SNPs were estimated in 1,707,351 for CG and 731,688 among CF plants. After filtering with VCFtools, 13,418 and 7,655 SNPs were retained for CG and CF, respectively. These SNPs were intersected with the reference SNP dataset of plants of the cacao ancestry genetic groups described by <xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al. (2008)</xref> (see Material and methods), retaining only intersected SNPs. The resulting datasets had 11,425 and 6,481 SNPs for the CG and CF plants, respectively, and were used for further analysis.</p>
<p>The distributions of the variables used during GATK hard filtration had similar profiles for both CG and CF SNP datasets, and the bell-shaped curve obtained for some of them suggested a low or absence of bias in the data supporting the identified SNPs (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figure S7</bold></xref>). The transition/transversion ratios estimated by VCFtools software showed comparable values among the SNP datasets (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table S7</bold></xref>). The means of missing data per individual were 1.81% (CG) and 1.54% (CF), but still 11 samples had missing data higher than 10%, with 17.1% being the highest. The average coverage depth per sample for both SNP datasets was 20.5&#xd7;, and in 12 cases, depth dropped below 10&#xd7;, reaching 6&#xd7; in one of them. Plants with relatively high missing data and low coverage were not removed from the datasets because they accounted for less than 3% of the total, and studies using up to 50% of individual missing data threshold have been reported in cacao genetic studies based on SNP markers (<xref ref-type="bibr" rid="B2">Adenet et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B37">Guti&#xe9;rrez et&#xa0;al., 2021</xref>).</p>
<p>The average SNPs per Mb were 34.6 and 19.6 for CG and CF SNP datasets, respectively. The SNP markers were distributed throughout the cacao genome, and SNP densities increased as windows moved from the center of the chromosomes to the telomeres (<xref ref-type="fig" rid="f1"><bold>Figure&#xa0;1</bold></xref>). For both cases, chromosome 1 had the higher average density per Mb (CG = 38.9 and CF = 22.2) while chromosome 7 showed the lowest values (CG = 28.9 and CF = 16.9).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Heatmap of SNP density per chromosome of cacao gene bank (CG) <bold>(A)</bold> and cacao farm (CF) <bold>(B)</bold> SNP datasets. A 1-Mb window was set for counting and plotting purposes. The plots were generated using the R package CMPlot.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1367632-g001.tif"/>
</fig>
<p>Most SNPs detected in both CG and CF SNP datasets lay on noncoding regions of the cacao reference genome, since the combination of variant categories &#x201c;intergenic region&#x201d;, &#x201c;upstream/downstream gene&#x201d;, and &#x201c;UTR and intron&#x201d; accounted for 83.96% of CG SNPs and 83.10% of CF SNPs. Missense and synonymous variants were 1,047 (9.16%) and 754 (6.60%) CG SNPs, and 625 (9.64%) and 447 (6.90%) CF SNPs, respectively. Low amounts of start/stop and splice-related variants were also detected (<xref ref-type="fig" rid="f2"><bold>Figure&#xa0;2</bold></xref>).</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Annotation of SNPs from Cuban cacao gene bank (CG) and Cuban cacao farm (CF) SNP datasets, analyzed independently with SnpEff software, using the available annotation for the <italic>Theobroma caca</italic>o Matina 1&#x2013;6 genome as a reference.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1367632-g002.tif"/>
</fig>
<p>For ontology analyses, lists of genes containing SNPs with moderate and high impact were built for each dataset. SNPs included in these impact categories had already been classified by SnpEff software as missense, splice_donor/acceptor, start_lost, stop_gained, and stop_lost variants. The lists built contained 1,052 and 636 genes for the CG and CF SNP datasets, respectively. Cellular and metabolic processes and biological regulation were the GO-Slim Biological Process terms with the highest number of hits (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figure S8</bold></xref>). In the case of GO-Slim Molecular Function, catalytic activity, binding, and transporter activity were the most abundant ones, and the protein class categories with the highest counts were metabolite interconversion enzyme, protein-modifying enzyme, transporter, and transmembrane signal receptor among the analyzed gene lists (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figure S8</bold></xref>).</p>
<p>Overrepresentation tests for molecular function and biological process (GO complete) of the gene lists revealed a more than expected representation of genes involved in protein kinase, ATP-related, and carbohydrate-binding activities for GO molecular function complete; while protein phosphorylation was the only biological process overrepresented (GO complete) (<xref ref-type="table" rid="T1"><bold>Table&#xa0;1</bold></xref>).</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Overrepresentation test results of genes containing moderate- and high-impact SNPs from CG and CF SNP datasets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">GO molecular function complete</th>
<th valign="top" align="center">REF No.</th>
<th valign="top" align="center">INPUT No.</th>
<th valign="top" align="center">Expec</th>
<th valign="top" align="center">Fold enrich</th>
<th valign="top" align="center">+/&#x2212;</th>
<th valign="top" align="center">Raw <italic>p</italic>-value</th>
<th valign="top" align="center">FDR</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="top" colspan="8" align="center">Cacao gene bank SNPs/genes</th>
</tr>
<tr>
<td valign="top" align="left">Transmembrane receptor protein serine/threonine kinase activity</td>
<td valign="top" align="center">43</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">1.56</td>
<td valign="top" align="center">5.12</td>
<td valign="top" align="center">+</td>
<td valign="top" align="center">3.70<italic>E</italic>&#x2212;04</td>
<td valign="top" align="center">2.64<italic>E</italic>&#x2212;02</td>
</tr>
<tr>
<td valign="top" align="left">Protein serine kinase activity</td>
<td valign="top" align="center">83</td>
<td valign="top" align="center">11</td>
<td valign="top" align="center">3.01</td>
<td valign="top" align="center">3.65</td>
<td valign="top" align="center">+</td>
<td valign="top" align="center">4.51<italic>E</italic>&#x2212;04</td>
<td valign="top" align="center">3.13<italic>E</italic>&#x2212;02</td>
</tr>
<tr>
<td valign="top" align="left">ABC-type transporter activity</td>
<td valign="top" align="center">110</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">3.99</td>
<td valign="top" align="center">4.01</td>
<td valign="top" align="center">+</td>
<td valign="top" align="center">8.90<italic>E</italic>&#x2212;06</td>
<td valign="top" align="center">7.94<italic>E</italic>&#x2212;04</td>
</tr>
<tr>
<td valign="top" align="left">ATP binding</td>
<td valign="top" align="center">2,127</td>
<td valign="top" align="center">160</td>
<td valign="top" align="center">77.24</td>
<td valign="top" align="center">2.07</td>
<td valign="top" align="center">+</td>
<td valign="top" align="center">2.25<italic>E</italic>&#x2212;17</td>
<td valign="top" align="center">1.40<italic>E</italic>&#x2212;14</td>
</tr>
<tr>
<td valign="top" align="left">ADP binding</td>
<td valign="top" align="center">94</td>
<td valign="top" align="center">12</td>
<td valign="top" align="center">3.41</td>
<td valign="top" align="center">3.52</td>
<td valign="top" align="center">+</td>
<td valign="top" align="center">3.44<italic>E</italic>&#x2212;04</td>
<td valign="top" align="center">2.52<italic>E</italic>&#x2212;02</td>
</tr>
<tr>
<td valign="top" align="left">Carbohydrate binding</td>
<td valign="top" align="center">289</td>
<td valign="top" align="center">27</td>
<td valign="top" align="center">10.49</td>
<td valign="top" align="center">2.57</td>
<td valign="top" align="center">+</td>
<td valign="top" align="center">3.34<italic>E</italic>&#x2212;05</td>
<td valign="top" align="center">2.69<italic>E</italic>&#x2212;03</td>
</tr>
<tr>
<td valign="top" align="left">Metal ion binding</td>
<td valign="top" align="center">2,869</td>
<td valign="top" align="center">145</td>
<td valign="top" align="center">104.18</td>
<td valign="top" align="center">1.39</td>
<td valign="top" align="center">+</td>
<td valign="top" align="center">8.55<italic>E</italic>&#x2212;05</td>
<td valign="top" align="center">6.47<italic>E</italic>&#x2212;03</td>
</tr>
<tr>
<th valign="top" colspan="8" align="center">GO biological process complete</th>
</tr>
<tr>
<td valign="top" align="left">Protein phosphorylation</td>
<td valign="top" align="center">1,160</td>
<td valign="top" align="center">95</td>
<td valign="top" align="center">42.12</td>
<td valign="top" align="center">2.26</td>
<td valign="top" align="center">+</td>
<td valign="top" align="center">1.82<italic>E</italic>&#x2212;12</td>
<td valign="top" align="center">4.23<italic>E</italic>&#x2212;09</td>
</tr>
<tr>
<th valign="top" colspan="8" align="center">Cacao farm SNPs/genes</th>
</tr>
<tr>
<td valign="top" align="left">Protein serine/threonine kinase activity</td>
<td valign="top" align="center">697</td>
<td valign="top" align="center">36</td>
<td valign="top" align="center">15.3</td>
<td valign="top" align="center">2.35</td>
<td valign="top" align="center">+</td>
<td valign="top" align="center">6.27<italic>E</italic>&#x2212;06</td>
<td valign="top" align="center">6.26<italic>E</italic>&#x2212;04</td>
</tr>
<tr>
<td valign="top" align="left">ATP-dependent activity</td>
<td valign="top" align="center">676</td>
<td valign="top" align="center">30</td>
<td valign="top" align="center">14.84</td>
<td valign="top" align="center">2.02</td>
<td valign="top" align="center">+</td>
<td valign="top" align="center">5.02<italic>E</italic>&#x2212;04</td>
<td valign="top" align="center">4.47<italic>E</italic>&#x2212;02</td>
</tr>
<tr>
<td valign="top" align="left">ATP binding</td>
<td valign="top" align="center">2,127</td>
<td valign="top" align="center">90</td>
<td valign="top" align="center">46.7</td>
<td valign="top" align="center">1.93</td>
<td valign="top" align="center">+</td>
<td valign="top" align="center">4.71<italic>E</italic>&#x2212;09</td>
<td valign="top" align="center">1.47<italic>E</italic>&#x2212;06</td>
</tr>
<tr>
<td valign="top" align="left">Carbohydrate binding</td>
<td valign="top" align="center">289</td>
<td valign="top" align="center">19</td>
<td valign="top" align="center">6.34</td>
<td valign="top" align="center">2.99</td>
<td valign="top" align="center">+</td>
<td valign="top" align="center">4.41<italic>E</italic>&#x2212;05</td>
<td valign="top" align="center">4.24<italic>E</italic>&#x2212;03</td>
</tr>
<tr>
<th valign="top" colspan="8" align="center">GO biological process complete</th>
</tr>
<tr>
<td valign="top" align="left">Protein phosphorylation</td>
<td valign="top" align="center">1,160</td>
<td valign="top" align="center">53</td>
<td valign="top" align="center">25.57</td>
<td valign="top" align="center">2.08</td>
<td valign="top" align="center">+</td>
<td valign="top" align="center">1.29<italic>E</italic>&#x2212;06</td>
<td valign="top" align="center">9.61<italic>E</italic>&#x2212;03</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Overrepresented GO complete terms according to the PANTHER classification system. Only terms/categories with significant results (p &lt; 0.05) and with false discovery rate (FDR &lt; 0.05) are shown. REF No., counting for the term/category in the reference; INPUT No., counting for the category/term in the input list; Expec, expected counting for the category/term; Fold Enrich, enrichment fold for the category/term; &#x201c;+/&#x2212;&#x201d; indicates if the category/term is overrepresented (+) or underrepresented (&#x2212;).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The new ddRADseq protocol used for cacao SNP genotyping, which proved to be efficient for high-quality SNP identification using samples from the gene bank and commercial cacao farms in the Baracoa region. The 11,425 and 6,481 variants contained in the CG and CF SNP datasets were spread throughout the 10 cacao chromosomes. Most of these SNPs were laid on noncoding regions of the cacao genome and biases toward ATP and protein phosphorylation-related activities were supported by the overrepresentation tests performed.</p>
</sec>
<sec id="s3_2">
<title>Population structure analysis of the Cuban cacao gene bank and cacao farms</title>
<p>The identification of the number of clusters or genetic groups (<italic>K</italic>) among the 238 samples of the CG proved to be a difficult task because no clear minimum for cross-validation error (CV error) was achieved when ADMIXTURE runs were analyzed, contrary to CF samples (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figure S9</bold></xref>) (<xref ref-type="bibr" rid="B4">Alexander et&#xa0;al., 2009</xref>; <xref ref-type="bibr" rid="B3">Alexander and Lange, 2011</xref>). Since we were seeking a scenario in which all identified genetic groups must contain at least some individuals with a high membership (<italic>Q</italic> &gt; 0.9), <italic>Q</italic>-matrices from every ADMIXTURE run of all <italic>K</italic> values assessed were analyzed. With CG plants, this premise of high membership groups was consistently fulfilled until <italic>K</italic> = 6, where samples with high kinship (<italic>Q</italic> &gt; 0.9) for all identified groups were found in 17 out of 20 <italic>Q</italic>-matrices (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table S8</bold></xref>). With <italic>K</italic> = 7, nine <italic>Q</italic>-matrices matched the premise; with <italic>K</italic> = 8 only three; this value keeps decreasing until zero for <italic>K</italic> &#x2265; 11.</p>
<p><italic>Q</italic>-matrices fulfilling the premise from <italic>K</italic> = 2 to <italic>K</italic> = 7 were analyzed with CLUMPAK (<xref ref-type="bibr" rid="B43">Kopelman et&#xa0;al., 2015</xref>). Among the <italic>Q</italic>-matrices included in the major clusters identified per <italic>K</italic> value, the ones belonging to the ADMIXTURE run showing the best combination of low CV error and low number of iterations to convergence for each <italic>K</italic> were selected for plotting purposes along with the membership matrix to the cacao ancestry genetic groups of <xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al. (2008)</xref> (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figure S11</bold></xref>). Since a higher congruence was detected between <italic>K</italic> = 7 and the membership to cacao ancestry genetic groups, seven genetic groups (CG1&#x2013;CG7) were assumed for CG plants (<xref ref-type="fig" rid="f3"><bold>Figure&#xa0;3A</bold></xref>).</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Memberships of 238 CG samples according to the ADMIXTURE program. Sample membership assuming <italic>K</italic> = 7 <bold>(A)</bold> as estimated by ADMIXTURE using cross-validation. Each column represents an individual. <bold>(B)</bold> Group assignment based on <italic>K</italic> = 7; Admixed plants (&#x201c;Adm&#x201d; group) in grey. <bold>(C)</bold> Membership to cacao ancestry genetic groups identified by <xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al. (2008)</xref> using ADMIXTURE under supervised mode. Ancestry plot combining CG plants and cacao reference plants is shown in <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figure S10</bold></xref>. Plots were generated using ggplot2 and ggpubr packages from the R program.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1367632-g003.tif"/>
</fig>
<p>For plant assignment to the seven CG genetic groups (<xref ref-type="fig" rid="f3"><bold>Figure&#xa0;3B</bold></xref>), all 212 individuals carrying <italic>Q</italic> &gt; 0.5 to any given group were assigned accordingly. Another 13 plants with maximum <italic>Q</italic> &lt; 0.5 and membership split into three groups were also allocated to the group with the highest <italic>Q</italic>. Finally, 13 samples were difficult to assign because they were highly mixed (membership to four or more groups, none with <italic>Q</italic> &gt; 0.5) or presented a unique membership pattern. Thus, these plants were classified as Admixed (&#x201c;Adm&#x201d; group). Pure and mixed plants were detected in most groups after the assignment, except for CG4 (<xref ref-type="fig" rid="f3"><bold>Figure&#xa0;3</bold></xref>).</p>
<p>In cacao farm population structure analysis, the minimum at <italic>K</italic> = 4 in the CV error vs. <italic>K</italic> plot built from ADMIXTURE runs with the 135 CF samples strongly supported the presence of four groups (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figure S9B</bold></xref>), and all the groups had plants with high kinship (<italic>Q</italic> &gt; 0.90). Thus, <italic>K</italic> = 4 was assumed as the most probable number of genetic groups (CF1&#x2013;CF4) (<xref ref-type="fig" rid="f4"><bold>Figure&#xa0;4A</bold></xref>). Plants carrying each possible combination of groups were detected in addition to the pure ones. CF plants were assigned to the group with the highest membership (<xref ref-type="fig" rid="f4"><bold>Figure&#xa0;4B</bold></xref>). CF2 (45) was the largest group, and CF3 (16) was the smallest one.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Memberships of CF samples according to the ADMIXTURE program. Sample membership assuming <italic>K</italic> = 4 <bold>(A)</bold> according to ADMIXTURE using cross-validation. Each column represents an individual. <bold>(B)</bold> Group assignment based on <italic>K</italic> = 4. <bold>(C)</bold> Membership to cacao ancestry genetic groups identified by <xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al. (2008)</xref> using ADMIXTURE under supervised mode. Ancestry plot combining CF plants and cacao reference plants is shown in <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figure S12</bold></xref>. Plots were generated using ggplot2 and ggpubr packages from the R program.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1367632-g004.tif"/>
</fig>
<p>Cacao genetic ancestries among CG and CF plants, according to <xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al. (2008)</xref>, shared some similarities. In both cases, Amelonado was the most abundant, with 49.22% in CG and 57.73% in CF of the total ancestry. Other common ancestries detected were Criollo (CG = 16.82% and CF = 19.13%), Iquitos (10.9% and 7.87%), Nanay (8.91% and 4.46%), and Contamana (6.51% and 9.83%). It excelled in the low representation of Nacional ancestry among CF samples in contrast to CG where it accounted for 5.36% of the total (<xref ref-type="fig" rid="f3"><bold>Figures&#xa0;3C</bold></xref>, <xref ref-type="fig" rid="f4"><bold>4C</bold></xref>). Criollo and Nacional ancestries were only found in hybrid plants with Amelonado, and for the other common ancestries, both pure plants and hybrids were detected. Plants with high membership (<italic>Q</italic> &gt; 0.9) to the Amelonado genetic group were identified in CG (30) and CF (21) while pure plants to Contamana (four), Iquitos (two), Nanay (one), and Mara&#xf1;on (one) were found in CG. Remarkably, this last one (C042) was located within the Admixed group, presumably because it was the unique individual showing a high <italic>Q</italic> to Mara&#xf1;on (<xref ref-type="fig" rid="f3"><bold>Figure&#xa0;3</bold></xref>). Curaray, Pur&#xfa;s, and Guiana ancestries were very low represented or absent among all studied plants. Each genetic group showed a distinctive cacao ancestry composition, and, in some cases, CG and CF groups were alike based on their cacao ancestries (<xref ref-type="fig" rid="f3"><bold>Figures&#xa0;3</bold></xref>, <xref ref-type="fig" rid="f4"><bold>4</bold></xref>). In this sense, CG1 and CF1 were basically conformed by plants with high membership to Amelonado (CG1 average <italic>Q</italic> = 0.86, CF1 average <italic>Q</italic> = 0.91), slightly combined with other ancestries; CG2 and CF2 mostly contained hybrids of Amelonado and Criollo ancestries, though CG2 also had contributions from other ancestries; and finally, CG3 and CF3 showed almost pure Contamana individuals as well as combinations of Contamana mainly with Amelonado and Criollo.</p>
<p>Further similarities based on cacao ancestries between the remaining CG and CF genetic groups were difficult to establish because of their particular ancestry combination or proportion they carried. On one side, CG4 individuals were a complex mix of Amelonado, Iquitos, Contamana, and Criollo ancestries; CG5 group mainly contained hybrids of Amelonado and Nacional; CG6 had Nanay as distinctive ancestry alone or in combination with Amelonado, Criollo, and others; and CG7 had Iquitos individuals and their hybrids with Amelonado plus a portion of Criollo ancestry. On the other side, CF4 had the highest mixture of cacao ancestries among CF genetic groups, with plants carrying the combinations Amelonado&#x2013;Iquitos&#x2013;Contamana&#x2013;Criollo, Amelonado&#x2013;Nanay&#x2013;Criollo, Amelonado&#x2013;Iquitos&#x2013;Criollo, and other minor combinations. Apparently, CF4 contained some of the ancestry combinations from CG4, CG6, and CG7, but the proportions were different; e.g., CG6 and CG7 plants had higher membership coefficients to Nanay and Iquitos, respectively, than CF4 plants carrying these ancestries (<xref ref-type="fig" rid="f3"><bold>Figures&#xa0;3C</bold></xref>, <xref ref-type="fig" rid="f4"><bold>4C</bold></xref>). It was noteworthy that for CG and CF plants, the conformation of the genetic groups, as K increased from K=2 to the most probable K value (seven for CG and four for CF), was highly related to the putative origin of the plants based on their cacao ancestries (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figure S11</bold></xref> and <xref ref-type="supplementary-material" rid="SM1"><bold>S13</bold></xref>).</p>
<p>Thus, using the model-based approach implemented in ADMIXTURE, seven and four genetic groups were detected among CG and CF samples, respectively. Cacao ancestry analysis revealed that some CG and CF groups had similar cacao ancestry compositions. Amelonado was the predominant ancestry in CG and CF plants; other commonly detected ancestries were Criollo, Contamana, Iquitos, and Nanay. Nacional ancestry was practically lacking in CF, while several CG plants had this ancestry as a hybrid with Amelonado. Mara&#xf1;on, Curaray, Pur&#xfa;s, and Guiana ancestries were underrepresented or absent.</p>
</sec>
<sec id="s3_3">
<title>Clustering and PCA of the Cuban cacao gene bank and cacao farms</title>
<p>Clustering analysis by UPGMA and PCA was used to confirm the genetic groups identified in CG and CF. Each analysis type was conducted with and without the 65 cacao plants used as a reference for the ancestry genetic groups. The dendrogram of either the 238 CG samples combined with the 65 references (<xref ref-type="fig" rid="f5"><bold>Figure&#xa0;5</bold></xref>) or the 135 CF plants and the references (<xref ref-type="fig" rid="f6"><bold>Figure&#xa0;6</bold></xref>) showed mostly congruent results with their respective population structure results. CG1, CG2, and CG4 plants, as well as CF1 and CF2, which all shared an important ancestry from Amelonado, basically formed independent clusters, which were allocated in the same branches as the reference plants of Amelonado in both dendrograms, though some plants from CG2 were located differently. Similarly, CG3 and CF3 samples carrying a high proportion of Contanama ancestry mainly clustered together in their respective dendrograms and closed to&#x2014;or mixed with&#x2014;Contamana reference plants (<xref ref-type="fig" rid="f5"><bold>Figures&#xa0;5</bold></xref>, <xref ref-type="fig" rid="f6"><bold>6</bold></xref>).</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Dendrogram with 238 CG samples and 65 reference plants of cacao genetic groups. Clustering by UPGMA was based on a Hamming distance matrix calculated from a merge vcf file containing both sets of individuals with the 11,425 SNPs from the CG SNP dataset. Coloring is based on group membership from the ADMIXTURE program (<italic>K</italic> = 7) for the 238 CG samples (CG Genetic Groups) and cacao ancestry genetic groups according to <xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al. (2008)</xref> for the 65 references (<sup>*</sup>). The plot was generated using ggtree and treeio packages from the R program.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1367632-g005.tif"/>
</fig>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Dendrogram with 135 CF samples and 65 cacao genetic group controls. Clustering by the UPGMA method was based on a Hamming distance matrix calculated from a merge vcf file containing both sets of individuals with the 6,481 SNPs from the CF SNP dataset. Coloring is based on group membership from the ADMIXTURE program (<italic>K</italic> = 4) for the 135 CF samples (CF Genetic Groups) and cacao ancestry genetic groups according to <xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al. (2008)</xref> for the 65 references (<sup>*</sup>). Light green background highlights hybrid individuals carrying Nanay ancestry. The plot was generated using ggtree and treeio packages from the R program.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1367632-g006.tif"/>
</fig>
<p>A high congruence was also seen in CG5 (Nacional Ancestry) and CG7 (Iquitos) clustering, which mostly formed independent clades together with their respective cacao references (<xref ref-type="fig" rid="f5"><bold>Figure&#xa0;5</bold></xref>). While CG6 (Nanay) were divided in two sub-groups, one of them laid next to the expected Nanay reference plants (<xref ref-type="fig" rid="f5"><bold>Figure&#xa0;5</bold></xref>). CF4 individuals were also split in two major clusters (<xref ref-type="fig" rid="f6"><bold>Figure&#xa0;6</bold></xref>), possibly as a result of their relatively high mix of cacao ancestry genetic groups revealed by ADMIXTURE (<xref ref-type="fig" rid="f4"><bold>Figure&#xa0;4C</bold></xref>). Surprisingly, samples with Nanay ancestry within CF4 grouped together in the dendrogram (highlighted in green color, <xref ref-type="fig" rid="f6"><bold>Figure&#xa0;6</bold></xref>), but apart from Nanay controls, which might be related to the lower Nanay ancestry of these individuals in comparison with CG6 plants. Interestingly, clustering analysis without cacao references improved the grouping of CG3 and CG6 individuals while having no effect on CF sample clustering (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figures S14, S15</bold></xref>). Finally, the dendrogram of CG samples revealed the proximity of C042 to the Mara&#xf1;on cacao reference plants (<xref ref-type="fig" rid="f5"><bold>Figure&#xa0;5</bold></xref>), as opposite to the population structure analysis revealed by ADMIXTURE (<xref ref-type="fig" rid="f3"><bold>Figure&#xa0;3</bold></xref>), which put C042 apart, although presenting 99% of ancestry to Mara&#xf1;on. This apparent discrepancy was due to the unique profile of C042 and the very low occurrence of Mara&#xf1;on ancestry in CG samples.</p>
<p>PCA results of the 238 CG and 135 CF plants mostly agreed with ADMIXTURE and clustering findings (<xref ref-type="fig" rid="f7"><bold>Figure&#xa0;7</bold></xref>). The first three principal components explained 33.6% and 39.7% of the total variance of CG and CF SNP datasets, respectively. In both cases, PC1 and PC2 (<xref ref-type="fig" rid="f7"><bold>Figures&#xa0;7A, C</bold></xref>) mostly separated genetic groups, carrying Amelonado (CG1, CF1) and Amelonado&#x2013;Criollo hybrids (CG2, CF2) from each other and from the rest of the groups. However, several CG2 plants intruded into other groups, similar to the clustering results of CG2 (<xref ref-type="fig" rid="f5"><bold>Figure&#xa0;5</bold></xref>), which would be associated with the presence of other ancestries apart from Amelonado and Criollo in these plants (<xref ref-type="fig" rid="f3"><bold>Figure&#xa0;3</bold></xref>). PC2 and PC3 plots (<xref ref-type="fig" rid="f7"><bold>Figures&#xa0;7B, D</bold></xref>) achieved the full separation of CG3 as well as CF3 and CF4. The other CG genetic groups were difficult to analyze because of the high mixed pattern observed, though CG6 and CG7 were mostly put apart from the rest of the samples without a clear separation between them. Nanay (found in CG6) and Iquitos (CG7) ancestries have proved to be difficult to separate from each other (<xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al., 2008</xref>; <xref ref-type="bibr" rid="B74">Osorio-Guarin et&#xa0;al., 2017</xref>). PCA, combining either CG or CF samples with cacao reference plants (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figure S16</bold></xref>), put together CG1 and CF1 with Amelonado references as expected. Furthermore, most CG2 and CF2 plants, formed by Amelonado and Criollo hybrids, were mainly spread between Amelonado and Criollo reference plants, except for the CG2 plants carrying additional cacao ancestries. CG3 and CF3 plants were closed to Contamana references&#x2014;or in between Amelonado and Contamana&#x2014;and those plants with the highest membership coefficient to Contamana ancestry were the closet ones to these ancestry references. In the case of CF4, samples were related to different cacao ancestry genetic groups, as expected.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Principal component analysis plots of the 238 CG samples <bold>(A, B)</bold> and of the 135 CF samples <bold>(C, D)</bold> using CG and CF SNP datasets, respectively. <bold>(A)</bold> PC1 and PC2 of CG samples, <bold>(B)</bold> PC2 and PC3 of CG samples, <bold>(C)</bold> PC1 and PC2 of CF samples, and <bold>(D)</bold> PC2 and PC3 of CF samples. Coloring is based on group membership from the ADMIXTURE program, assuming <italic>K</italic> = 7 and <italic>K</italic> = 4 for CG and CF samples, respectively. Plots were generated using ggplot2 and ggpubr packages from the R program.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1367632-g007.tif"/>
</fig>
<p>In general, clustering by UPGMA and PCA results were consistent with each other, and with population structure results supporting the presence of seven and four genetic groups among CG and CF plants, as well as the cacao ancestries identified in each genetic group. UPGMA performed better for CG samples than for CF, while PCA provided better support for CF genetic groups than for CG ones.</p>
</sec>
<sec id="s3_4">
<title>Differentiation between groups and genetic diversity of the Cuban cacao gene bank and cacao farms</title>
<p>AMOVA and <italic>F</italic><sub>ST</sub> pairwise comparisons were independently performed on CG and CF plants to assess the genetic variability and differentiation among the identified groups. Similar contributions to variability were detected between CG (29.51%) and CF (30.15%) genetic groups, but the major contribution to the variability came from within groups with 70.49% (CG) and 69.85% (CF) of the total variance (<xref ref-type="table" rid="T2"><bold>Tables&#xa0;2</bold></xref>, <xref ref-type="table" rid="T3"><bold>3</bold></xref>). For CF samples, farms and productive poles were also evaluated as putative sources of variation, but even less contribution to variability was detected using these levels (<xref ref-type="table" rid="T3"><bold>Table&#xa0;3</bold></xref>).</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>AMOVA results from CG plants assuming seven genetic groups.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Source of variation</th>
<th valign="top" align="center"><italic>Df</italic>
</th>
<th valign="top" align="center">SS</th>
<th valign="top" align="center">MS</th>
<th valign="top" align="center">Sigma</th>
<th valign="top" align="center">Variance (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Between groups</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">103,965.08</td>
<td valign="top" align="center">17,327.51</td>
<td valign="top" align="center">540.74</td>
<td valign="top" align="center">29.51</td>
</tr>
<tr>
<td valign="top" align="left">Within groups</td>
<td valign="top" align="center">218</td>
<td valign="top" align="center">281,633.74</td>
<td valign="top" align="center">1,291.90</td>
<td valign="top" align="center">1,291.90</td>
<td valign="top" align="center">70.49</td>
</tr>
<tr>
<td valign="top" align="left">Total</td>
<td valign="top" align="center">224</td>
<td valign="top" align="center">385,598.82</td>
<td valign="top" align="center">1,721.42</td>
<td valign="top" align="center">1,832.64</td>
<td valign="top" align="center">100.00</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Df, degree of freedom; SS, square sum; MS, mean square. The Adm group was excluded from the analysis, p &lt; 0.001.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>AMOVA results from CF samples assuming different organizing levels.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" rowspan="2" align="center">Source of variation</th>
<th valign="top" colspan="3" align="center">Identified CF groups</th>
<th valign="top" colspan="3" align="center">Cacao farms</th>
<th valign="top" colspan="3" align="center">Productive poles</th>
</tr>
<tr>
<th valign="top" align="center"><italic>Df</italic>
</th>
<th valign="top" align="center">Sigma</th>
<th valign="top" align="center">Var (%)</th>
<th valign="top" align="center"><italic>Df</italic>
</th>
<th valign="top" align="center">Sigma</th>
<th valign="top" align="center">Var (%)</th>
<th valign="top" align="center"><italic>Df</italic>
</th>
<th valign="top" align="center">Sigma</th>
<th valign="top" align="center">Var (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">Between organizing levels</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">293.83</td>
<td valign="top" align="center">30.15</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">60.15</td>
<td valign="top" align="center">6.65</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">46.92</td>
<td valign="top" align="center">5.14</td>
</tr>
<tr>
<td valign="top" align="center">Within organizing levels</td>
<td valign="top" align="center">131</td>
<td valign="top" align="center">680.68</td>
<td valign="top" align="center">69.85</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">844.88</td>
<td valign="top" align="center">93.35</td>
<td valign="top" align="center">132</td>
<td valign="top" align="center">865.58</td>
<td valign="top" align="center">94.86</td>
</tr>
<tr>
<td valign="top" align="center">Total</td>
<td valign="top" align="center">134</td>
<td valign="top" align="center">974.51</td>
<td valign="top" align="center">100.00</td>
<td valign="top" align="center">134</td>
<td valign="top" align="center">905.03</td>
<td valign="top" align="center">100.00</td>
<td valign="top" align="center">134</td>
<td valign="top" align="center">912.50</td>
<td valign="top" align="center">100.00</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Organizing levels: identified groups with AMIXTURE software, cacao farms, and productive poles. Df, degree of freedom; Var, variance. Levels refer to the different hierarchy evaluated: identified groups (defined by ADMIXTURE), cacao farms (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table S1</bold></xref>), and productive poles (farm location: Jamal, San Luis and Paso de Cuba/Sabanilla), p &lt; 0.001.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p><italic>F</italic><sub>ST</sub> pairwise comparisons among either CG or CF genetic groups revealed all groups were significantly different (<italic>p</italic> = 0) from each other and supported a moderate to very large genetic differentiation (<xref ref-type="table" rid="T4"><bold>Tables&#xa0;4</bold></xref>, <xref ref-type="table" rid="T5"><bold>5</bold></xref>). <italic>F</italic><sub>ST</sub> values for CG genetic groups (from 0.071 to 0.407) had a broader variation range than for CF groups (from 0.093 to 0.282). The highest <italic>F</italic><sub>ST</sub> values were obtained for the pairs CG1&#x2013;CG3 (0.407) in CG and CF1&#x2013;CF3 (0.282) in CF genetic groups. The cacao ancestries of the groups with the highest differentiation were alike, as CG1 and CF1 were mostly Amelonado plants, and CG3 and CF3&#x2019;s distinctive ancestry was Contamana. On the other hand, the lowest <italic>F</italic><sub>ST</sub> values were estimated for the pairs CG6&#x2013;CG7 (0.071) and CF2&#x2013;CF4 (0.093) from CG and CF, respectively. Taking the groups individually, CG3 (<italic>F</italic><sub>ST</sub> ranging from 0.225 to 0.407) and CF1 (<italic>F</italic><sub>ST</sub> from 0.124 to 0.282) had the largest differentiation from the rest of the CG and CF genetic groups, respectively.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p><italic>F</italic><sub>ST</sub> pairwise comparison among the seven genetic groups identified in CG plants.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="top" align="center">CG1</th>
<th valign="top" align="center">CG2</th>
<th valign="top" align="center">CG3</th>
<th valign="top" align="center">CG4</th>
<th valign="top" align="center">CG5</th>
<th valign="top" align="center">CG6</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">CG2</td>
<td valign="top" align="center">0.144</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="center">CG3</td>
<td valign="top" align="center">0.407</td>
<td valign="top" align="center">0.225</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="center">CG4</td>
<td valign="top" align="center">0.243</td>
<td valign="top" align="center">0.174</td>
<td valign="top" align="center">0.279</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="center">CG5</td>
<td valign="top" align="center">0.175</td>
<td valign="top" align="center">0.125</td>
<td valign="top" align="center">0.23</td>
<td valign="top" align="center">0.171</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="center">CG6</td>
<td valign="top" align="center">0.213</td>
<td valign="top" align="center">0.142</td>
<td valign="top" align="center">0.252</td>
<td valign="top" align="center">0.162</td>
<td valign="top" align="center">0.129</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="center">CG7</td>
<td valign="top" align="center">0.176</td>
<td valign="top" align="center">0.11</td>
<td valign="top" align="center">0.238</td>
<td valign="top" align="center">0.172</td>
<td valign="top" align="center">0.113</td>
<td valign="top" align="center">0.071</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Adm samples were excluded from the analysis. All F<sub>ST</sub> values were significant (p = 0).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p><italic>F</italic><sub>ST</sub> pairwise pairwise comparison among the four groups identified in CF plants.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="top" align="center">CF1</th>
<th valign="top" align="center">CF2</th>
<th valign="top" align="center">CF3</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center"><bold>CF2</bold>
</td>
<td valign="top" align="center">0.211</td>
<td valign="top" align="left"/>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="center"><bold>CF3</bold>
</td>
<td valign="top" align="center">0.282</td>
<td valign="top" align="center">0.192</td>
<td valign="top" align="left"/>
</tr>
<tr>
<td valign="top" align="center"><bold>CF4</bold>
</td>
<td valign="top" align="center">0.124</td>
<td valign="top" align="center">0.093</td>
<td valign="top" align="center">0.105</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>All F<sub>ST</sub> values were significant (p = 0).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Genetic diversity parameters for all CG individuals (<italic>H</italic><sub>obs</sub> = 0.264, <italic>H</italic><sub>exp</sub> = 0.283, PIC = 0.235) were similar to the ones from CF plants (0.296, 0.286, 2.36) though <italic>H</italic><sub>obs</sub> was slightly higher for CF samples (<xref ref-type="table" rid="T6"><bold>Table&#xa0;6</bold></xref>). These parameters exhibited variability among both the CG and CF genetic groups. Genetic groups carrying the highest proportion of Amelonado ancestry (CG1 and CF1) had the lowest genetic diversity values. The highest <italic>H</italic><sub>obs</sub> were obtained in groups mostly conformed by Amelonado/Criollo hybrids (CG2 and CF2). CG6 and CG7 were very alike in terms of genetic diversity; these two groups had already shown the lowest <italic>F</italic><sub>ST</sub> values in group differentiation analysis. The CG4 group had a unique behavior since its <italic>H</italic><sub>exp</sub> (0.215) and PIC (0.162) values were remarkably lower than the estimated for CG samples, while <italic>H</italic><sub>obs</sub> (0.285) was slightly higher.</p>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Genetic diversity in CG and CF plants (total) and among CG and CF genetic groups.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="top" align="center">Group</th>
<th valign="top" align="center"><italic>N</italic>
</th>
<th valign="top" align="center"><italic>H</italic><sub>obs</sub>
</th>
<th valign="top" align="center"><italic>H</italic><sub>exp</sub>
</th>
<th valign="top" align="center">PIC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" rowspan="8" align="center"><bold>CG&#xa0;samples</bold>
</td>
<td valign="top" align="center">CG1</td>
<td valign="top" align="center">62</td>
<td valign="top" align="center">0.143</td>
<td valign="top" align="center">0.157</td>
<td valign="top" align="center">0.135</td>
</tr>
<tr>
<td valign="top" align="center">CG2</td>
<td valign="top" align="center">68</td>
<td valign="top" align="center">0.346</td>
<td valign="top" align="center">0.299</td>
<td valign="top" align="center">0.234</td>
</tr>
<tr>
<td valign="top" align="center">CG3</td>
<td valign="top" align="center">10</td>
<td valign="top" align="center">0.225</td>
<td valign="top" align="center">0.251</td>
<td valign="top" align="center">0.191</td>
</tr>
<tr>
<td valign="top" align="center">CG4</td>
<td valign="top" align="center">10</td>
<td valign="top" align="center">0.285</td>
<td valign="top" align="center">0.215</td>
<td valign="top" align="center">0.162</td>
</tr>
<tr>
<td valign="top" align="center">CG5</td>
<td valign="top" align="center">22</td>
<td valign="top" align="center">0.284</td>
<td valign="top" align="center">0.266</td>
<td valign="top" align="center">0.209</td>
</tr>
<tr>
<td valign="top" align="center">CG6</td>
<td valign="top" align="center">34</td>
<td valign="top" align="center">0.295</td>
<td valign="top" align="center">0.276</td>
<td valign="top" align="center">0.224</td>
</tr>
<tr>
<td valign="top" align="center">CG7</td>
<td valign="top" align="center">19</td>
<td valign="top" align="center">0.295</td>
<td valign="top" align="center">0.271</td>
<td valign="top" align="center">0.216</td>
</tr>
<tr>
<td valign="top" align="center"><bold>Total</bold>
</td>
<td valign="top" align="center"><bold>225</bold>
</td>
<td valign="top" align="center"><bold>0.264</bold>
</td>
<td valign="top" align="center"><bold>0.283</bold>
</td>
<td valign="top" align="center"><bold>0.235</bold>
</td>
</tr>
<tr>
<td valign="top" rowspan="5" align="center"><bold>CF&#xa0;samples</bold>
</td>
<td valign="top" align="center">CF1</td>
<td valign="top" align="center">33</td>
<td valign="top" align="center">0.114</td>
<td valign="top" align="center">0.129</td>
<td valign="top" align="center">0.115</td>
</tr>
<tr>
<td valign="top" align="center">CF2</td>
<td valign="top" align="center">45</td>
<td valign="top" align="center">0.427</td>
<td valign="top" align="center">0.292</td>
<td valign="top" align="center">0.223</td>
</tr>
<tr>
<td valign="top" align="center">CF3</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">0.296</td>
<td valign="top" align="center">0.305</td>
<td valign="top" align="center">0.236</td>
</tr>
<tr>
<td valign="top" align="center">CF4</td>
<td valign="top" align="center">41</td>
<td valign="top" align="center">0.303</td>
<td valign="top" align="center">0.296</td>
<td valign="top" align="center">0.241</td>
</tr>
<tr>
<td valign="top" align="center"><bold>Total</bold>
</td>
<td valign="top" align="center"><bold>135</bold>
</td>
<td valign="top" align="center"><bold>0.296</bold>
</td>
<td valign="top" align="center"><bold>0.286</bold>
</td>
<td valign="top" align="center"><bold>0.236</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>N, number of individuals; H<sub>obs</sub>, observed heterozygosity; H<sub>exp</sub>, expected heterozygosity; PIC, Polymorphic Information Content. Adm group was excluded from the analysis.</p>
<p>The bold values represent the Total value of the parameters for CG and CF samples.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Summarizing, genetic groups from either CG or CF plants were significantly different from each other. Although genetic differentiation was higher among CG groups than among CF groups. The highest <italic>F</italic><sub>ST</sub> values came from pairwise comparison of groups carrying mostly Amelonado ancestry and those with Contamana background. Similar genetic diversity parameters were obtained in CG and CF samples. Genetic groups with the highest Amelonado ancestry proportion had the lowest genetic diversity parameters.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<title>Discussion</title>
<sec id="s4_1">
<title>SNP calling and SNP dataset characterization</title>
<p>Few protocols based on the double-digest approach of GBS technologies have been described for genetic studies in <italic>Theobroma cacao</italic>. Here, we used a ddRADseq approach (<xref ref-type="bibr" rid="B77">Peterson et&#xa0;al., 2012</xref>) with the enzymes <italic>Eco</italic>RI and <italic>Nla</italic>III to build three DNA sequencing libraries containing 406 samples from the Cuban cacao gene bank (CG) and cacao farms (CF). After data processing, the final number of SNPs identified in the CG (11,425 in 238 plants) and CF (6,481 in 135 plants) were higher than in other double-digestion-based GBS protocols used in cacao. <xref ref-type="bibr" rid="B46">Lachenaud et&#xa0;al. (2018)</xref> studied the population structure of 181 cacao clones from CIRAD&#x2019;s Paracou-Combi station, French Guiana, with 3,409 SNPs derived from DArTseq technology based on a PstI/MseI genome digestion. <xref ref-type="bibr" rid="B75">Osorio-Guarin et&#xa0;al. (2020)</xref> used the enzyme combination BsaXI/CspCI to conduct the sequencing of 229 cacao accessions of Colombian germplasm collection and identified 8,131 or 9,003 SNPs depending on the reference genome used (Matina 1&#x2013;6 or Criollo B97-61/B2, respectively). <xref ref-type="bibr" rid="B2">Adenet et&#xa0;al. (2020)</xref> assessed the cacao genetic diversity of 147 plants in Martinique and identified 4,113 SNP markers using GBS libraries built from double digestion with the same enzymes as <xref ref-type="bibr" rid="B46">Lachenaud et&#xa0;al. (2018)</xref>.</p>
<p>The higher number of SNPs detected in our case is probably related to the technology and GBS protocol used. First, the combination of the enzymes NlaIII (four-cutter) and EcoRI (six-cutter) theoretically generates 81,158 DNA fragments between 300 bp and 500 bp from the Matina genome estimated by RADinitio software (<xref ref-type="bibr" rid="B80">Rivera-Colon et&#xa0;al., 2021</xref>). This number is higher than the 42,849 fragments predicted by the <italic>in silico</italic> Matina genome digestion with the enzyme combination described by <xref ref-type="bibr" rid="B75">Osorio-Guarin et&#xa0;al. (2020)</xref>, even though they employed a broader fragment size selection (200&#x2013;700). Second, the amount of DNA used (1 &#xb5;g) in library preparation was larger than in other protocols (<xref ref-type="bibr" rid="B2">Adenet et&#xa0;al. (2020)</xref>&#x2014;200 ng&#x2014;and a greater amount of starting DNA helps in preventing biases during the PCR enrichment step, which could lead to genotyping errors during the SNP calling process (<xref ref-type="bibr" rid="B5">Andrews et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B80">Rivera-Colon et&#xa0;al., 2021</xref>). Last but not the least, the sequencing strategy we followed (150 bp, pair-end) should contribute to identifying more SNPs than the <xref ref-type="bibr" rid="B75">Osorio-Guarin et&#xa0;al. (2020)</xref> or <xref ref-type="bibr" rid="B2">Adenet et&#xa0;al. (2020)</xref> approaches, which used a 100-pb/pair-end and a 150-bp/single-end sequencing configuration, respectively. Interestingly, the number of filtered SNPs could be raised to 28,151 (CG) and 13,791 (CF) if the condition of SNP minimal separation of 1,000 base pairs during the VCFtools filtering step is removed.</p>
<p>A major goal of cacao genetic studies is to determine the presence of cacao genetic group ancestries in the plants under study, as described by <xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al. (2008)</xref>, which forces the use of cacao reference clones of those genetic groups. Two approaches have been reported to get the genotypes of the controls in GBS-based studies: 1&#xb0; to process reference plants in the same way as the samples under study (<xref ref-type="bibr" rid="B46">Lachenaud et&#xa0;al., 2018</xref>) and 2&#xb0; to exploit published data to get genotypes and hence the SNPs of a group of reference clones (<xref ref-type="bibr" rid="B75">Osorio-Guarin et&#xa0;al., 2020</xref>). We opted for the last choice and built a SNP dataset with 65 cacao reference plants, according to <xref ref-type="bibr" rid="B20">Cornejo et&#xa0;al. (2018)</xref> (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table S2</bold></xref>).</p>
<p>After intersecting the VCFtools-filtered SNPs with the cacao reference SNP dataset, 85.15% (11,425) of CG and 84.66% (6,481) of CF-filtered SNPs were retained. These percentages are higher than the ones reported by <xref ref-type="bibr" rid="B75">Osorio-Guarin et&#xa0;al. (2020)</xref>, who obtained 3,712 SNPs out of 9,003 (45.65%) after intersecting their dataset with a SNP dataset built from raw sequence data of 69 cacao reference plants using Criollo cacao as reference genome. The capacity of the SNPs contained in both CG and CF SNP datasets to properly differentiate cacao genetic reference plants into 10 genetic groups was successfully confirmed (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Tables S2&#x2013;S6</bold></xref>; <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figures S1&#x2013;S6</bold></xref>) and validated the strategy followed.</p>
<p>The transitions/transversions ratio (Ti/Tv) estimated for the SNP datasets (1.647 for CG and 1.651 for CF) were similar to the 1.682 Ti/Tv value calculated for a cacao SNP dataset by <xref ref-type="bibr" rid="B76">Osorio-Guarin et&#xa0;al. (2018)</xref>. This ratio has been used as a quality control parameter for checking the overall SNP quality during GBS experiments since SNP datasets of the same species should have similar values (<xref ref-type="bibr" rid="B36">Guo et&#xa0;al., 2014</xref>). The distribution of the per-SNP parameters used in GATK hard filtration showed the expected profile according to GATK hard filtering best practices (<xref ref-type="bibr" rid="B15">Caetano-Anolles, 2022</xref>). Put together, these results witness the overall good quality of the SNP datasets obtained for both sample groups.</p>
<p>The distribution of the SNPs along the cacao genome was different from that of <xref ref-type="bibr" rid="B75">Osorio-Guarin et&#xa0;al. (2020)</xref>, who obtained the lowest and highest SNP density in chromosome 8 (25.13 per Mb) and chromosome 10 (32.15 per Mb), respectively. In our case, chromosome 1 had the highest average density, while chromosome 7 showed the lowest value for both CG and CF samples; the differences in the library preparation protocols (enzymes, fragment size) justify such behavior. We also detected an increase in SNP density as the locus moved from the chromosome center toward the telomeres. Such a pattern has already been described in GBS-based genetic studies of other crop species (<xref ref-type="bibr" rid="B45">Kumar et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B95">Yu et&#xa0;al., 2021</xref>). Centromeres and pericentromeric chromosome regions usually show a tendency of increased DNA methylation (<xref ref-type="bibr" rid="B1">Achrem et&#xa0;al., 2020</xref>); therefore, a lower amount of DNA fragments from this region should be expected when restriction enzymes sensitive to DNA methylation are used for genomic DNA digestion. EcoRI, one of the enzymes we used, is partially blocked by some combinations of overlapping in CpG-methylated DNA (<xref ref-type="bibr" rid="B70">NEB, New England Biolabs, 2022</xref>), which could support the lower number of SNPs detected toward the center of the chromosomes (<xref ref-type="fig" rid="f1"><bold>Figure&#xa0;1</bold></xref>). On the other hand, many crop species such as barley, wheat, maize, tomato, and cotton showed high recombination rates in distal regions of the chromosome (<xref ref-type="bibr" rid="B53">Lloyd, 2022</xref>), which could favor the occurrence of genetic variation, including SNPs, in these parts of the genome.</p>
<p>The annotation of SNPs from both datasets located more than 83% of the variants in noncoding regions of the genome. <xref ref-type="bibr" rid="B20">Cornejo et&#xa0;al. (2018)</xref> also reported a large majority of the identified variants in noncoding regions using whole genome sequencing. However, the percentage of missense variants (average 9.4% for both datasets) and synonymous variants (~ 6.75%) that we obtained were higher than the 4.35% and 2.97% of missense and synonymous variants, respectively, obtained by these authors. This result supports a bias toward coding regions of the cacao genome with the ddRADseq sequencing protocol here described, likely associated with the enzyme combination used for genomic DNA digestion and the DNA fragment size selected.</p>
<p>Overrepresentation test results of genes comprising moderate and high-impact SNPs also revealed a bias since the gene molecular functions: protein kinase, ATP-related, and carbohydrate-binding activities were overrepresented while protein phosphorylation was the only biological process overrepresented. These aspects are common to many processes in plants, such as sensing, signaling, abiotic and biotic stress response, and growth, among others (<xref ref-type="bibr" rid="B82">Saijo and Loo, 2020</xref>; <xref ref-type="bibr" rid="B18">Chen et&#xa0;al., 2021</xref>). Therefore, these SNP datasets provided a suitable platform to deepen the genetic basis of physiological processes and agronomic indicators of cacao plants, in which the aforementioned functions and processes play a central role.</p>
</sec>
<sec id="s4_2">
<title>Population analysis of the Cuban cacao gene bank and of cacao farms</title>
<p>The establishment of the Cuban cacao gene bank started more than 40 years ago and has been enriched throughout the years by solidary donations, the exchange of biological materials, and field expeditions. Presently, the collection hosts 282 cacao accessions, which constitute the genetic basis of the Cuban cacao improvement program and an important source of the genetic material used for cacao farming in the Baracoa region.</p>
<p>Using the model-based clustering of ADMIXTURE software, seven genetic groups were identified among the CG plants, while four were detected among cacao farm (CF) samples. The procedure we followed for best <italic>K</italic> value identification, including the premise that we set, helped in the proper detection of the genetic group number in CG as cross-validation (CV) error changes suggested no obvious <italic>K</italic> value. The success of CV error changes in best <italic>K</italic> identification depends in part on the degree of differentiation between the populations under study, as quantified by Wright&#x2019;s fixation index, <italic>F</italic><sub>ST</sub> (<xref ref-type="bibr" rid="B3">Alexander and Lange, 2011</xref>). <italic>F</italic><sub>ST</sub> value for the CG6&#x2013;CG7 pair was the lowest one (<xref ref-type="table" rid="T3"><bold>Table&#xa0;3</bold></xref>) among CG genetic groups, and since these groups were the last ones to be differentiated under <italic>K</italic> = 7 (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figure S11</bold></xref>), further group identification would be a difficult task. The number of genetic groups in cacao collections varies from one study to another: <xref ref-type="bibr" rid="B12">Boza et&#xa0;al. (2013)</xref> described nine lineages in a Dominican Republic collection revealed by 14 SSR markers; <xref ref-type="bibr" rid="B76">Osorio-Guarin et&#xa0;al. (2018)</xref> identified four groups within 565 clones in a CORPIOCA collection from Colombia using 96 SNPs; and three clusters were found in 133 Vietnamese cacao cultivars studied by a combination of SSR and SNP markers (<xref ref-type="bibr" rid="B26">Everaert et&#xa0;al., 2020</xref>).</p>
<p>Amelonado is the predominant ancestry among CG (49.22%) and CF (57.73%) plants, mostly as hybrids with the other cacao ancestries detected: Criollo, Contamana, Iquitos, Nacional, and Nanay. The prevalence of hybrids in cacao germplasm has been described in the collection from the Dominican Republic (<xref ref-type="bibr" rid="B12">Boza et&#xa0;al., 2013</xref>), Jamaica (<xref ref-type="bibr" rid="B51">Lindo et&#xa0;al., 2018</xref>), China (<xref ref-type="bibr" rid="B90">Wang et&#xa0;al., 2020</xref>), Vietnam (<xref ref-type="bibr" rid="B26">Everaert et&#xa0;al., 2020</xref>), Uganda (<xref ref-type="bibr" rid="B33">Gopaulchan et&#xa0;al., 2019</xref>), Nigeria (<xref ref-type="bibr" rid="B72">Olasupo et&#xa0;al., 2018</xref>), and others. However, groups contributing to hybrids are different; for instance, Amelonado contributed the most to collections in the Dominican Republic (51.7%) and China (59%), while Mara&#xf1;on is the more common ancestry in germplasm from Jamaica (29.9%) and Uganda (61.5% of the trees had &#x2265; 80% Mara&#xf1;on lineage). Amelonado/Criollo, Amelonado/Nacional, and Amelonado/Criollo/Nacional hybrids excelled among the identified lineages in CG because of their putative connection to the natural hybrids Trinitario and Refractario (<xref ref-type="bibr" rid="B68">Motilal, 2018</xref>). Trinitario clones are recognized for their high productivity and high cocoa quality, and Refractario constitutes a source of resistance to witches&#x2019; broom disease, a plague not reported in Cuba (<xref ref-type="bibr" rid="B57">Mart&#xed;nez de la Parte and P&#xe9;rez Vicente, 2015</xref>) but detected in Caribbean islands close to Cuba (<xref ref-type="bibr" rid="B25">Evans, 2016</xref>; <xref ref-type="bibr" rid="B87">Ten Hoopen and Umaharan, 2017</xref>).</p>
<p>Cacao ancestry analysis in CF revealed Amelonado (CF1) and Amelonado/Criollo hybrid (CF2) plants as the most abundant (57.78%), which agrees with other in-farm cacao genetic diversity studies (<xref ref-type="bibr" rid="B12">Boza et&#xa0;al., 2013</xref>; <xref ref-type="bibr" rid="B38">Ji et&#xa0;al., 2013</xref>; <xref ref-type="bibr" rid="B21">Cosme et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B2">Adenet et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B34">Gopaulchan et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B50">Li et&#xa0;al., 2021</xref>) and the preference of farmers for planting grafted seedlings derived from putative Trinitario clones because of their better agronomic profile. Unfortunately, we did not detect pure Criollo plants in the CF plants as they were identified in farms from Honduras, Nicaragua, and Puerto Rico (<xref ref-type="bibr" rid="B38">Ji et&#xa0;al., 2013</xref>; <xref ref-type="bibr" rid="B21">Cosme et&#xa0;al., 2016</xref>). Another ancestry absent in Baracoa farm samples was Mara&#xf1;on, which has been identified in cacao fields in the Dominican Republic (<xref ref-type="bibr" rid="B12">Boza et&#xa0;al., 2013</xref>), Martinique (<xref ref-type="bibr" rid="B2">Adenet et&#xa0;al., 2020</xref>), Dominica (<xref ref-type="bibr" rid="B34">Gopaulchan et&#xa0;al., 2020</xref>), and Uganda (<xref ref-type="bibr" rid="B33">Gopaulchan et&#xa0;al., 2019</xref>).</p>
<p>
<xref ref-type="bibr" rid="B10">Bidot Mart&#xed;nez et&#xa0;al. (2015)</xref> studied the population structure of anciently introduced cacao plants in Cuba, also known as traditional Cuban cacao. These are unique plants remaining within cacao farms from the central and eastern regions of the country. Some of the plants studied carried higher Criollo and Mara&#xf1;on proportions than those found in samples from cacao commercial farms in the Baracoa region (CF). Considering that only one plant from the CG had an important contribution from Mara&#xf1;on ancestry, these plants could represent an opportunity to increase the genetic resources available in the cacao gene bank for the strengthening of the Cuban cacao genetic improvement program.</p>
<p>Clustering analysis by UPGMA of CG and CF samples either alone (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Figure S14, S15</bold></xref>) or combined with cacao reference plants (<xref ref-type="fig" rid="f5"><bold>Figures&#xa0;5</bold></xref>, <xref ref-type="fig" rid="f6"><bold>6</bold></xref>) showed mostly congruent results with clustering by ADMIXTURE except for the CG2 and CF4 groups. Some CG2 individuals were located in different clades of the dendrogram apart from the main CG2 branch. Such behavior is probably a consequence of the relaxed rules followed for group assignment and the presence of several hybrid combinations within this group. The clustering of CF4 samples in a single branch with their expected cacao controls was difficult to achieve. The difficulties of clustering analysis to group Nanay and Iquitos hybrids from CF4 samples with their respective cacao ancestry genetic group references, as occurred with CG6 and CG7 samples (<xref ref-type="fig" rid="f5"><bold>Figure&#xa0;5</bold></xref>), may be related to the lower membership coefficient of these CF individuals to Nanay (average <italic>Q</italic> = 0.4265) and Iquitos (0.4996), in comparison with the average memberships for Nanay (0.5767) and Iquitos (0.6665) of CG6 and CG67, respectively. The rest of the groups largely clustered together, with the expected cacao references confirming the results derived from ADMIXTURE software. On the other hand, PCA results of CG and CF plants showed adequate correspondence with ADMIXTURE and dendrogram results, though some CG genetic groups were not properly separated from each other in the PCA, but they were set apart from the rest of the groups.</p>
</sec>
<sec id="s4_3">
<title>Differentiation between groups of the Cuban cacao gene bank and cacao farms</title>
<p>In spite of the small inconsistencies in the different approaches employed to assess population structure in CG and CF plants, the variability between the identified genetic groups revealed by AMOVA results was good enough to differentiate them, since all <italic>F</italic><sub>ST</sub> values from pairwise comparison concluded significant differences (<italic>p</italic> = 0). The low contribution to variability between levels when farms and productive poles were evaluated as sources of variation in AMOVA with CF samples suggests the same planting policies are followed in different regions in Baracoa, which in turn obey a national list of 15 cacao clones and eight related hybrids to be used in cacao farming (<xref ref-type="bibr" rid="B65">MINAGRI, 2019</xref>).</p>
<p>Group differentiation between CF genetic groups was less noticeable than CG genetic groups. The highest differences among CG and CF groups were detected between CG1&#x2013;CG3 (0.407) and CF1&#x2013;CF3 (0.282). CG1 and CF1 groups had Amelonado ancestry, while CG3 and CF3 contained Contamana backgrounds. Counting plants carrying a high membership coefficient (<italic>Q</italic> &gt; 0.9) to any cacao ancestry revealed Amelonado (51) and Contamana (six, including two plants from CF3 with <italic>Q</italic> &gt; 0.80) as the ancestries with the highest count of almost pure individuals among all CG and CF plants. Genetic groups containing plants with high membership coefficients to well-differentiated genetic backgrounds are expected to be the ones with the highest differentiation. The lowest differentiation in CG was achieved between CG6 and CG7 groups (0.071), which are mainly formed by hybrids of Nanay and Iquitos, respectively. <xref ref-type="bibr" rid="B66">Motamayor et&#xa0;al. (2008)</xref> have already recognized difficulties in the proper separation of Nanay and Iquitos groups in reduced diversity scenarios probably because individuals from these groups had been collected in a relatively small geographical area. Our results confirm those difficulties even though a large genetic differentiation (<italic>F</italic><sub>ST</sub> = 0.367) was detected between the reference plants of these cacao ancestry genetic groups using the same SNPs (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table S6</bold></xref>). The fact that we are working with hybrids instead of pure individuals would contribute to the low differentiation detected between the CG6 and CG7 samples. <italic>F</italic><sub>ST</sub> values for the rest of the pairwise comparisons varied from 0.105 to 0.407 and supported a moderate to very large differentiation between the groups (<xref ref-type="bibr" rid="B31">Frankham et&#xa0;al., 2002</xref>; <xref ref-type="bibr" rid="B84">Shu et&#xa0;al., 2021</xref>).</p>
</sec>
<sec id="s4_4">
<title>Genetic diversity of the Cuban cacao gene bank and cacao farms</title>
<p>Genetic diversity statistics of CG plants showed moderate to low values of observed (<italic>H</italic><sub>obs</sub> = 0.264) and expected (<italic>H</italic><sub>exp</sub> = 0.283) heterozygosities and PIC (0.235), similar to the ones obtained for CF plants (0.296, 0.286, 0.236). Observed and expected heterozygosities from other cacao gene banks, estimated with SNP data, mostly had higher values. That is the case of Yunnan collection from China (<italic>H</italic><sub>obs</sub> = 0.361, <italic>H</italic><sub>exp</sub> = 0.306) (<xref ref-type="bibr" rid="B90">Wang et&#xa0;al. (2020)</xref>, CORPIOCA collection in Colombia (0.353, 0.314) (<xref ref-type="bibr" rid="B74">Osorio-Guarin et&#xa0;al., 2017</xref>), germplasm bank of Tenguel-Guayas, Ecuador (0.479, 0.378) (<xref ref-type="bibr" rid="B16">Carranza et&#xa0;al., 2020</xref>), CRIG germplasm collection of Ghana (0.274, 0.343) (<xref ref-type="bibr" rid="B86">Takrama et&#xa0;al., 2014</xref>) and Uganda (0.304, 0.322) (<xref ref-type="bibr" rid="B33">Gopaulchan et&#xa0;al., 2019</xref>). Only the Jamaican collection had a lower <italic>H</italic><sub>exp</sub> (0.240), while <italic>H</italic><sub>obs</sub> (0.280) was again higher (<xref ref-type="bibr" rid="B51">Lindo et&#xa0;al., 2018</xref>). These cacao plants were genotyped with SNP markers selected for cacao classification. The selection was based on several SNP properties, such as level of polymorphism, distribution across the 10 cacao chromosomes, and SNP capacity to properly distinguish reference clones of cacao ancestry genetic groups (<xref ref-type="bibr" rid="B38">Ji et&#xa0;al., 2013</xref>; <xref ref-type="bibr" rid="B28">Fang et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B69">Motilal et&#xa0;al., 2017</xref>). Variant filtration based on polymorphic information was not applied to our SNP datasets. Therefore, lower genetic diversity statistics values should be expected on a per-locus basis, resulting from the random combination of low and high polymorphic SNP markers.</p>
<p>The within-group genetic diversity was also estimated among both CG and CF genetic groups. CG1 and CF1 had the lowest values for all genetic diversity parameters, while CG2 and CF2 showed the highest <italic>H</italic><sub>obs</sub> values. The lowest values of genetic diversity of CG1 and CF1 are consistent with the prevalence of Amelonado ancestry within this group. Amelonado and Criollo clones are recognized by their highly homozygous genomes and self-fertilization (<xref ref-type="bibr" rid="B8">Argout et&#xa0;al., 2011</xref>; <xref ref-type="bibr" rid="B67">Motamayor et&#xa0;al., 2013</xref>), justifying the low genetic diversity found in these groups. On the contrary, CG2 and CF2 excelled in the occurrence of Amelonado/Criollo hybrids and in the case of CG2 of other hybrid combinations (<xref ref-type="fig" rid="f3"><bold>Figures&#xa0;3</bold></xref>, <xref ref-type="fig" rid="f4"><bold>4</bold></xref>). The crossing of highly homozygous Amelonado and Criollo plants should produce highly heterozygous Amelonado/Criollo hybrids, which, combined with the presence of other hybrids, could lead to the high values of CG2 and CF2 genetic diversity parameters, especially <italic>H</italic><sub>obs</sub>. A high similarity was observed in CG6 and CG7 genetic diversity parameters. These results are consistent with low differentiation detected between these groups, as already discussed.</p>
<p>The genetic diversity of CG4 revealed some distinctions among the groups. This is a reduced, quite homogenous group of cacao plants with a unique mixture of cacao ancestries. Nine of its 10 individuals had a 0.9999 membership to this group (the last one was 0.8432), and cacao genetic group ancestries are split into Amelonado (0.4643), Iquitos (0.3689), Contamana (0.1355), and Criollo (0.0302). Genetic parameters of CG4 confirmed the observed singularities since the low <italic>H</italic><sub>exp</sub> and PIC obtained agree with the homogeneity of the individuals, and the cacao ancestries mix detected supports the excess of heterozygosity identified (<italic>H</italic><sub>obs</sub> &gt; <italic>H</italic><sub>exp</sub>).</p>
</sec>
<sec id="s4_5">
<title>Final considerations</title>
<p>These results constitute the first attempt to use SNP markers in the assessment of the genetic diversity of cacao resources in the Baracoa region, which is responsible for most of the cacao production in Cuba. The cacao ancestry genetic group distributions among the CG and CF plants confirmed the poor utilization of diverse genetic groups in Cuban cacao farming, as also described in cacao agronomical practices worldwide (<xref ref-type="bibr" rid="B96">Zhang and Motilal, 2016</xref>; <xref ref-type="bibr" rid="B20">Cornejo et&#xa0;al., 2018</xref>). More cacao ancestries and ancestry combinations were detected among cacao gene bank accessions than in plants from commercial cacao farms of Baracoa. Therefore, there is a chance to increase the ancestries exploited in cacao production and, hence, the genetic diversity of the in-farm cacao resources using local cacao resources. To this end, cacao clones from the gene bank with ancestries different from those currently exploited in Baracoa cacao productive areas should be introduced into cacao farming practices. As an example, the results indicated a low representation of Nacional hybrids in cacao farms, but several clones of this type were detected in the germplasm collection. Refractario clones, which carry Nacional ancestry as hybrid clones, are known for their putative resistance to <italic>Moniliophthora perniciosa</italic> (witches&#x2019; broom disease), a disease absent in Cuba but present in Jamaica and the Dominican Republic. The introduction of cacao plants with a Nacional background in the productive areas of Baracoa may contribute to face future sanitary contingencies like the intrusion of witches&#x2019; broom disease from neighboring Caribbean Islands (<xref ref-type="bibr" rid="B25">Evans, 2016</xref>; <xref ref-type="bibr" rid="B87">Ten Hoopen and Umaharan, 2017</xref>).</p>
<p>Pure cacao plants belonging to the groups Amelonado, Contamana, Iquitos, Nanay, and Mara&#xf1;on were identified in the collection, but only one individual per group of the last two ancestries was found with <italic>Q</italic> &gt; 0.9. Efforts should be made to increase the number of low-represented or absent ancestries such as Curaray, Guiana, and Pur&#xfa;s in the Cuban national gene bank. Equally important would be the incorporation of pure individuals from the groups Criollo and Nacional, though several hybrids of these groups with Amelonado were identified. However, the population structure of the cacao collection accessions here described will contribute to the strengthening of the ongoing cacao improvement program by providing new approaches and revitalizing poorly exploited cacao clones present in the collection with attractive genetic backgrounds.</p>
<p>The high quality of the SNP datasets obtained, the genome widespread distributions of the variant sites, and the congruence among the results here presented to validate the use of the ddRADseq protocol described for genetic studies in <italic>Theobroma cacao</italic>. The exploitation of these datasets to conduct association studies could contribute to the identification of new genes and genome regions related to morpho-agronomic properties relevant to the cacao production. Particularly interesting in using RADseq in cacao is the potential identification of extrachromosomal SNPs (<xref ref-type="bibr" rid="B47">Laczk&#xf3; et&#xa0;al., 2022</xref>). In our case, an average of 93% of the reads mapped to the 10 cacao chromosomes, leaving room for mitochondrial and plastid SNP identification. These aspects are still to be analyzed and could provide new ways for genetic studies in <italic>Theobroma cacao</italic>.</p>
</sec>
</sec>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: <uri xlink:href="https://www.ebi.ac.uk/eva/">https://www.ebi.ac.uk/eva/</uri>, PRJEB71753.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>AR-R: Data curation, Formal analysis, Investigation, Methodology, Software, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. KM: Conceptualization, Investigation, Methodology, Writing &#x2013; review &amp; editing. MM-G: Investigation, Writing &#x2013; review &amp; editing. PC-: Investigation, Writing&#xa0;&#x2013; review &amp; editing. GE-L: Methodology, Writing &#x2013; review &amp; editing. IB-M: Conceptualization, Funding acquisition, Investigation, Project administration, Resources, Writing &#x2013; review &amp; editing. PB: Conceptualization, Funding acquisition, Investigation, Methodology, Project administration, Resources, Supervision, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing.</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work was funded by the &#x201c;Projet de Recherche et D&#xe9;veloppement&#x201d; of the Cooperation program from the &#x201c;Acad&#xe9;mie de Recherche et d&#x2019;Enseignement Sup&#xe9;rieur&#x201d; (ARES), Belgium (ARES CCD Programme PRD-PFS 2017: Cuba-Cacao). Funds from the Oficina de Gesti&#xf3;n de Fondos y Proyectos Internacionales (OGFPI) were also provided (Code: PN223LH010-022) to cover local expenditures in Cuba.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>The use of several bioinformatic software programs in this study was possible because of the computational resources provided by C&#xc9;CI clusters, the &#x201c;Consortium des Equipements de Calcul Intensif&#x201d; of Wallonia region, Belgium. We also want to thank Administration des relations internationals (ADRI) personnel at UCLouvain for the support provided during the development of the ARES-CCD Project on Cuban cacao, especially for the warm welcome of Cuban fellows at UCLouvain.</p>
</ack>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2024.1367632/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2024.1367632/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet_1.pdf" id="SM1" mimetype="application/pdf"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Achrem</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Szucko</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Kalinka</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>The epigenetic regulation of centromeres and telomeres in plants and animals</article-title>. <source>Comp. Cytogenet.</source> <volume>14</volume>, <fpage>265</fpage>&#x2013;<lpage>311</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3897/CompCytogen.v14i2.51895</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adenet</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Regina</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Rogers</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Bharath</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Argout</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Rochefort</surname> <given-names>K.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Study of the genetic diversity of cocoa populations (<italic>Theobroma cacao</italic> L.) of Martinique (FWI) and potential for processing and the cocoa industry</article-title>. <source>Gen. Resour. Crop Evol.</source> <volume>67</volume>, <fpage>1969</fpage>&#x2013;<lpage>1979</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10722-020-00953-0</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alexander</surname> <given-names>D. H.</given-names>
</name>
<name>
<surname>Lange</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Enhancements to the ADMIXTURE algorithm for individual ancestry estimation</article-title>. <source>BMC Bioinform.</source> <volume>12</volume>, <elocation-id>246</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/1471-2105-12-246</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alexander</surname> <given-names>D. H.</given-names>
</name>
<name>
<surname>Novembre</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lange</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Fast model-based estimation of ancestry in unrelated individuals</article-title>. <source>Genome Res.</source> <volume>19</volume>, <fpage>1655</fpage>&#x2013;<lpage>1664</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/gr.094052.109</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andrews</surname> <given-names>K. R.</given-names>
</name>
<name>
<surname>Good</surname> <given-names>J. M.</given-names>
</name>
<name>
<surname>Miller</surname> <given-names>M. R.</given-names>
</name>
<name>
<surname>Luikart</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Hohenlohe</surname> <given-names>P. A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Harnessing the power of RADseq for ecological and evolutionary genomics</article-title>. <source>Nat. Rev. Genet.</source> <volume>17</volume>, <fpage>81</fpage>&#x2013;<lpage>92</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nrg.2015.28</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arevalo-Gardini</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Meinhardt</surname> <given-names>L. W.</given-names>
</name>
<name>
<surname>Zu&#xf1;iga</surname> <given-names>L. C.</given-names>
</name>
<name>
<surname>Ar&#xe9;valo-Gardni</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Motilal</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>D</given-names>
</name>
</person-group>. (<year>2019</year>). <article-title>Genetic identity and origin of &#x201c;Piura Porcelana&#x201d;&#x2014;a fine-flavored traditional variety of cacao (<italic>Theobroma cacao</italic>) from the Peruvian Amazon. <italic>Tree Genet</italic>
</article-title>. <source>Genomes</source> <volume>15</volume>, <elocation-id>11</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11295-019-1316-y</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Argout</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Fouet</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Wincker</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Gramacho</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Legavre</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Sabau</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2008</year>). <article-title>Towards the understanding of the cocoa transcriptome: Production and analysis of an exhaustive dataset of ESTs of <italic>Theobroma cacao</italic> L. generated from various tissues and under various conditions</article-title>. <source>BMC Genom.</source> <volume>9</volume>, <elocation-id>512</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/1471-2164-9-512</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Argout</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Salse</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Aury</surname> <given-names>J. M.</given-names>
</name>
<name>
<surname>Guiltinan</surname> <given-names>M. J.</given-names>
</name>
<name>
<surname>Droc</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Gouzy</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2011</year>). <article-title>The genome of <italic>Theobroma cacao</italic>
</article-title>. <source>Nat. Genet.</source> <volume>43</volume>, <fpage>101</fpage>&#x2013;<lpage>108</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/ng.736</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Badrie</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Bekele</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Sikora</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Sikora</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Cocoa agronomy, quality, nutritional, and health aspects</article-title>. <source>Crit. Rev. Food Sci. Nutr.</source> <volume>55</volume>, <fpage>620</fpage>&#x2013;<lpage>659</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/10408398.2012.669428</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bidot Mart&#xed;nez</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Riera Nelson</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Flamand</surname> <given-names>M.-C.</given-names>
</name>
<name>
<surname>Bertin</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Genetic diversity and population structure of anciently introduced Cuban cacao <italic>Theobroma cacao</italic> plants</article-title>. <source>Gen. Resour. Crop Evol.</source> <volume>62</volume>, <fpage>67</fpage>&#x2013;<lpage>84</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10722-014-0136-z</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bidot Mart&#xed;nez</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Vald&#xe9;s de la Cruz</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Riera Nelson</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Bertin</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Morphological characterization of traditional cacao (<italic>Theobroma cacao</italic> L.) plants in Cuba</article-title>. <source>Gen. Resour. Crop Evol.</source> <volume>64</volume>, <fpage>73</fpage>&#x2013;<lpage>99</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10722-015-0333-4</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boza</surname> <given-names>E. J.</given-names>
</name>
<name>
<surname>Irish</surname> <given-names>B. M.</given-names>
</name>
<name>
<surname>Meerow</surname> <given-names>A. W.</given-names>
</name>
<name>
<surname>Tondo</surname> <given-names>C. L.</given-names>
</name>
<name>
<surname>Rodr&#xed;guez</surname> <given-names>O. A.</given-names>
</name>
<name>
<surname>Ventura-L&#xf3;pez</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2013</year>). <article-title>Genetic diversity, conservation, and utilization of <italic>Theobroma cacao</italic> L.: genetic resources in the Dominican Republic</article-title>. <source>Gen. Resour. Crop Evol.</source> <volume>60</volume>, <fpage>605</fpage>&#x2013;<lpage>619</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10722-012-9860-4</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="web">
<person-group person-group-type="author">
<collab>Broad Institute</collab>
</person-group> (<year>2016</year>) <article-title>PicardTools: A set of command line tools (in Java) for manipulating high-throughput sequencing (HTS) data and formats such as SAM/BAM/CRAM and VCF</article-title>. Available online at: <uri xlink:href="https://broadinstitute.github.io/picard/">https://broadinstitute.github.io/picard/</uri> (Accessed <access-date>December 10, 2021</access-date>).</citation>
</ref>
<ref id="B14">
<citation citation-type="web">
<person-group person-group-type="author">
<collab>CacaoNet</collab>
</person-group> (<year>2022</year>) <article-title>A global strategy for the conservation and use of cacao genetic resources, as the foundation for a sustainable cocoa economy. Global network for cacao genetic resources</article-title>. Available online at: <uri xlink:href="https://www.cacaonet.org/global-strategy/abstract">https://www.cacaonet.org/global-strategy/abstract</uri> (Accessed <access-date>December 10, 2022</access-date>).</citation>
</ref>
<ref id="B15">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Caetano-Anolles</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2022</year>) <article-title>Hard-filtering germline short variants. Genome Analysis Tool kit, GATK</article-title>. Available online at: <uri xlink:href="https://gatk.broadinstitute.org/hc/en-us/articles/360035890471-Hard-filtering-germline-short-variants">https://gatk.broadinstitute.org/hc/en-us/articles/360035890471-Hard-filtering-germline-short-variants</uri> (Accessed <access-date>May 15, 2022</access-date>).</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carranza</surname> <given-names>M. S.</given-names>
</name>
<name>
<surname>Zapata</surname> <given-names>Y. P.</given-names>
</name>
<name>
<surname>Gallego</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Rodr&#xed;guez</surname> <given-names>J. N.</given-names>
</name>
<name>
<surname>Morante Carriel</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Cruz Rosero</surname> <given-names>N.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Genetic diversity of Ecuadorian cocoa from the germplasm bank of Teh&#xec;nguel-Guyas Ecuador based in SNPP&#x2019;S</article-title>. <source>Bioagro</source> <volume>32</volume>, <fpage>75</fpage>&#x2013;<lpage>86</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Catchen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hohenlohe</surname> <given-names>P. A.</given-names>
</name>
<name>
<surname>Bassham</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Amores</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Cresko</surname> <given-names>W. A.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Stacks: an analysis tool set for population genomics</article-title>. <source>Mol. Ecol.</source> <volume>22</volume>, <fpage>3124</fpage>&#x2013;<lpage>3140</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/mec.12354</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ding</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Protein kinases in plant responses to drought, salt, and cold stress</article-title>. <source>J. Integr. Plant Biol.</source> <volume>63</volume>, <fpage>53</fpage>&#x2013;<lpage>78</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/jipb.13061</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cingolani</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Platts</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L. L.</given-names>
</name>
<name>
<surname>Coon</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Nguyen</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2012</year>). <article-title>A program for annotating and predicting the effects of single nucleotide polymorphisms, SnpEff</article-title>. <source>Fly</source> <volume>6</volume>, <fpage>80</fpage>&#x2013;<lpage>92</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.4161/fly.19695</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cornejo</surname> <given-names>O. E.</given-names>
</name>
<name>
<surname>Yee</surname> <given-names>M. C.</given-names>
</name>
<name>
<surname>Dominguez</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Andrews</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Sockell</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Strandberg</surname> <given-names>E.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>Population genomic analyses of the chocolate tree, <italic>Theobroma cacao</italic> L., provide insights into its domestication process</article-title>. <source>Commun. Biol.</source> <volume>1</volume>, <fpage>167</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s42003-018-0168-6</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cosme</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Cuevas</surname> <given-names>H. E.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Oleksyk</surname> <given-names>T. K.</given-names>
</name>
<name>
<surname>Irish</surname> <given-names>B. M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Genetic diversity of naturalized cacao (<italic>Theobroma cacao</italic> L.) in Puerto Rico</article-title>. <source>Tree Genet. Genomes</source> <volume>12</volume>, <fpage>88</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11295-016-1045-4</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Danecek</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Auton</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Abecasis</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Albers</surname> <given-names>C. A.</given-names>
</name>
<name>
<surname>Banks</surname> <given-names>E.</given-names>
</name>
<name>
<surname>DePristo</surname> <given-names>M. A.</given-names>
</name>
<etal/>
</person-group>. (<year>2011</year>). <article-title>The variant call format and VCFtools</article-title>. <source>Bioinform</source> <volume>27</volume>, <fpage>2156</fpage>&#x2013;<lpage>2158</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btr330</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>dos Santos Menezes</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Mucherino-Mu&#xf1;oz</surname> <given-names>J. J.</given-names>
</name>
<name>
<surname>Ferreira</surname> <given-names>C. A.</given-names>
</name>
<name>
<surname>da Silva Chaves</surname> <given-names>S. F.</given-names>
</name>
<name>
<surname>Barbosa</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Lemos</surname> <given-names>L. S. L.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). &#x201c;<article-title>Genomic designing for biotic stress resistant cocoa tree</article-title>,&#x201d; in <source>Genomic designing for biotic stress resistant technical crops</source>. Ed. <person-group person-group-type="editor">
<name>
<surname>Kole</surname> <given-names>C.</given-names>
</name>
</person-group> (<publisher-name>Springer Nature</publisher-name>, <publisher-loc>Switzerland</publisher-loc>), <fpage>49</fpage>&#x2013;<lpage>113</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dray</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Dufour</surname> <given-names>A. B.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>The ade4 package: implementing the duality diagram for ecologists</article-title>. <source>J. Stat. Software</source> <volume>22</volume>, <fpage>1</fpage>&#x2013;<lpage>20</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.18637/jss.v022.i04</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Evans</surname> <given-names>H. C.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Witches&#x2019; Broom disease (<italic>Moniliophthora perniciosa</italic>): history and biology</article-title>,&#x201d; in <source>Cacao diseases</source>. Eds. <person-group person-group-type="editor">
<name>
<surname>Bailey</surname> <given-names>B. A.</given-names>
</name>
<name>
<surname>Meinhardt</surname> <given-names>L. W.</given-names>
</name>
</person-group> (<publisher-name>Springer International Publishing</publisher-name>, <publisher-loc>Switzerland</publisher-loc>), <fpage>137</fpage>&#x2013;<lpage>177</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Everaert</surname> <given-names>H.</given-names>
</name>
<name>
<surname>De Wever</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>T. K. H.</given-names>
</name>
<name>
<surname>Vu</surname> <given-names>T. L. A.</given-names>
</name>
<name>
<surname>Maebe</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Rottiers</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Genetic classification of Vietnamese cacao cultivars assessed by SNP and SSR markers</article-title>. <source>Tree Genet. Genomes.</source> <volume>16</volume>, <fpage>43</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11295-020-01439-x</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Excoffier</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Smouse Pe Fau-Quattro</surname> <given-names>J. M.</given-names>
</name>
<name>
<surname>Quattro</surname> <given-names>J. M.</given-names>
</name>
</person-group> (<year>1992</year>). <article-title>Analysis of molecular variance inferred from metric distances among DNA haplotypes: application to human mitochondrial DNA restriction data</article-title>. <source>Genetics.</source> <volume>131</volume>, <fpage>479</fpage>&#x2013;<lpage>491</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/genetics/131.2.479</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Meinhardt</surname> <given-names>L. W.</given-names>
</name>
<name>
<surname>Mischke</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Bellato</surname> <given-names>C. M.</given-names>
</name>
<name>
<surname>Motilal</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Accurate determination of genetic identity for a single cacao bean, using molecular markers with a nanofluidic system, ensures cocoa authentication</article-title>. <source>J. Agric. Food Chem.</source> <volume>62</volume>, <fpage>481</fpage>&#x2013;<lpage>487</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/jf404402v</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="web">
<person-group person-group-type="author">
<collab>FAOSTAT</collab>
</person-group> (<year>2023</year>) <article-title>Food and agriculture organizations of the united nations. Statictic division</article-title>. Available online at: <uri xlink:href="https://www.fao.org/faostat/en/#data/QCL/visualize">https://www.fao.org/faostat/en/#data/QCL/visualize</uri> (Accessed <access-date>October 12, 2023</access-date>).</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Figueira</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Janick</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Levy</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Goldsbrough</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>Reexamining the classification of <italic>Theobroma cacao</italic> L. using molecular markers</article-title>. <source>J. Am. Soc Hortic. Sci.</source> <volume>119</volume>, <fpage>1073</fpage>&#x2013;<lpage>1082</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.21273/JASHS.119.5.1078</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Frankham</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Ballou</surname> <given-names>J. D.</given-names>
</name>
<name>
<surname>Briscoe</surname> <given-names>D. A.</given-names>
</name>
</person-group> (<year>2002</year>). <source>Introduction to conservation genetics</source> (<publisher-loc>Cambridge</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>), <fpage>pp 641</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1017/CBO9780511808999</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<collab>GO Consortium C</collab>
</person-group> (<year>2017</year>). <article-title>Expansion of the Gene Ontology knowledgebase and resources</article-title>. <source>Nucleic Acids Res.</source> <volume>45</volume>, <fpage>D331</fpage>&#x2013;<lpage>D338</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/nar/gkw1108</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gopaulchan</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Motilal</surname> <given-names>L. A.</given-names>
</name>
<name>
<surname>Bekele</surname> <given-names>F. L.</given-names>
</name>
<name>
<surname>Clause</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ariko</surname> <given-names>J. O.</given-names>
</name>
<name>
<surname>Ejang</surname> <given-names>H. P.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Morphological and genetic diversity of cacao (<italic>Theobroma cacao</italic> L.) in Uganda</article-title>. <source>Physiol. Mol. Biol. Plants</source> <volume>25</volume>, <fpage>361</fpage>&#x2013;<lpage>375</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s12298-018-0632-2</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gopaulchan</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Motilal</surname> <given-names>L. A.</given-names>
</name>
<name>
<surname>Kalloo</surname> <given-names>R. K.</given-names>
</name>
<name>
<surname>Mahabir</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Moses</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Joseph</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Genetic diversity and ancestry of cacao (<italic>Theobroma cacao</italic> L.) in Dominica revealed by single nucleotide polymorphism markers</article-title>. <source>Genome</source> <volume>63</volume>, <fpage>583</fpage>&#x2013;<lpage>595</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1139/gen-2019-0214</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gruber</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Unmack</surname> <given-names>P. J.</given-names>
</name>
<name>
<surname>Berry</surname> <given-names>O. F.</given-names>
</name>
<name>
<surname>Georges</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>dartr: An r package to facilitate analysis of SNP data generated from reduced representation genome sequencing</article-title>. <source>Mol. Ecol. Resour.</source> <volume>18</volume>, <fpage>691</fpage>&#x2013;<lpage>699</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/1755-0998.12745</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Sheng</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Clark</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Samuels</surname> <given-names>D. C.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Three-stage quality control strategies for DNA re-sequencing data</article-title>. <source>Brief. Bioinf.</source> <volume>15</volume>, <fpage>879</fpage>&#x2013;<lpage>889</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bib/bbt069</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guti&#xe9;rrez</surname> <given-names>O. A.</given-names>
</name>
<name>
<surname>Martinez</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Livingstone</surname> <given-names>D. S.</given-names>
</name>
<name>
<surname>Turnbull</surname> <given-names>C. J.</given-names>
</name>
<name>
<surname>Motamayor</surname> <given-names>J. C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Selecting SNP markers reflecting population origin for cacao (<italic>Theobroma cacao</italic> L.) germplasm identification</article-title>. <source>Beverage Plant Res.</source> <volume>1</volume>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48130/bpr-2021-0015</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ji</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Motilal</surname> <given-names>L. A.</given-names>
</name>
<name>
<surname>Boccara</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Lachenaud</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Meinhardt</surname> <given-names>L. W.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Genetic diversity and parentage in farmer varieties of cacao (<italic>Theobroma cacao</italic> L.) from Honduras and Nicaragua as revealed by single nucleotide polymorphism (SNP) markers</article-title>. <source>Gen. Resour. Crop Evol.</source> <volume>60</volume>, <fpage>441</fpage>&#x2013;<lpage>453</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10722-012-9847-1</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jombart</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>adegenet: a R package for the multivariate analysis of genetic markers</article-title>. <source>Bioinform</source> <volume>24</volume>, <fpage>1403</fpage>&#x2013;<lpage>1405</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btn129</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jombart</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Ahmed</surname> <given-names>I.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>adegenet 1.3-1: new tools for the analysis of genome-wide SNP data</article-title>. <source>Bioinform</source> <volume>27</volume>, <fpage>3070</fpage>&#x2013;<lpage>3071</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btr521</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kamvar</surname> <given-names>Z. N.</given-names>
</name>
<name>
<surname>Brooks</surname> <given-names>J. C.</given-names>
</name>
<name>
<surname>Grunwald</surname> <given-names>N. J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Novel R tools for analysis of genome-wide population genetic data with emphasis on clonality</article-title>. <source>Front. Genet.</source> <volume>6</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fgene.2015.00208</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kamvar</surname> <given-names>Z. N.</given-names>
</name>
<name>
<surname>Tabima</surname> <given-names>J. F.</given-names>
</name>
<name>
<surname>Grunwald</surname> <given-names>N. J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Poppr: an R package for genetic analysis of populations with clonal, partially clonal, and/or sexual reproduction</article-title>. <source>PeerJ</source> <volume>2</volume>, <elocation-id>e281</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.7717/peerj.281</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kopelman</surname> <given-names>N. M.</given-names>
</name>
<name>
<surname>Mayzel</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jakobsson</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Rosenberg</surname> <given-names>N. A.</given-names>
</name>
<name>
<surname>Mayrose</surname> <given-names>I.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Clumpak: a program for identifying clustering modes and packaging population structure inferences across K</article-title>. <source>Mol. Ecol. Resour.</source> <volume>15</volume>, <fpage>1179</fpage>&#x2013;<lpage>1191</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/1755-0998.12387</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Krueger</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2017</year>) <article-title>Trim Galore: a wrapper script to automate quality and adapter trimming</article-title>. Available online at: <uri xlink:href="http://www.bioinformatics.babraham.ac.uk/projects/trim_galore">http://www.bioinformatics.babraham.ac.uk/projects/trim_galore</uri> (Accessed <access-date>December 5, 2021</access-date>).</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Chhokar</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Sheoran</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Sharma</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Jaiswal</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Characterization of genetic diversity and population structure in wheat using array-based SNP markers</article-title>. <source>Mol. Biol. Rep.</source> <volume>47</volume>, <fpage>293</fpage>&#x2013;<lpage>306</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11033-019-05132-8</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lachenaud</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Cl&#xe9;ment</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Argout</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Scalabrin</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Doar&#xe9;</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>The Guiana cacao genetic group (<italic>Theobroma cacao</italic> L.): a new core collection in French Guiana</article-title>. <source>Bot. Lett.</source> <volume>165</volume>, <fpage>248</fpage>&#x2013;<lpage>254</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/23818107.2018.1465466</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laczk&#xf3;</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Jord&#xe1;n</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sramk&#xf3;</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>The RadOrgMiner pipeline: Automated genotyping of organellar loci from RADseq data</article-title>. <source>Methods Ecol. Evol.</source> <volume>13</volume>, <fpage>1962</fpage>&#x2013;<lpage>1975</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/2041-210X.13937</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Durbin</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Fast and accurate short read alignment with Burrows-Wheeler transform</article-title>. <source>Bioinform</source> <volume>25</volume>, <fpage>1754</fpage>&#x2013;<lpage>1760</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btp324</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Handsaker</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Wysoker</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Fennell</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Ruan</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Homer</surname> <given-names>N.</given-names>
</name>
<etal/>
</person-group>. (<year>2009</year>). <article-title>The sequence alignment/map format and SAMtools</article-title>. <source>Bioinform</source> <volume>25</volume>, <fpage>2078</fpage>&#x2013;<lpage>2079</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bioinformatics/btp352</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Motilal</surname> <given-names>L. A.</given-names>
</name>
<name>
<surname>Lachenaud</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Mischke</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Meinhardt</surname> <given-names>L. W.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Traditional varieties of cacao (<italic>Theobroma cacao</italic>) in Madagascar: their origin and dispersal revealed by SNP markers</article-title>. <source>Beverage Plant Res.</source> <volume>1</volume>, <elocation-id>4</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.48130/BPR-2021-0004</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lindo</surname> <given-names>A. A.</given-names>
</name>
<name>
<surname>Robinson</surname> <given-names>D. E.</given-names>
</name>
<name>
<surname>Tennant</surname> <given-names>P. F.</given-names>
</name>
<name>
<surname>Meinhardt</surname> <given-names>L. W.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Molecular characterization of cacao (<italic>Theobroma cacao</italic>) germplasm from Jamaica using single nucleotide polymorphism (SNP) markers</article-title>. <source>Trop. Plant Biol.</source> <volume>11</volume>, <fpage>93</fpage>&#x2013;<lpage>106</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s12042-018-9203-5</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>C. C.</given-names>
</name>
<name>
<surname>Shringarpure</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Lange</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Novembre</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Exploring population structure with admixture models and principal component analysis</article-title>,&#x201d; in <source>Methods in molecular biology: statistical population genomics</source>, vol. <volume>2090</volume> . Ed. <person-group person-group-type="editor">
<name>
<surname>Dutheil</surname> <given-names>J. Y.</given-names>
</name>
</person-group> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Humana</publisher-name>), <fpage>67</fpage>&#x2013;<lpage>86</lpage>.</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lloyd</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Crossover patterning in plants</article-title>. <source>Plant Reprod</source> <volume>36</volume>, <fpage>55</fpage>&#x2013;<lpage>72</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00497-022-00445-4</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lukman, Zhang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Susilo</surname> <given-names>A. W.</given-names>
</name>
<name>
<surname>Dinarti</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Bailey</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Mischke</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2014</year>). <article-title>Genetic identity, ancestry and parentage in farmer selections of cacao from aceh, Indonesia revealed by single nucleotide polymorphism (SNP) markers</article-title>. <source>Trop. Plant Biol.</source> <volume>7</volume>, <fpage>133</fpage>&#x2013;<lpage>143</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s12042-014-9144-6</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mahabir</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Motilal</surname> <given-names>L. A.</given-names>
</name>
<name>
<surname>Gopaulchan</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Ramkissoon</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sankar</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Umaharan</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Development of a core SNP panel for cacao (<italic>Theobroma cacao</italic> L.) identity analysis</article-title>. <source>Genome</source> <volume>63</volume>, <fpage>103</fpage>&#x2013;<lpage>114</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1139/gen-2019-0071</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>M&#xe1;rquez-Rivero</surname> <given-names>J. J.</given-names>
</name>
<name>
<surname>Aguirre-G&#xf3;mez</surname> <given-names>M. B.</given-names>
</name>
</person-group> (<year>2008</year>). <source>Manual t&#xe9;cnico de manejo agrot&#xe9;cnico de las plantaciones de cacao</source> (<publisher-loc>CIDISAV</publisher-loc>: <publisher-name>Ciudad de la Habana, Cuba</publisher-name>).</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mart&#xed;nez de la Parte</surname> <given-names>E.</given-names>
</name>
<name>
<surname>P&#xe9;rez Vicente</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Incidencia de enfermedades f&#xfa;ngicas en plantaciones de cacao de las provincias orientales de Cuba</article-title>. <source>Rev. Protecci&#xf3;n Veg.</source> <volume>30</volume>, <fpage>87</fpage>&#x2013;<lpage>96</lpage>.</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mart&#xed;nez-Su&#xe1;rez</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Men&#xe9;ndez-Grenot</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Varela-Nualles</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Moya-L&#xf3;pez</surname> <given-names>C. C.</given-names>
</name>
<name>
<surname>Hern&#xe1;ndez</surname> <given-names>J&#xc1;.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Estabilidad de la producci&#xf3;n de cultivares h&#xed;bridos de cacao en la regi&#xf3;n de Baracoa</article-title>. <source>Caf&#xe9; y Cacao</source> <volume>15</volume>, <fpage>3</fpage>&#x2013;<lpage>10</lpage>.</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Matos-Cueto</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Clap&#xe9;-Borges</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Nari&#xf1;o-Nari&#xf1;o</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Resistencia a <italic>Phytophthora palmivora</italic> de 48 accesiones de cacao del Banco de Germoplasma de la Estaci&#xf3;n Experimental AgroForestal Baracoa, Cuba</article-title>. <source>Caf&#xe9; y Cacao</source> <volume>15</volume>, <fpage>28</fpage>&#x2013;<lpage>32</lpage>.</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Men&#xe9;ndez-Grenot</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Clap&#xe9;-Borges</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Lambertt-Lobaina</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Rodr&#xed;guez-Terrero</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Nari&#xf1;o-Nari&#xf1;o</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Caracterizaci&#xf3;n y an&#xe1;lisis mofoagron&#xf3;mico de 74 genotipos de <italic>Theobroma cacao</italic> Lin. para mejorar la estructura clonal del cultivo en Cuba</article-title>. <source>Caf&#xe9; y Cacao</source> <volume>13</volume>, <fpage>3</fpage>&#x2013;<lpage>11</lpage>.</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Men&#xe9;ndez-Grenot</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Rodr&#xed;guez-Terrero</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Clap&#xe9;-Borges</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Evaluaci&#xf3;n agron&#xf3;mica y de calidad de genotipos introducidos y prospectados de alto potencial productivo</article-title>. <source>Caf&#xe9; y Cacao</source> <volume>11</volume>, <fpage>25</fpage>&#x2013;<lpage>28</lpage>.</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Men&#xe9;ndez-Grenot</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Rodr&#xed;guez-Terrero</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Mart&#xed;nez-Su&#xe1;rez</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Selecci&#xf3;n de h&#xed;bridos avanzados F1 que mejoren las actuales estructuras clonales en Cuba</article-title>. <source>Caf&#xe9; y Cacao</source> <volume>15</volume>, <fpage>15</fpage>&#x2013;<lpage>21</lpage>.</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mi</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Muruganujan</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ebert</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Mills</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Protocol Update for large-scale genome and gene function analysis with the PANTHER classification system (v.14.0)</article-title>. <source>Nat. Protoc.</source> <volume>14</volume>, <fpage>703</fpage>&#x2013;<lpage>721</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41596-019-0128-8</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mijangos</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Gruber</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Berry</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Pacioni</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Georges</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>dartR v2: An accessible genetic analysis platform for conservation, ecology and agriculture</article-title>. <source>Methods Ecol. Evol.</source> <volume>13</volume>, <fpage>2150</fpage>&#x2013;<lpage>2158</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/2041-210X.13918</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="book">
<person-group person-group-type="author">
<collab>MINAGRI</collab>
</person-group> (<year>2019</year>). &#x201c;<article-title>Ministerio de la Agricultura. Rep&#xfa;blica de Cuba. Resoluci&#xf3;n 421. Lista Oficial de Variedades Comerciales</article-title>,&#x201d; in <source>Gaceta oficial</source>. 1058-O92. (<publisher-name>Ministerio de Justicia</publisher-name>: <publisher-loc>La Habana, Cuba</publisher-loc>) pp. <fpage>30</fpage>.</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Motamayor</surname> <given-names>J. C.</given-names>
</name>
<name>
<surname>Lachenaud</surname> <given-names>P.</given-names>
</name>
<name>
<surname>da Silva</surname> <given-names>E. M. J. W.</given-names>
</name>
<name>
<surname>Loor</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Kuhn</surname> <given-names>D. N.</given-names>
</name>
<name>
<surname>Brown</surname> <given-names>J. S.</given-names>
</name>
<etal/>
</person-group>. (<year>2008</year>). <article-title>Geographic and genetic population differentiation of the Amazonian chocolate tree (<italic>Theobroma cacao</italic> L)</article-title>. <source>PloS One</source> <volume>3</volume>, <elocation-id>e3311</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0003311</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Motamayor</surname> <given-names>J. C.</given-names>
</name>
<name>
<surname>Mockaitis</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Schmutz</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Haiminen</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Livingstone</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Cornejo</surname> <given-names>O.</given-names>
</name>
<etal/>
</person-group>. (<year>2013</year>). <article-title>The genome sequence of the most widely cultivated cacao type and its use to identify candidate genes regulating pod color</article-title>. <source>Genome Biol.</source> <volume>14</volume>, <elocation-id>r53</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/gb-2013-14-6-r53</pub-id>
</citation>
</ref>
<ref id="B68">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Motilal</surname> <given-names>L. A.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>The role of gene banks in preserving the genetic diversity of cacao</article-title>,&#x201d; in <source>Achieving sustainable cultivation of cocoa</source>. Ed. <person-group person-group-type="editor">
<name>
<surname>Umaharan</surname> <given-names>P.</given-names>
</name>
</person-group> (<publisher-name>Burleigh Dodds Science Publishing, Cambridge, UK</publisher-name>, <publisher-loc>London</publisher-loc>), <fpage>pp 55</fpage>.</citation>
</ref>
<ref id="B69">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Motilal</surname> <given-names>L. A.</given-names>
</name>
<name>
<surname>Sankar</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Gopaulchan</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Umaharan</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Cocoa</article-title>,&#x201d; in <source>Biotechnology of plantation crops</source>. Eds. <person-group person-group-type="editor">
<name>
<surname>Chowdappa</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Karun</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Rajesh</surname> <given-names>M. K.</given-names>
</name>
<name>
<surname>Ramesh</surname> <given-names>S. V.</given-names>
</name>
</person-group> (<publisher-name>Daya Publishing house</publisher-name>, <publisher-loc>New Delhi</publisher-loc>), <fpage>pp 313</fpage>&#x2013;<lpage>pp 354</lpage>.</citation>
</ref>
<ref id="B70">
<citation citation-type="web">
<person-group person-group-type="author">
<collab>NEB, New England Biolabs</collab>
</person-group> (<year>2022</year>) <article-title>EcoRI product information</article-title>. Available online at: <uri xlink:href="https://international.neb.com">https://international.neb.com</uri> (Accessed <access-date>November 15, 2022</access-date>).</citation>
</ref>
<ref id="B71">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Nu&#xf1;ez-Gonz&#xe1;lez</surname> <given-names>N.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>El cacao y el chocolate en Cuba</article-title>
<source>Fundaci&#xf3;n fernando ortiz</source>. <edition>Segunda Edici&#xf3;n ed</edition> (<publisher-loc>La Habana, Cuba</publisher-loc>: <publisher-name>Fundaci&#xf3;n Fernando Ortiz</publisher-name>), <fpage>pp 336</fpage>.</citation>
</ref>
<ref id="B72">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Olasupo</surname> <given-names>F. O.</given-names>
</name>
<name>
<surname>Adewale</surname> <given-names>D. B.</given-names>
</name>
<name>
<surname>Aikpokpodion</surname> <given-names>P. O.</given-names>
</name>
<name>
<surname>Muyiwa</surname> <given-names>A. A.</given-names>
</name>
<name>
<surname>Bhattacharjee</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Gutierrez</surname> <given-names>O. A.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>Genetic identity and diversity of Nigerian cacao genebank collections verified by single nucleotide polymorphisms (SNPs): a guide to field genebank management and utilization</article-title>. <source>Tree Genet. Genomes</source> <volume>14</volume>, <fpage>32</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11295-018-1244-2</pub-id>
</citation>
</ref>
<ref id="B73">
<citation citation-type="web">
<person-group person-group-type="author">
<collab>ONEI, Oficina Nacional de Estad&#xed;stica e Informaci&#xf3;n</collab>
</person-group> (<year>2021</year>) <article-title>Anuario Estad&#xed;stico de Cuba 2020</article-title>. Available online at: <uri xlink:href="http://www.onei.gob.cu/">http://www.onei.gob.cu/</uri> (Accessed <access-date>December 28, 2022</access-date>).</citation>
</ref>
<ref id="B74">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Osorio-Guarin</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Berdugo-Cely</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Coronado</surname> <given-names>R. A.</given-names>
</name>
<name>
<surname>Zapata</surname> <given-names>Y. P.</given-names>
</name>
<name>
<surname>Quintero</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Gallego-Sanchez</surname> <given-names>G.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). <article-title>Colombia a source of cacao genetic diversity as revealed by the population structure analysis of germplasm bank of <italic>theobroma cacao</italic> L</article-title>. <source>Front. Plant Sci.</source> <volume>8</volume>
<elocation-id>1994</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2017.01994</pub-id>
</citation>
</ref>
<ref id="B75">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Osorio-Guarin</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Berdugo-Cely</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Coronado-Silva</surname> <given-names>R. A.</given-names>
</name>
<name>
<surname>Baez</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Jaimes</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yockteng</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Genome-Wide Association Study Reveals Novel Candidate Genes Associated with Productivity and Disease Resistance to <italic>Moniliophthora</italic> spp. in Cacao (<italic>Theobroma cacao</italic> L.)</article-title>. <source>G3 (Bethesda)`</source> <volume>10</volume>, <fpage>1713</fpage>&#x2013;<lpage>1725</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1534/g3.120.401153</pub-id>
</citation>
</ref>
<ref id="B76">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Osorio-Guarin</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Quackenbush</surname> <given-names>C. R.</given-names>
</name>
<name>
<surname>Cornejo</surname> <given-names>O. E.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Ancestry informative alleles captured with reduced representation library sequencing in <italic>Theobroma cacao</italic>
</article-title>. <source>PloS One</source> <volume>13</volume>, <elocation-id>e0203973</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0203973</pub-id>
</citation>
</ref>
<ref id="B77">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peterson</surname> <given-names>B. K.</given-names>
</name>
<name>
<surname>Weber</surname> <given-names>J. N.</given-names>
</name>
<name>
<surname>Kay</surname> <given-names>E. H.</given-names>
</name>
<name>
<surname>Fisher</surname> <given-names>H. S.</given-names>
</name>
<name>
<surname>Hoekstra</surname> <given-names>H. E.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Double digest RADseq: an inexpensive method for <italic>de novo</italic> SNP discovery and genotyping in model and non-model species</article-title>. <source>PloS One</source> <volume>7</volume>, <elocation-id>e37135</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0037135</pub-id>
</citation>
</ref>
<ref id="B78">
<citation citation-type="web">
<person-group person-group-type="author">
<collab>QIAgen</collab>
</person-group> (<year>2019</year>) <article-title>DNeasy<sup>&#xae;</sup> Plant pro kit handbook</article-title>. Available online at: <uri xlink:href="https://www.qiagen.com/us/resources/download.aspx?id=7f236f8b-84ba-4926-a7d8-813c91ab4a30&amp;lang=en">https://www.qiagen.com/us/resources/download.aspx?id=7f236f8b-84ba-4926-a7d8-813c91ab4a30&amp;lang=en</uri> (Accessed <access-date>15, 2021</access-date>).</citation>
</ref>
<ref id="B79">
<citation citation-type="book">
<person-group person-group-type="author">
<collab>R Core Team</collab>
</person-group> (<year>2020</year>). <source>R: A language and environment for statistical computing</source> (<publisher-loc>Vienna, Austria</publisher-loc>: <publisher-name>R Foundation for Statistical Computing</publisher-name>). Available at: <uri xlink:href="https://www.R-project.org/">https://www.R-project.org/</uri>.</citation>
</ref>
<ref id="B80">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rivera-Colon</surname> <given-names>A. G.</given-names>
</name>
<name>
<surname>Rochette</surname> <given-names>N. C.</given-names>
</name>
<name>
<surname>Catchen</surname> <given-names>J. M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Simulation with RADinitio improves RADseq experimental design and sheds light on sources of missing data</article-title>. <source>Mol. Ecol. Resour.</source> <volume>21</volume>, <fpage>363</fpage>&#x2013;<lpage>378</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/1755-0998.13163</pub-id>
</citation>
</ref>
<ref id="B81">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rochette</surname> <given-names>N. C.</given-names>
</name>
<name>
<surname>Catchen</surname> <given-names>J. M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Deriving genotypes from RAD-seq short-read data using Stacks</article-title>. <source>Nat. Protoc.</source> <volume>12</volume>, <fpage>2640</fpage>&#x2013;<lpage>2659</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nprot.2017.123</pub-id>
</citation>
</ref>
<ref id="B82">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saijo</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Loo</surname> <given-names>E. P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Plant immunity in signal integration between biotic and abiotic stress responses</article-title>. <source>New Phytol.</source> <volume>225</volume>, <fpage>87</fpage>&#x2013;<lpage>104</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/nph.15989</pub-id>
</citation>
</ref>
<ref id="B83">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Scheben</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Batley</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Edwards</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Genotyping-by-sequencing approaches to characterize crop genomes: choosing the right tool for the right application</article-title>. <source>Plant Biotechnol. J.</source> <volume>15</volume>, <fpage>149</fpage>&#x2013;<lpage>161</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/pbi.12645</pub-id>
</citation>
</ref>
<ref id="B84">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>T.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Genetic variation and population structure in China summer maize germplasm</article-title>. <source>Sci. Rep.</source> <volume>11</volume>, <fpage>8012</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-021-84732-6</pub-id>
</citation>
</ref>
<ref id="B85">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Souza</surname> <given-names>H. A. V.</given-names>
</name>
<name>
<surname>Muller</surname> <given-names>L. A. C.</given-names>
</name>
<name>
<surname>Brand&#xe3;o</surname> <given-names>R. L.</given-names>
</name>
<name>
<surname>Lovato</surname> <given-names>M. B.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Isolation of high quality and polysaccharide-free DNA from leaves of <italic>Dimorphandra mollis</italic> (<italic>Leguminosae</italic>), a tree from the Brazilian Cerrado</article-title>. <source>Genet. Mol. Res.</source> <volume>11</volume>, <fpage>756</fpage>&#x2013;<lpage>764</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.4238/2012.March.22.6</pub-id>
</citation>
</ref>
<ref id="B86">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Takrama</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Kun</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Meinhardt</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Mischke</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Opoku</surname> <given-names>S. Y.</given-names>
</name>
<name>
<surname>Padi</surname> <given-names>F. K.</given-names>
</name>
<etal/>
</person-group>. (<year>2014</year>). <article-title>Verification of genetic identity of introduced cacao germplasm in Ghana using single nucleotide polymorphism (SNP) markers</article-title>. <source>Afr. J. Biotechnol.</source> <volume>13</volume>, <fpage>2127</fpage>&#x2013;<lpage>2136</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.5897/AJB</pub-id>
</citation>
</ref>
<ref id="B87">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ten Hoopen</surname> <given-names>G. M.</given-names>
</name>
<name>
<surname>Umaharan</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Preventing the spread and mitigating the impact of cocoa diseases in the Caribbean</article-title>,&#x201d; in <source>Cocoa research and development symposium</source> (<publisher-name>Cocoa Research Centre</publisher-name>, <publisher-loc>Saint Augustine, Trinit&#xe9;-et-Tobago</publisher-loc>). CIRAD-BIOS-UPR Bioagresseurs (TTO).</citation>
</ref>
<ref id="B88">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Thioulouse</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Dray</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Dufour</surname> <given-names>A.-B.</given-names>
</name>
<name>
<surname>Siberchicot</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Jombart</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Pavoine</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Multivariate analysis of ecological data with ade4</source> (<publisher-loc>Springer</publisher-loc>: <publisher-name>New York, NY, USA</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-1-4939-8850-1</pub-id>
</citation>
</ref>
<ref id="B89">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Van der Auwera</surname> <given-names>G. A.</given-names>
</name>
<name>
<surname>O&#x2019;Connor</surname> <given-names>B. D.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Genomics in the cloud</article-title>,&#x201d; in <source>Using docker, GATK, and WDL in terra</source>, <edition>1st Edition</edition> (<publisher-loc>Sebastopol, CA</publisher-loc>: <publisher-name>O&#x2019;Reilly Media Inc</publisher-name>), <fpage>pp 467</fpage>.</citation>
</ref>
<ref id="B90">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Motilal</surname> <given-names>L. A.</given-names>
</name>
<name>
<surname>Meinhardt</surname> <given-names>L. W.</given-names>
</name>
<name>
<surname>Yin</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Molecular characterization of a cacao germplasm collection maintained in yunnan, China using single nucleotide polymorphism (SNP) markers</article-title>. <source>Trop. Plant Biol.</source> <volume>13</volume>, <fpage>359</fpage>&#x2013;<lpage>370</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s12042-020-09267-y</pub-id>
</citation>
</ref>
<ref id="B91">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weir</surname> <given-names>B. S.</given-names>
</name>
<name>
<surname>Cockerham</surname> <given-names>C. C.</given-names>
</name>
</person-group> (<year>1984</year>). <article-title>Estimating F-Statistics for the analysis of population structure</article-title>. <source>Evol</source> <volume>38</volume>, <fpage>1358</fpage>&#x2013;<lpage>1370</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2307/2408641</pub-id>
</citation>
</ref>
<ref id="B92">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wickramasuriya</surname> <given-names>A. M.</given-names>
</name>
<name>
<surname>Dunwell</surname> <given-names>J. M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Cacao biotechnology: current status and future prospects</article-title>. <source>Plant Biotechnol. J.</source> <volume>16</volume>, <fpage>4</fpage>&#x2013;<lpage>17</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/pbi.12848</pub-id>
</citation>
</ref>
<ref id="B93">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yin</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yin</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>rMVP: A memory-efficient, visualization-enhanced, and parallel-accelerated tool for genome-wide association study</article-title>. <source>Genomics Proteomics Bioinf.</source> <volume>19</volume>, <fpage>619</fpage>&#x2013;<lpage>628</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.gpb.2020.10.007</pub-id>
</citation>
</ref>
<ref id="B94">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Smith</surname> <given-names>D. K.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Guan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Lam</surname> <given-names>T. T. Y.</given-names>
</name>
<name>
<surname>McInerny</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>ggtree: an r package for visualization and annotation of phylogenetic trees with their covariates and other associated data</article-title>. <source>Methods Ecol. Evol.</source> <volume>8</volume>, <fpage>28</fpage>&#x2013;<lpage>36</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/2041-210X.12628</pub-id>
</citation>
</ref>
<ref id="B95">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Gu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Genetic diversity and population structure of popcorn germplasm resources using genome-wide SNPs through genotyping-by-sequencing</article-title>. <source>Gen. Resour. Crop Evol.</source> <volume>68</volume>, <fpage>2379</fpage>&#x2013;<lpage>2389</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10722-021-01137-0</pub-id>
</citation>
</ref>
<ref id="B96">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Motilal</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Origin, dispersal, and current global distribution of cacao genetic diversity</article-title>,&#x201d;. Eds. <person-group person-group-type="editor">
<name>
<surname>Bailey</surname> <given-names>B. A.</given-names>
</name>
<name>
<surname>Meinhardt</surname> <given-names>L. W.</given-names>
</name>
</person-group> <source>Cacao diseases: A history of old enemies and new encounters</source>. (<publisher-loc>Cacao Diseases</publisher-loc>: <publisher-name>Springer International Publishing, Switzerland</publisher-name>), <fpage>137</fpage>&#x2013;<lpage>177</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>
