<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2022.787703</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Phenotypic Variation and the Impact of Admixture in the <italic>Oryza rufipogon</italic> Species Complex (<italic>ORSC</italic>)</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Eizenga</surname> <given-names>Georgia C.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<xref ref-type="author-notes" rid="fn003"><sup>&#x2021;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/395501/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Kim</surname> <given-names>HyunJung</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
<xref ref-type="author-notes" rid="fn003"><sup>&#x2021;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/495479/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Jung</surname> <given-names>Janelle K. H.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn003"><sup>&#x2021;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/75521/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Greenberg</surname> <given-names>Anthony J.</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Edwards</surname> <given-names>Jeremy D.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/899268/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Naredo</surname> <given-names>Maria Elizabeth B.</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/595244/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Banaticla-Hilario</surname> <given-names>Maria Celeste N.</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1793950/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Harrington</surname> <given-names>Sandra E.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1084526/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Shi</surname> <given-names>Yuxin</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1791455/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Kimball</surname> <given-names>Jennifer A.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x2020;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Harper</surname> <given-names>Lisa A.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>McNally</surname> <given-names>Kenneth L.</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/72481/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>McCouch</surname> <given-names>Susan R.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/75889/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Dale Bumpers National Rice Research Center, USDA-ARS</institution>, <addr-line>Stuttgart, AR</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Plant Breeding and Genetics Section, School of Integrative Plant Science, Cornell University</institution>, <addr-line>Ithaca, NY</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>Bayesic Research LLC.</institution>, <addr-line>Ithaca, NY</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>International Rice Research Institute</institution>, <addr-line>Los Ba&#x00F1;os</addr-line>, <country>Philippines</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Anna Maria Mastrangelo, Research Centre for Cereal and Industrial Crops, Council for Agricultural and Economics Research (CREA), Italy</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Ryo Ishikawa, Kobe University, Japan; Robert Henry, The University of Queensland, Australia</p></fn>
<corresp id="c001">&#x002A;Correspondence: Georgia C. Eizenga, <email>georgia.eizenga@usda.gov</email>; <ext-link ext-link-type="uri" xlink:href="https://orcid.org/0000-0003-0330-9225">orcid.org/0000-0003-0330-9225</ext-link></corresp>
<corresp id="c002">Susan R. McCouch, <email>srm4@cornell.edu</email>; <ext-link ext-link-type="uri" xlink:href="https://orcid.org/0000-0001-9246-3106">orcid.org/0000-0001-9246-3106</ext-link></corresp>
<fn fn-type="present-address" id="fn002"><p><sup>&#x2020;</sup>Present addresses: HyunJung Kim, LG Chemical Ltd., Seoul, South Korea; Maria Elizabeth B. Naredo, Philippine Coconut Authority &#x2013; Albay Research Center, Guinobatan, Philippines; Jennifer A. Kimball, University of Minnesota-Twin Cities, St. Paul, MN, United States</p></fn>
<fn fn-type="other" id="fn003"><p><sup>&#x2021;</sup>These authors share first authorship</p></fn>
<fn fn-type="other" id="fn004"><p>This article was submitted to Plant Breeding, a section of the journal Frontiers in Plant Science</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>13</day>
<month>06</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>787703</elocation-id>
<history>
<date date-type="received">
<day>01</day>
<month>10</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>13</day>
<month>04</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2022 At least a portion of this work is authored by Georgia C. Eizenga and Jeremy D. Edwards on behalf of the U.S. Government and as regards Dr. Eizenga, Dr Edwards and the U.S. Government, is not subject to copyright protection in the United States. Foreign and other copyrights may apply.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>At least a portion of this work is authored by Georgia C. Eizenga and Jeremy D. Edwards on behalf of the U.S. Government and as regards Dr. Eizenga, Dr Edwards and the U.S. Government, is not subject to copyright protection in the United States. Foreign and other copyrights may apply.</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Crop wild relatives represent valuable reservoirs of variation for breeding, but their populations are threatened in natural habitats, are sparsely represented in genebanks, and most are poorly characterized. The focus of this study is the <italic>Oryza rufipogon</italic> species complex (<italic>ORSC</italic>), wild progenitor of Asian rice (<italic>Oryza sativa</italic> L.). The <italic>ORSC</italic> comprises perennial, annual and intermediate forms which were historically designated as <italic>O. rufipogon, O. nivara</italic>, and <italic>O. sativa</italic> f. <italic>spontanea</italic> (or <italic>Oryza</italic> spp., an annual form of mixed <italic>O. rufipogon/O. nivara</italic> and <italic>O. sativa</italic> ancestry), respectively, based on non-standardized morphological, geographical, and/or ecologically-based species definitions and boundaries. Here, a collection of 240 diverse <italic>ORSC</italic> accessions, characterized by genotyping-by-sequencing (113,739 SNPs), was phenotyped for 44 traits associated with plant, panicle, and seed morphology in the screenhouse at the International Rice Research Institute, Philippines. These traits included heritable phenotypes often recorded as characterization data by genebanks. Over 100 of these <italic>ORSC</italic> accessions were also phenotyped in the greenhouse for 18 traits in Stuttgart, Arkansas, and 16 traits in Ithaca, New York, United States. We implemented a Bayesian Gaussian mixture model to infer accession groups from a subset of these phenotypic data and ascertained three phenotype-based group assignments. We used concordance between the genotypic subpopulations and these phenotype-based groups to identify a suite of phenotypic traits that could reliably differentiate the <italic>ORSC</italic> populations, whether measured in tropical or temperate regions. The traits provide insight into plant morphology, life history (perenniality versus annuality) and mating habit (self- versus cross-pollinated), and are largely consistent with genebank species designations. One phenotypic group contains predominantly <italic>O. rufipogon</italic> accessions characterized as perennial and largely out-crossing and one contains predominantly <italic>O. nivara</italic> accessions characterized as annual and largely inbreeding. From these groups, 42 &#x201C;core&#x201D; <italic>O. rufipogon</italic> and 25 &#x201C;core&#x201D; <italic>O. nivara</italic> accessions were identified for domestication studies. The third group, comprising 20% of our collection, has the most accessions identified as <italic>Oryza</italic> spp. (51.2%) and levels of <italic>O. sativa</italic> admixture accounting for more than 50% of the genome. This third group is potentially useful as a &#x201C;pre-breeding&#x201D; pool for breeders attempting to incorporate novel variation into elite breeding lines.</p>
</abstract>
<kwd-group>
<kwd>rice</kwd>
<kwd>Bayesian Gaussian mixture models</kwd>
<kwd><italic>Oryza rufipogon</italic></kwd>
<kwd><italic>Oryza sativa</italic></kwd>
<kwd><italic>Oryza nivara</italic></kwd>
<kwd>genebank accessions</kwd>
<kwd>crop wild relatives</kwd>
<kwd><italic>Oryza rufipogon</italic> species complex</kwd>
</kwd-group>
<contract-num rid="cn001">0606461</contract-num>
<contract-num rid="cn001">1026555</contract-num>
<contract-sponsor id="cn001">National Science Foundation<named-content content-type="fundref-id">10.13039/100000001</named-content></contract-sponsor>
<counts>
<fig-count count="4"/>
<table-count count="1"/>
<equation-count count="5"/>
<ref-count count="61"/>
<page-count count="15"/>
<word-count count="11904"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>Introduction</title>
<p>Wild relatives of domesticated crop species are some of the greatest sources of untapped genetic variation available to plant breeders as they confront the challenges of a changing climate. Yet populations of crop wild relatives are underrepresented in genebanks, threatened in their natural environments, and poorly characterized in both <italic>ex situ</italic> and <italic>in situ</italic> collections. The <italic>Oryza rufipogon</italic> species complex (<italic>ORSC</italic>), previously referred to as <italic>O. rufipogon</italic> Griff. or <italic>O. perennis</italic> Moench, is the wild progenitor of Asian rice (<italic>Oryza sativa</italic> L.) and is comprised of perennial, annual, and intermediate ecotypes (<xref ref-type="bibr" rid="B48">Sano et al., 1980</xref>; <xref ref-type="bibr" rid="B42">Oka, 1988</xref>). As summarized in <xref ref-type="supplementary-material" rid="TS1">Supplementary Table 1</xref>, the perennial ecotype exhibits vigorous vegetative growth, is largely out-crossing, and is found in areas which are continuously wet while the annual ecotype shows less vigorous vegetative growth, is primarily in-breeding, and is found in areas that are seasonally wet (<xref ref-type="bibr" rid="B43">Oka and Morishima, 1967</xref>; <xref ref-type="bibr" rid="B39">Morishima et al., 1984</xref>; <xref ref-type="bibr" rid="B54">Vaughan et al., 2008</xref>). In accordance with these differences in life history, the perennial form is capable of reproducing clonally via stolons, typically has low seed productivity, is late flowering and photoperiod sensitive, while the annual form lacks stolons, has high seed productivity, is frequently early-flowering and photoperiod insensitive (<xref ref-type="bibr" rid="B39">Morishima et al., 1984</xref>; <xref ref-type="bibr" rid="B3">Barbier, 1989a</xref>,<xref ref-type="bibr" rid="B4">b</xref>; <xref ref-type="bibr" rid="B5">Barbier et al., 1991</xref>; <xref ref-type="bibr" rid="B37">Morishima, 2001</xref>; <xref ref-type="bibr" rid="B55">Vaughan et al., 2003</xref>, <xref ref-type="bibr" rid="B54">2008</xref>; <xref ref-type="bibr" rid="B8">Cai et al., 2004</xref>). An extensive survey of both annual and perennial ecotypes of the <italic>ORSC</italic> led <xref ref-type="bibr" rid="B50">Sharma and Shastry (1965)</xref> to treat the annual form as a separate species, designated <italic>O. nivara</italic>, as originally suggested by <xref ref-type="bibr" rid="B10">Chatterjee (1948)</xref>, to distinguish it from the perennial <italic>O. rufipogon</italic>. A weedy, intermediate form is classified as <italic>O. sativa</italic> f. <italic>spontanea</italic> Roschev (<xref ref-type="bibr" rid="B46">Roschevicz, 1931</xref>; <xref ref-type="bibr" rid="B42">Oka, 1988</xref>) and sometimes erroneously designated <italic>O. spontanea</italic> (<xref ref-type="bibr" rid="B41">Oka, 1974</xref>; <xref ref-type="bibr" rid="B48">Sano et al., 1980</xref>; <xref ref-type="bibr" rid="B52">Vaughan, 1989</xref>; <xref ref-type="bibr" rid="B21">Hill, 2010</xref>). Weedy rice is considered an annual of mixed <italic>O. rufipogon/O. nivara</italic> and <italic>O. sativa</italic> ancestry (<xref ref-type="bibr" rid="B38">Morishima et al., 1961</xref>; <xref ref-type="bibr" rid="B9">Chang, 1976</xref>; <xref ref-type="bibr" rid="B56">Vaughan et al., 2001</xref>; <xref ref-type="bibr" rid="B49">Sharma, 2003</xref>).</p>
<p>To this day, wide differences in the nomenclature and trait-based definitions assigned to the wild ancestor of <italic>O. sativa</italic> persist in rice genebank databases and research labs around the world. According to the online database of the International Rice Germplasm Collection (IRGC)<sup><xref ref-type="fn" rid="footnote1">1</xref></sup> the list of available <italic>ORSC</italic> accessions includes 846 <italic>O. rufipogon</italic>, 1,455 <italic>O. nivara</italic>, and 1,134 accessions listed as either &#x2018;<italic>Oryza</italic> spp.&#x2019; or &#x2018;<italic>Oryza</italic> hybr.&#x2019;. Many of the <italic>Oryza</italic> spp. or <italic>Oryza</italic> hybr. were previously designated as <italic>O. spontanea (O. sativa</italic> f. <italic>spontanea)</italic> or referred to as hybrids between <italic>O. sativa</italic> and <italic>O. rufipogon</italic> and/or <italic>O. nivara</italic>. The online wild strain database of the Japanese National Institute of Genetics, <italic>Oryzabase</italic> (<xref ref-type="bibr" rid="B30">Kurata and Yamazaki, 2006</xref>; <xref ref-type="bibr" rid="B60">Yamazaki et al., 2010</xref>), acknowledges confusion in the wild rice taxonomy and nomenclature, including a distinction between <italic>O. nivara</italic> and <italic>O. rufipogon</italic> based on ecology and life habit, but clearly states its use of the <italic>sensu lato</italic> classification of <italic>O. rufipogon</italic> as a single species with a continuous range of annual, perennial, and intermediate types as held by core collectors Morishima and colleagues (<xref ref-type="bibr" rid="B38">Morishima et al., 1961</xref>). There are 682 accessions in <italic>Oryzabase</italic> designated as &#x2018;<italic>O. rufipogon</italic>&#x2019; with additional, but incomplete information on former species designations (i.e., &#x2018;<italic>O. perennis&#x2019;, &#x2018;O. perennis (O. nivara),&#x2019;</italic> or &#x2018;<italic>O. sativa f. spontanea</italic>,&#x2019;), and life habit designations, such as annual or perennial<sup><xref ref-type="fn" rid="footnote2">2</xref></sup>.</p>
<p>While the species designations originally assigned to accessions in the IRGC have persisted largely unchanged through the decades, there has been a paradigm shift in biology from morphology-based to genetic identity-based species definitions. In the case of the <italic>ORSC</italic>, studies using molecular markers (isozymes, RFLPs, SSRs, SINEs and other indels, and SNPs) present conflicting and often contradictory results that collectively fail to support the existence of two, well-differentiated species. Global studies of genomic diversity in the <italic>ORSC</italic> document three to eight genetically distinct groups (<xref ref-type="bibr" rid="B33">Londo et al., 2006</xref>; <xref ref-type="bibr" rid="B61">Zheng and Ge, 2010</xref>; <xref ref-type="bibr" rid="B24">Huang et al., 2012</xref>; <xref ref-type="bibr" rid="B1">Banaticla-Hilario et al., 2013a</xref>,<xref ref-type="bibr" rid="B2">b</xref>; <xref ref-type="bibr" rid="B32">Liu et al., 2015</xref>; <xref ref-type="bibr" rid="B29">Kim et al., 2016</xref>) more closely associated with geography than with life-habit. Pronounced genetic differentiation has been reported between <italic>O. rufipogon</italic> and <italic>O. nivara</italic> only in studies involving local collections of germplasm from South Asia (<xref ref-type="bibr" rid="B47">Samal et al., 2018</xref>) and Southeast Asia (<xref ref-type="bibr" rid="B31">Kuroda et al., 2006</xref>; <xref ref-type="bibr" rid="B1">Banaticla-Hilario et al., 2013a</xref>), where both species occur sympatrically but remain differentiated due to differences in flowering time and ecological adaptation. Genetic differentiation has not been documented in China or Oceania where native populations of <italic>O. nivara</italic> are largely absent (<xref ref-type="bibr" rid="B53">Vaughan, 1994</xref>; <xref ref-type="bibr" rid="B32">Liu et al., 2015</xref>).</p>
<p>Several studies based on genome-wide SNPs have examined relationships among <italic>ORSC</italic> accessions and among <italic>ORSC</italic> and <italic>O. sativa</italic> subpopulations. A study by <xref ref-type="bibr" rid="B24">Huang et al. (2012)</xref> examined a panel of 446 <italic>ORSC</italic> accessions and 1,083 <italic>O. sativa</italic> cultivars that had been genotyped with &#x223C;5M SNPs and identified three major groups of <italic>ORSC</italic> accessions using a neighbor-joining tree. These groups were significantly correlated with geographic distribution and were referred to as <italic>OrI, OrII</italic> and <italic>OrIII</italic>. Subsequently, <italic>OrI</italic> was divided into two sub-clades, where <italic>OrIa</italic> included <italic>ORSC</italic> accessions and rice belonging to the <italic>O. sativa-indica</italic> subpopulation and <italic>OrIb</italic> included <italic>ORSC</italic> and <italic>O. sativa-aus</italic> rices; <italic>OrIII</italic> was also divided into two sub-clades, where <italic>OrIIIa</italic> included <italic>ORSC</italic> accessions from Southern China as well as <italic>O. sativa-japonica</italic> rices, and <italic>OrIIIb</italic> included <italic>ORSC</italic> accessions that clustered independently of <italic>O. sativa</italic>.</p>
<p>The <italic>ORSC</italic> dataset generated by <xref ref-type="bibr" rid="B24">Huang et al. (2012)</xref> was subsequently re-analyzed by three independent research groups: <xref ref-type="bibr" rid="B14">Civ&#x00E1;n et al. (2015)</xref>, <xref ref-type="bibr" rid="B57">Wang et al. (2017)</xref>, and <xref ref-type="bibr" rid="B13">Choi and Purugganan (2018)</xref>. In each case the authors reached very different conclusions about the subpopulation structure of the <italic>ORSC</italic> and relationships with the cultivated gene pools of <italic>O. sativa.</italic> These subsequent studies revealed evidence of substantial gene flow from domesticated groups into wild populations and suggested that introgression rather than phylogenetic relationship was a primary driver of the <italic>ORSC</italic> groups reported by <xref ref-type="bibr" rid="B24">Huang et al. (2012)</xref>. Use of larger K values (population number) and local ancestry estimation techniques helped to resolve the heavily <italic>O. sativa</italic>-admixed wild groups as distinct from independent wild populations. The analysis by <xref ref-type="bibr" rid="B57">Wang et al. (2017)</xref> combined the sequence data of 435 <italic>ORSC</italic> accessions (<xref ref-type="bibr" rid="B24">Huang et al., 2012</xref>) with sequencing data from 203 <italic>O. sativa</italic> accessions included in the Rice Minicore collection (<xref ref-type="bibr" rid="B58">Wang et al., 2016</xref>). Subsequent STRUCTURE analysis (<italic>K</italic> = 9) ascertained six <italic>ORSC</italic> subpopulations: four unique <italic>ORSC</italic> subpopulations that were correlated with geographic distribution and two that clustered with <italic>O. sativa</italic> subpopulations. Of these, <italic>Or-A</italic> accessions had the broadest range with the highest proportion from Oceania, <italic>Or-B</italic> accessions were almost exclusively found in China, <italic>Or-C</italic> had the highest proportion from West India and Sri Lanka, and <italic>Or-D</italic> was mainly from the SE Asia, Bangladesh, and East India, <italic>Or-E</italic> accessions clustered with <italic>O. sativa</italic>-<italic>aus</italic> accessions, and <italic>Or-F</italic> clustered with <italic>O. sativa-indica</italic> (<xref ref-type="bibr" rid="B57">Wang et al., 2017</xref>). Of note, about 42% of these <italic>ORSC</italic> accessions were deemed to be substantially admixed, thus could not be assigned to a single ancestry group.</p>
<p>To further disentangle the population dynamics of the <italic>ORSC</italic>, a different collection of 286 accessions from the IRGC was genotyped using genotyping-by-sequencing (113,739 SNPs) and evaluated at 25 polymorphic sites in the chloroplast genome (<xref ref-type="bibr" rid="B29">Kim et al., 2016</xref>). Six wild subpopulation groups were identified; three were closely related to <italic>O. sativa-indica, -aus</italic>, and -<italic>japonica</italic>, respectively, while the other three subpopulations were genetically divergent, had unique chloroplast haplotypes, and were located at the geographical extremes of the species range.</p>
<p>Another study examined local differentiation between <italic>O. rufipogon</italic> and <italic>O. nivara</italic> using a collection of 52 pairs of sympatric <italic>O. rufipogon</italic> and <italic>O. nivara</italic> accessions that were phenotyped for 32 traits (<xref ref-type="bibr" rid="B1">Banaticla-Hilario et al., 2013a</xref>) and genotyped with 29 SSR markers (<xref ref-type="bibr" rid="B2">Banaticla-Hilario et al., 2013b</xref>). The genotypic neighbor-joining tree showed no clear-cut genetic separation of <italic>O. nivara</italic> and <italic>O. rufipogon</italic> accessions though, species separation was apparent at the local scale. Nonetheless, Bayesian clustering was able to differentiate the two species in sympatric population pairs across the entire range of their distribution (<xref ref-type="bibr" rid="B1">Banaticla-Hilario et al., 2013a</xref>). Furthermore, this study identified a cluster of <italic>O. nivara</italic> accessions from Nepal that diverged significantly from the other population groups (similar to the findings reported by <xref ref-type="bibr" rid="B29">Kim et al., 2016</xref>), as well as two or three other genetically and geographically distinct subpopulations of <italic>O. nivara</italic>, suggesting that reproductive barriers in this inbreeding group may tend to intensify under sympatric conditions, and that <italic>O. nivara</italic> may have originated more than once from its perennial ancestor. Local differences in flowering time or in floral and panicle structure associated with the mating system would impact adaptation and have been known to drive reproductive isolation in closely related species (<xref ref-type="bibr" rid="B20">Grillo et al., 2009</xref>; <xref ref-type="bibr" rid="B32">Liu et al., 2015</xref>). Statistical analysis of the phenotypic data showed 27 of the 32 traits were significantly different between <italic>O. rufipogon</italic> and <italic>O. nivara</italic> with spikelet width, anther length, culm length and photoperiod explaining 25.4 percent of the variation between the two species. These studies also noted that Southeast Asian accessions appeared to be more recently diverged and/or had more interspecific gene flow compared to those from South Asia.</p>
<p>To augment our understanding of diversity and population structure in the <italic>ORSC</italic> and to facilitate the selection of materials for use in plant breeding, we phenotyped the collection genotyped by <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref>. Phenotyping was performed at three locations (Los Ba&#x00F1;os, Philippines, Arkansas, United States and New York, United States) for heritable traits, with emphasis on those commonly used to characterize rice accessions in genebanks (<xref ref-type="bibr" rid="B6">Bioversity International et al., 2007</xref>). Several of the traits were associated with differences in life habit and/or mating system and served to substantiate the distinction between <italic>O. nivara</italic> and <italic>O. rufipogon</italic> as distinct ecotypes within the <italic>ORSC</italic>. A significant proportion of <italic>ORSC</italic> accessions carried unexpected combinations of phenotypes and were associated with admixture from <italic>O. sativa</italic>.</p>
<p>To enable a more robust interpretation of the phenotypic variation observed, we also merged genotypic datasets for the 286 <italic>ORSC</italic> accessions analyzed by <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> and the 446 <italic>ORSC</italic> accessions analyzed by <xref ref-type="bibr" rid="B24">Huang et al. (2012)</xref>, <xref ref-type="bibr" rid="B14">Civ&#x00E1;n et al. (2015)</xref>, <xref ref-type="bibr" rid="B57">Wang et al. (2017)</xref>, and <xref ref-type="bibr" rid="B13">Choi and Purugganan (2018)</xref> and analyzed the extent of unique or shared variation in each of the studies. Based on these analyses, we identify a set of &#x2018;core accessions&#x2019; representing <italic>O. rufipogon</italic> and <italic>O. nivara</italic> along with a large group of <italic>ORSC</italic> accessions that represent unique sources of naturally occurring, highly admixed pre-breeding material for use in plant breeding. This work adds value to the <italic>ORSC</italic> genetic resources currently conserved in the IRGC and enhances opportunities for expanded utilization in research and plant improvement.</p>
</sec>
<sec id="S2" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="S2.SS1">
<title>Plant Materials and Phenotypic Data Collection</title>
<p>For this study, 240 <italic>ORSC</italic> accessions from the International Rice Germplasm Collection (IRGC) previously genotyped by <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref>, were selected for phenotypic evaluation (<xref ref-type="supplementary-material" rid="TS2">Supplementary Table 2</xref>). This collection included 102 <italic>ORSC</italic> accessions from Southeast Asia, 79 accessions from South Asia, 55 accessions from East Asia and four accessions from Australasia. Of these, 222 accessions were evaluated in a screenhouse at the IRRI, Los Ba&#x00F1;os, Philippines and 130 were evaluated under greenhouse conditions in the United States, with 112 accessions phenotyped in more than one location. At IRRI two plants were grown of each accession with one plant per pot and each pot was considered a single replication. Seed of each accession was planted between June 20 and July 4, 2008, and the plants were phenotyped for the 44 morphological traits related to the vegetative, reproductive and harvest growth stages (<xref ref-type="supplementary-material" rid="TS3">Supplementary Table 3</xref>). Twenty-two of these 44 traits were measured five times on an individual plant to improve the accuracy of the measurement (<xref ref-type="supplementary-material" rid="TS4">Supplementary Table 4</xref>).</p>
<p>In the United States, 104 accessions were phenotyped at both Cornell University (CU) in Ithaca, New York and the Dale Bumpers National Rice Research Center (DB) near Stuttgart, Arkansas. Three plants were grown of each accession with one plant per pot and each pot was considered a single replication. At CU, 104 accessions were planted between August 31 and September 8, 2008, and each plant was phenotyped for 16 morphological traits. At DB, 129 <italic>ORSC</italic> accessions were grown in the greenhouse with three plants per accession, one plant per pot, each pot was considered one replication, and plants were phenotyped for 18 morphological traits. In Arkansas, each accession was phenotyped in two different years, thus a total of six plants were characterized for most accessions. The accessions were planted August 24 to September 6, 2007 (86 accessions); August 1 to September 12, 2008 (124 accessions) and August 12 to 13, 2009 (41 accessions). <xref ref-type="supplementary-material" rid="TS2">Supplementary Table 2</xref> lists the location where each accession was grown.</p>
<p><xref ref-type="supplementary-material" rid="TS3">Supplementary Table 3</xref> summarizes the 57 phenotypic traits characterized across the three locations and indicates the location(s) where the trait data were collected. Additional details about the collection of phenotypic data at IRRI are included in <xref ref-type="bibr" rid="B6">Bioversity International et al. (2007)</xref>. Also listed are the corresponding Planteome acronyms and the trait ontology or crop ontology terms (<xref ref-type="bibr" rid="B15">Cooper et al., 2018</xref>). At IRRI, for the 22 quantitative traits with five measurements per plant, the within-plant variances were very small, thus a mean was calculated for subsequent analyses. The replicated trait data collected for each accession by location was used to calculate a best linear unbiased estimator (BLUE) for each trait and accession by location as described in the next section. The BLUEs are listed by location in <xref ref-type="supplementary-material" rid="TS4">Supplementary Table 4</xref>.</p>
</sec>
<sec id="S2.SS2">
<title>Data Quality</title>
<p>Using the replicated trait measurements taken at each location and a relationship matrix constructed from the GBS genotypes (<xref ref-type="bibr" rid="B29">Kim et al., 2016</xref>), we fit multi-trait Bayesian hierarchical models (<xref ref-type="bibr" rid="B19">Greenberg et al., 2011</xref>) to the data from each site separately. The model is:</p>
<disp-formula id="S2.Ex1">
<mml:math id="M1" display='block'><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mtext>&#x2009;&#x2009;&#x2009;</mml:mtext><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x22C5;</mml:mo></mml:mrow></mml:msub><mml:mo>&#x223C;</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mi>v</mml:mi></mml:msub><mml:mmultiscripts><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mtext>&#x03BC;</mml:mtext><mml:mrow><mml:mi>j</mml:mi><mml:mo stretchy='false'>[</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>]</mml:mo><mml:mo>&#x22C5;</mml:mo></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mo>;</mml:mo><mml:msup><mml:mtext>&#x03A3;</mml:mtext><mml:mi>e</mml:mi></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mprescripts/><mml:mrow><mml:mi>e</mml:mi><mml:mo>,</mml:mo><mml:mi>d</mml:mi></mml:mrow><mml:mrow></mml:mrow></mml:mmultiscripts></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msubsup><mml:mtext>&#x03BC;</mml:mtext><mml:mrow><mml:mi>j</mml:mi><mml:mo>&#x22C5;</mml:mo></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x223C;</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mi>d</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mtext>&#x03BC;</mml:mtext><mml:mo>+</mml:mo><mml:msub><mml:mi>u</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mo>&#x22C5;</mml:mo></mml:mrow></mml:msub><mml:mtext>&#x0393;</mml:mtext><mml:mo>;</mml:mo><mml:msup><mml:mtext>&#x03A3;</mml:mtext><mml:mi>s</mml:mi></mml:msup><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x2009;&#x2009;</mml:mtext><mml:msub><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mo>&#x22C5;</mml:mo></mml:mrow></mml:msub><mml:mo>&#x223C;</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mi>d</mml:mi></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mn>0</mml:mn><mml:mi>d</mml:mi></mml:msub><mml:mo>;</mml:mo><mml:msup><mml:mtext>&#x03A3;</mml:mtext><mml:mi>a</mml:mi></mml:msup><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Location parameters of the form <bold><italic>y</italic></bold><sub>i</sub><bold>.</bold> are row-vectors of the corresponding parameter matrices that have data points as rows and traits as columns. The overall intercept &#x03BC;was modeled with a high-variance (<bold><italic>I</italic></bold><inline-formula><mml:math id="INEQ3"><mml:msubsup><mml:mi mathvariant="normal">&#x03C3;</mml:mi><mml:mn>0</mml:mn><mml:mn>2</mml:mn></mml:msubsup></mml:math></inline-formula> = 10<sup>6</sup>) Gaussian prior. The errors were modeled using a multivariate Student-<italic>t</italic> distribution with three degrees of freedom to dampen the effects of outliers (<xref ref-type="bibr" rid="B18">Greenberg et al., 2010</xref>; <xref ref-type="bibr" rid="B19">Greenberg et al., 2011</xref>). We computed the parameters using a C++ program based on an openly-available library<sup><xref ref-type="fn" rid="footnote3">3</xref></sup> that implements methods described by <xref ref-type="bibr" rid="B19">Greenberg et al. (2011)</xref>.</p>
<p>The marker effects enter through the eigenvectors <bold><italic>U</italic></bold> of the relationship matrix, weighted by square roots of their eigenvalues. We estimated the relationship matrix from all non-singleton SNPs in each data set using the van Raden method (<xref ref-type="bibr" rid="B51">VanRaden, 2008</xref>). We used all the eigenvectors that correspond to non-zero eigenvalues. If this regression is performed using a Gaussian prior on the coefficients &#x03B3;<sub><italic>j</italic>.</sub>, it is the Bayesian analog of the (matrix-variate) mixed effect model (<xref ref-type="bibr" rid="B26">Kang et al., 2010</xref>; <xref ref-type="bibr" rid="B22">Hoffman, 2013</xref>). The accession means <inline-formula><mml:math id="INEQ5"><mml:msubsup><mml:mi mathvariant="normal">&#x03BC;</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mo>.</mml:mo></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>c</mml:mi><mml:mo>&#x2062;</mml:mo><mml:mi>c</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>are analogous to BLUEs. Modes of their posterior distributions were used in the subsequent analyses.</p>
<p>The covariance matrices &#x03A3;<italic><sup>x</sup></italic> were modeled using Wishart distributions with weakly-informative Wishart priors with two degrees of freedom for &#x03A3;<italic><sup>e</sup></italic> and &#x03A3;<italic><sup>s</sup></italic> and four degrees of freedom for &#x03A3;<italic><sup>a</sup></italic> (<xref ref-type="bibr" rid="B17">Gelman et al., 2004</xref>; <xref ref-type="bibr" rid="B19">Greenberg et al., 2011</xref>). This implies a uniform prior on narrow-sense (marker) heritability. Estimated model parameters can then be used to calculate marker heritability:</p>
<disp-formula id="S2.Ex2">
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi mathvariant="normal">&#x0393;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi/>
<mml:mo>&#x22C5;</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi mathvariant="normal">&#x0393;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi/>
<mml:mo>&#x22C5;</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>-</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi mathvariant="normal">&#x0393;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi/>
<mml:mo>&#x22C5;</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi mathvariant="normal">&#x0393;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi/>
<mml:mo>&#x22C5;</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>-</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="normal">&#x03A3;</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mi>s</mml:mi>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mi mathvariant="normal">&#x03A3;</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mi>e</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>N</italic><sub>acc</sub> is the number of accessions, and the rest of the notation is as described above. We used the variance of genome-estimated breeding values (GEBV).</p>
<disp-formula id="S2.Ex3">
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi mathvariant="normal">&#x0393;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi/>
<mml:mo>&#x22C5;</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>U</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi mathvariant="normal">&#x0393;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi/>
<mml:mo>&#x22C5;</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>-</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>rather than values from the marker covariance matrix <inline-formula><mml:math id="INEQ10"><mml:msubsup><mml:mi mathvariant="normal">&#x03A3;</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mo>,</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mi>a</mml:mi></mml:msubsup></mml:math></inline-formula>because the latter reflects the additive covariance only when the prior on the principal component regression (see above) is Gaussian. The <inline-formula><mml:math id="INEQ11"><mml:msubsup><mml:mi mathvariant="normal">&#x03A3;</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mo>,</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mi>s</mml:mi></mml:msubsup></mml:math></inline-formula> values estimate the portion of total genetic variance not explained by genotyped markers. They include the non-additive effects and additive effects not tagged by SNPs. The sum of the GEBV and background variance, divided by total phenotypic variance, is broad-sense heritability (<xref ref-type="bibr" rid="B34">Lynch and Walsh, 1998</xref>).</p>
</sec>
<sec id="S2.SS3">
<title>Mixture Model</title>
<p>Using estimates of phenotypic values for each accession, we fit a Gaussian mixture model (<xref ref-type="bibr" rid="B45">Robert, 1996</xref>) using a variational Bayes approach (<xref ref-type="bibr" rid="B36">McGrory and Titterington, 2007</xref>). The model is</p>
<disp-formula id="S2.Ex4">
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msubsup>
<mml:mi mathvariant="normal">&#x03BC;</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2063;</mml:mo>
<mml:mo>&#x22C5;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x223C;</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>C</mml:mi>
</mml:mfrac>
<mml:mo>&#x2062;</mml:mo>
<mml:mrow>
<mml:munderover>
<mml:mo largeop="true" movablelimits="false" symmetric="true">&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>M</mml:mi>
</mml:msub>
</mml:munderover>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">&#x03BC;</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2063;</mml:mo>
<mml:mo>&#x22C5;</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">&#x03A3;</mml:mi>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:msup>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="S2.Ex5">
<mml:math id="M5" display='block'><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mtext>&#x2009;&#x2009;&#x2009;&#x2009;</mml:mtext><mml:msub><mml:mi>z</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>&#x223C;</mml:mo><mml:mi>B</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>n</mml:mi><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mtext>&#x03C0;</mml:mtext><mml:mi>m</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x2009;&#x2009;&#x2009;&#x2009;</mml:mtext><mml:msub><mml:mtext>&#x03C0;</mml:mtext><mml:mi>m</mml:mi></mml:msub><mml:mo>&#x223C;</mml:mo><mml:mi>D</mml:mi><mml:mi>i</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>h</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mtext>&#x03B1;</mml:mtext><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x2009;&#x2009;&#x2009;</mml:mtext><mml:msub><mml:mtext>&#x03BC;</mml:mtext><mml:mrow><mml:mi>m</mml:mi><mml:mo>&#x22C5;</mml:mo></mml:mrow></mml:msub><mml:mo>&#x223C;</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>M</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:mn>0</mml:mn><mml:mo>;</mml:mo><mml:msub><mml:mtext>&#x03BB;</mml:mtext><mml:mn>0</mml:mn></mml:msub><mml:msub><mml:mtext>&#x03A3;</mml:mtext><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mtext>&#x03A3;</mml:mtext><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>&#x223C;</mml:mo><mml:mi>W</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mtext>&#x03A3;</mml:mtext><mml:mmultiscripts><mml:mo>;</mml:mo><mml:mprescripts/><mml:mn>0</mml:mn><mml:mrow></mml:mrow></mml:mmultiscripts><mml:msub><mml:mtext>&#x03C5;</mml:mtext><mml:mn>0</mml:mn></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the notation is the same as for the hierarchical model above and <bold>&#x03BC;</bold><italic><sub><italic>m</italic></sub></italic><bold><italic><sub>&#x22C5;</sub></italic></bold> is a row vector of group means. Each group has a separate mean and covariance matrix. The prior population size <bold>&#x03B1;</bold> is set to a small value, shrinking groups with few members to zero. The number of groups must be set <italic>a priori</italic>.</p>
<p>We perform inference by running the model multiple times from different starting points and picking the best fit, as assessed using the deviance information criterion (DIC, <xref ref-type="bibr" rid="B36">McGrory and Titterington, 2007</xref>). Small DIC values indicate better fit. Our implementation of the model is available as the R package MuGaMix<sup><xref ref-type="fn" rid="footnote4">4</xref></sup>. We use two methods to assess how many groups are statistically supported: the DIC and the number of groups with at least two accessions. We include the pipeline (implemented in R) that runs the mixture model, processes the results, and generates all the plots and summary tables presented in this manuscript in the <xref ref-type="supplementary-material" rid="SM1">Supplementary Material File 1</xref>.</p>
</sec>
<sec id="S2.SS4">
<title>Merged Population Structure Analysis</title>
<p>For population structure analysis, the SNP datasets from <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> and <xref ref-type="bibr" rid="B24">Huang et al. (2012)</xref> were merged into a single dataset. First, the <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> SNPs from <italic>ORSC</italic> and <italic>O. sativa</italic> accessions were imputed using Beagle 5.0 (<xref ref-type="bibr" rid="B7">Browning et al., 2018</xref>). Next the intersection between the SNPs in the imputed <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> and the (already imputed) SNPs in the <xref ref-type="bibr" rid="B24">Huang et al. (2012)</xref> was determined and a merged file was generated using BCFtools version 1.14 (<xref ref-type="bibr" rid="B16">Danecek et al., 2021</xref>). Population structure analysis was done using fastSTRUCTURE version 1.0 (<xref ref-type="bibr" rid="B44">Raj et al., 2014</xref>) with <italic>K</italic> values ranging from 2 to 9.</p>
</sec>
<sec id="S2.SS5">
<title>RFMix Analysis</title>
<p>The local ancestry of the <italic>ORSC</italic> individuals was determined using RFMix (<xref ref-type="bibr" rid="B35">Maples et al., 2013</xref>). The training panel included ten individuals from each subpopulation (W1-W6) with the highest global ancestry as reported by <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> and included individuals from each of the five <italic>O. sativa</italic> subpopulations, <italic>indica</italic> (IND) (10 individuals), <italic>aus</italic>, (AUS) (9 individuals), <italic>temperate japonica</italic> (TEJ) (9 individuals), <italic>tropical japonica</italic> (TRJ) (10 individuals), and <italic>aromatic</italic> (ARO) (6 individuals). Of the 60 <italic>ORSC</italic> individuals in the training panel, 49 were phenotyped in the current study. Prior to RFMix analysis, the genotypic data were imputed and phased using Beagle 5.0 (<xref ref-type="bibr" rid="B7">Browning et al., 2018</xref>).</p>
</sec>
</sec>
<sec id="S3" sec-type="results">
<title>Results</title>
<sec id="S3.SS1">
<title>Evaluation of Phenotypic Data</title>
<p>Broad-sense and marker heritability distributions for each trait and location are presented in <xref ref-type="supplementary-material" rid="TS5">Supplementary Table 5</xref>. Heritabilities range from 0.98 to 0.33 and are generally moderate to high. Markers account for at least 33% of total genetic variance across all traits and locations, but it is the generally high <italic>H</italic><sup>2</sup> that gives confidence in within-site reproducibility of our phenotypic measurements.</p>
<p>Fourteen traits were measured at more than one site. We tested among-site reproducibility of these measurements by estimating correlations among phenotypes evaluated on the same accessions and generally see positive correlations (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 2</xref> and <xref ref-type="supplementary-material" rid="TS6">Supplementary Table 6</xref>), again suggesting that phenotypic measurements are reproducible across sites despite the diverse climatic conditions. One notable exception is panicle number (PNNB). Values of this trait are positively correlated between Cornell and Dale Bumpers (temperate and sub-tropical sites, respectively), but both are negatively correlated with IRRI measurements (tropical site), suggesting a genotype by environment interaction.</p>
</sec>
<sec id="S3.SS2">
<title>Inferring Phenotypic Groups</title>
<p>Using estimates of 32 phenotypic values (binary traits were excluded) for each accession in the IRRI data set (n = 222), we fit a Gaussian mixture model (<xref ref-type="bibr" rid="B45">Robert, 1996</xref>) using a variational Bayes approach (<xref ref-type="bibr" rid="B36">McGrory and Titterington, 2007</xref>). To infer the underlying number of groups that is statistically supported and biologically meaningful, we ran analyses assuming two to 15 groups <italic>a priori</italic> and assessed model fit (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 3</xref>) using the deviance information criterion and the number of non-empty groups. Both metrics lend statistical support for the division of our accessions into as many as 10 phenotypic groups. However, to make biologically relevant inferences we need a parsimonious classification that is not only statistically significant, but also robust to model uncertainty, reproducible, and interpretable given other biological information. The variational Bayes approach we used is fast but produces only point estimates of parameters. To test robustness of these estimates to initial conditions, we started the model fitting process from five independent sets of initial values. Model fit metrics were consistent across initial value sets (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 3</xref>).</p>
<p>Moving beyond model fit estimates, we tested stability of group membership across model specifications and starting value sets. To anchor these inferences, we started by noting taxonomic classification of our accessions, as found in the IRRI-GRIN-Global database. Based on traditional classification methods, accessions in this database are listed as either <italic>O. rufipogon</italic>, <italic>O. nivara</italic>, or, if the taxonomic assignment is uncertain, as <italic>Oryza</italic> spp. or <italic>Oryza</italic> hybr. We combined the latter two categories into one, calling it <italic>Oryza</italic> spp. We tracked each accession, starting from the species designations, across phenotypic groups inferred given different <italic>a priori</italic> group numbers (<italic>N</italic><sub><italic>G</italic></sub>) using a Sankey plot (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 4</xref>; <xref ref-type="bibr" rid="B27">Kennedy and Sankey, 1898</xref>; implemented in the ggsankey R package)<sup><xref ref-type="fn" rid="footnote5">5</xref></sup>. A line, which is like a ribbon in such plots, represents a single accession, while boxes arranged along the <italic>y</italic>-axis are groups. Groups derived from models assuming a different number of groups are shown along the <italic>x</italic>-axis. We can thus follow a given accession as its line connects to each group across models with different <italic>N</italic><sub><italic>G</italic></sub>. If we set <italic>N</italic><sub><italic>G</italic></sub> = 3, the model infers one large group comprising most accessions. The second largest group is composed predominantly of accessions classified as <italic>Oryza</italic> spp. Increasing the <italic>a priori N<sub><italic>G</italic></sub></italic> to four, we can assign most <italic>O. rufipogon</italic> and <italic>O. nivara</italic> accessions to separate groups. If we further increase <italic>N</italic><sub><italic>G</italic></sub>, distinct groups corresponding to the traditional species designations are maintained, with an additional group that comprises most <italic>Oryza</italic> spp. accessions. However, additional small groups inferred with higher <italic>a priori N</italic><sub><italic>G</italic></sub> do not appear to maintain coherent membership across model runs (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 4</xref>). Thus, while there is statistical support for finer classification, these groupings are not robust to model specification. We therefore focused our analyses on inferences assuming four <italic>a priori</italic> groups.</p>
<p>Given that models with <italic>N</italic><sub><italic>G</italic></sub> = 4 appear to be the most parsimonious, while still capturing biologically meaningful stratification of the <italic>ORSC</italic>, we next tested these models for robustness. We started model fitting from 20 sets of initial conditions and averaged posterior probabilities that a given accession belongs to a particular group across runs. We see a set of accessions that reproducibly belong to the P1 or P4 groups (<xref ref-type="fig" rid="F1">Figure 1A</xref>), corresponding to <italic>O. rufipogon</italic> and <italic>O. nivara</italic>, respectively. The P3 group has the least reproducible membership. This is because for some initial conditions, a small number of accessions are assigned to it, while for others these lines are placed in P2 (see <xref ref-type="supplementary-material" rid="S9">Supplementary Material 1</xref> for a detailed analysis of these shifts). This may be due to the approximate nature of the variational Bayes inference. Therefore, we added the posterior probabilities of P2 and P3 assignment in subsequent analyses and called the resulting group P2/P3.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>Phenotypic group composition. <bold>(A)</bold> Bar plot depicting posterior probabilities that a given accession (along the <italic>x</italic>-axis) belongs to each of the four phenotypic groups. The bars are stacked, with the total probability summing to 1. The bars below the x-axis indicate the <italic>O. sativa</italic> genome fraction, with the darker bars having a higher <italic>O. sativa</italic> fraction. (The actual values are in <xref ref-type="supplementary-material" rid="TS2">Supplementary Table 2</xref>) <bold>(B)</bold> Sankey plots comparing compositions of phenotypic groups inferred from data collected at Dale Bumpers and Cornell assuming <italic>N<sub><italic>G</italic></sub></italic> = 3 compared to species designations and the four groups inferred from data collected at IRRI. Each accession is a single &#x201C;ribbon&#x201D; running from left to right and colored according to the grouping to the left.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-787703-g001.tif"/>
</fig>
<p>The resulting P1 phenotype group contains 106 accessions, of which 82 (78%) are classified as perennial <italic>O. rufipogon</italic> species. In contrast, P4 has 73 accessions with 43 (59%) classified as <italic>O. nivara.</italic> About half of the P2/P3 group are designated as <italic>Oryza</italic> spp. in the IRGC, with the remaining accessions about equally divided between <italic>O. nivara</italic> and <italic>O. rufipogon</italic> (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 5</xref> and <xref ref-type="supplementary-material" rid="TS2">Supplementary Table 2</xref>).</p>
<p>To further test the reproducibility of phenotypic group inference, we used data collected at DB and CU. These data sets include different, but overlapping, phenotype sets that were measured on fewer than half the accessions represented in the IRRI data (<xref ref-type="supplementary-material" rid="TS3">Supplementary Tables 2</xref>, <xref ref-type="supplementary-material" rid="TS3">3</xref>). Despite this, we can still identify a similar set of three groups (<xref ref-type="fig" rid="F1">Figure 1B</xref>). This time, setting <italic>N</italic><sub><italic>G</italic></sub> = 3 <italic>a priori</italic> is sufficient to separate <italic>O. rufipogon</italic> and <italic>O. nivara</italic> from each other and from a group that largely corresponds to the <italic>Oryza</italic> spp. designation from IRRI-GRIN-Global. Taking these observations together with the above statistical robustness analyses, we see that a core set of accessions can be reproducibly assigned to a phenotypic group, while inference for others is less certain.</p>
</sec>
<sec id="S3.SS3">
<title>Subsets of Traits Defining Phenotypic Groups</title>
<p>Having established a robust phenotype-driven grouping of <italic>ORSC</italic> accessions, we next wanted to define a minimal combination of phenotypic traits that is sufficient to reproduce these divisions. Our mixture model allows both phenotype means and covariance to vary among groups. We started by ranking traits according to how well their values correlate with our groups (<xref ref-type="fig" rid="F2">Figure 2A</xref>). We used a multivariate regression with group identity probabilities as the response variable and trait values as predictors. Since the last group membership is fully determined by the other three, only P1 and the combined P2 and P3 membership probabilities were used. We estimated regression coefficients and their covariance and used them to calculate a Hotelling <italic>T</italic><sup>2</sup> statistic (<xref ref-type="bibr" rid="B23">Hotelling, 1931</xref>) of each trait-group association. This is a multivariate version of the square of Student&#x2019;s <italic>t</italic> statistic widely used to determine statistical significance of individual regression coefficients. However, in this case we were not interested in the significance of associations. Instead, we ranked traits by their <italic>T</italic><sup>2</sup> values (<xref ref-type="fig" rid="F2">Figure 2A</xref>). Two traits, culm length (CULT) and number of empty spikelets per panicle (UNFILLED), stand out for their association with phenotypic groups. Their Hotelling <italic>T</italic><sup>2</sup> statistics (0.71 for CULT and 0.70 for UNFILLED) are over twice the value of the next seven highest trait-group associations for which <italic>T</italic><sup>2</sup> ranges from 0.17 to 0.28 (<xref ref-type="table" rid="T1">Table 1</xref> and <xref ref-type="supplementary-material" rid="TS7">Supplementary Table 7</xref>). Accessions belonging to the P1 group are on average taller, with P4 the shortest and P2/P3 intermediate. P2/P3 accessions typically have the most empty spikelets per panicle, while P4 accessions have the least (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 6</xref>).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Phenotypic identifiers of group divergence based on the IRRI data. <bold>(A)</bold> Traits sorted by the strength of the association of their values to <italic>N<sub><italic>G</italic></sub></italic> = 4 group assignments based on Hotelling <italic>T</italic><sup>2</sup> values <bold>(B)</bold> Among-trait correlation estimates within each phenotypic group. Squares mark trait pairs, with the three bars within each square colored according to correlation magnitude in the three phenotypic groups. Dots represent probabilities of observing the correlation differences by chance (lower probabilities reflect high confidence of correlation difference). The inset shows a segment of the plot in more detail. (Trait acronyms and actual values are in <xref ref-type="supplementary-material" rid="TS7">Supplementary Table 7</xref>) <bold>(C)</bold> Sankey plot comparing accession membership across groups inferred from trait subsets listed in <xref ref-type="table" rid="T1">Table 1</xref>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-787703-g002.tif"/>
</fig>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p>Phenotypic traits identified by trait value (9 traits), correlation switching (11 traits) and the union of both sets of traits (16 traits) based on the analysis of the phenotypic trait data collected at IRRI.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Descriptive trait name</td>
<td valign="top" align="center">Trait acronym (IRRI)</td>
<td valign="top" align="center">Top traits by value (9)</td>
<td valign="top" align="center">Top traits by correlation (11)</td>
<td valign="top" align="center">Union of value &#x0026; correlation (16)</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Anther length</td>
<td valign="top" align="center">ANTLT</td>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">Awn length</td>
<td valign="top" align="center">AWNLT</td>
<td valign="top" align="center">X</td>
<td/>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">Awn width</td>
<td valign="top" align="center">AWNWD</td>
<td valign="top" align="center">X</td>
<td/>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">Culm diameter</td>
<td valign="top" align="center">CUDI</td>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">Culm length</td>
<td valign="top" align="center">CULT</td>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">Days to 50% heading</td>
<td valign="top" align="center">DTHD</td>
<td/>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">Distance of nearest spikelet to the panicle base</td>
<td valign="top" align="center">DIST</td>
<td valign="top" align="center">X</td>
<td/>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">Flag leaf lamina width</td>
<td valign="top" align="center">FLFWD</td>
<td/>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">Ligule length</td>
<td valign="top" align="center">LIGLT</td>
<td/>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">No. empty spikelets per panicle</td>
<td valign="top" align="center">UNFILLED</td>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">Panicle fertility</td>
<td valign="top" align="center">FERT</td>
<td/>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">Panicle length</td>
<td valign="top" align="center">PNLG</td>
<td/>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">Penultimate (2nd) leaf length</td>
<td valign="top" align="center">2LLT</td>
<td valign="top" align="center">X</td>
<td/>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">Spikelet length</td>
<td valign="top" align="center">SPKLT</td>
<td/>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">Spikelet width</td>
<td valign="top" align="center">SPKWD</td>
<td/>
<td valign="top" align="center">X</td>
<td valign="top" align="center">X</td>
</tr>
<tr>
<td valign="top" align="left">Sterile lemma width</td>
<td valign="top" align="center">STLWD</td>
<td valign="top" align="center">X</td>
<td/>
<td valign="top" align="center">X</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn><p><italic>These were used to construct the groups shown in the Sankey plot (<xref ref-type="fig" rid="F2">Figure 2C</xref>).</italic></p></fn>
</table-wrap-foot>
</table-wrap>
<p>To find pairs of traits that change correlations across groups, we first estimated within-group associations and then used a permutation test (randomly assigning accessions to groups) to assess statistical evidence for across-group variation (<xref ref-type="fig" rid="F2">Figure 2B</xref>). The most significant correlation changes occur between the number of empty spikelets (UNFILLED) and either panicle length (PNLG) or culm diameter (CUDI). These correlations are positive (0.54 and 0.48, respectively) only in P4 (<xref ref-type="supplementary-material" rid="TS7">Supplementary Table 7</xref>).</p>
<p>To test if subsets of traits can reproduce our phenotypic groups, we took 11 traits that belong to pairs that change correlations with probability at least 0.001 (<xref ref-type="table" rid="T1">Table 1</xref>). This set also includes the top two traits (CULT and UNFILLED) whose values correlate best with group assignments. To expand the trait subset, we added seven phenotypes with <italic>T</italic><sup>2</sup> &#x003E; 0.17. Hotelling values showed a marked drop below this cut-off. These two sets have four traits in common [CULT, UNFILLED, CUDI, and anther length (ANTLT)]. We also used the union of the two sets, containing 16 traits, or half the number in the full data set. Although covariance matrices include both correlations and variances, all traits with appreciable among-group variance changes are already accounted for by using correlation and mean differences (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figures 6</xref>, <xref ref-type="supplementary-material" rid="SM2">7</xref>).</p>
<p>We used three data subsets (nine traits associated with phenotypic groups by value only, 11 that switch correlations, and 16 that were associated either by value or correlation) to re-fit our mixture model, setting the <italic>a priori</italic> number of groups at three. We ran our analyses 20 times and averaged posterior probabilities of accessions belonging to each group. Using all three data sets, we consistently see three groups that separate <italic>O. rufipogon</italic> from <italic>O. nivara</italic>. The nine- and 16-trait data sets assign most P2/P3 accessions to P1 (<xref ref-type="fig" rid="F2">Figure 2C</xref>). The 11 traits most likely to change correlations across groups appear to be sufficient to recapitulate the grouping inferred from the whole data set. The two traits, CULT and UNFILLED, most closely aligned with groups by value and are included in this collection.</p>
<p>Our approach identified a minimally sufficient set of traits that captures most of the phenotypic group structure in the IRRI data. This does not mean that other sets of traits could not achieve similar results but an exhaustive search for all possible combinations that could achieve the same result was not undertaken considering time and the computational expense. Instead, we pursued two approaches using the Cornell and Dale Bumpers data sets to find alternative sets of traits that could be used to re-infer the three major phenotypic groups identified using the full data set. First, we used the process outlined above to identify traits associated with the phenotypic groups that were inferred from IRRI data by value or that change correlations with other traits. <xref ref-type="supplementary-material" rid="SM2">Supplementary Figures 8</xref>, <xref ref-type="supplementary-material" rid="SM2">9</xref> visualize the Dale Bumpers data analyses and <xref ref-type="supplementary-material" rid="SM2">Supplementary Figures 10</xref>, <xref ref-type="supplementary-material" rid="SM2">11</xref> visualize the Cornell data analyses with the actual values in <xref ref-type="supplementary-material" rid="TS7">Supplementary Table 7</xref>. This approach did not identify additional trait subsets that could reliably identify phenotypic groups (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figures 12A,C</xref>). Second, we identified traits evaluated at Dale Bumpers and Cornell that are the same or similar to the 11 phenotypes (<xref ref-type="supplementary-material" rid="TS3">Supplementary Table 3</xref>) that defined the grouping patterns in the IRRI data. Re-running our mixture model with these analogous traits reproduced the original groups, albeit at lower resolution (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figures 12B,D</xref>). Using fewer than 11 traits thus appears to significantly degrade group assignment precision and reproducibility.</p>
</sec>
<sec id="S3.SS4">
<title>Relationships Between Phenotypic Groups and Genetic Subpopulations</title>
<p>While our phenotypic groups largely correspond to IRRI-GRIN-Global species designations, we wanted to know how the groups related to previously characterized genotypic subpopulations. The accessions used in this study were genotyped by <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> who reported six <italic>ORSC</italic> subpopulations (W1-W6). Based on RFMix (<xref ref-type="bibr" rid="B35">Maples et al., 2013</xref>) analysis, we identified introgressions among <italic>ORSC</italic> subpopulations and between the <italic>ORSC</italic> and five <italic>O. sativa</italic> subpopulations (IND, AUS, TRJ, TEJ, ARO). We partitioned accessions with more than half of their genome introgressed from <italic>O. sativa</italic> into a separate &#x201C;admixture with <italic>O. sativa</italic>&#x201D; group (ADM/OSAT). By means of a Sankey plot, we related the species designation, phenotypic group and genetic subpopulation for the <italic>ORSC</italic> accessions with complete information (<xref ref-type="fig" rid="F3">Figure 3A</xref> and <xref ref-type="supplementary-material" rid="TS2">Supplementary Table 2</xref>). Across the <italic>ORSC</italic> subpopulations, the majority of the W1, all W3, and about half of the W6 accessions (as per <xref ref-type="bibr" rid="B29">Kim et al., 2016</xref>) clustered in the P1 group, taxonomically designated as <italic>O. rufipogon</italic>. Conversely, all of the W5 accessions and most of the W2 clustered in the P4 group, taxonomically designated as <italic>O. nivara</italic>. Finally, accessions with significant <italic>O. sativa</italic> introgression fell almost exclusively into the P2/P3 group and are classified as <italic>Oryza</italic> spp. The P2/P3 group identified using the reduced set of 11 traits also includes most of the W6 subpopulation (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 13</xref>). Eight accessions taxonomically classified as <italic>O. rufipogon</italic> in the IRGC were genetically designated as W2 or W4 by <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> and clustered with the P4 group in this study. These accessions had less than 20% of their genome from <italic>O. sativa</italic> and were re-classified as <italic>O. nivara</italic> in our analyses (including in the results reported above) and are identified in <xref ref-type="supplementary-material" rid="TS2">Supplementary Table 2</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>Genotypic composition, local ancestry and phenotypic groups of <italic>ORSC</italic> accessions. <bold>(A)</bold> Sankey plot depicting correspondence between traditional species designations, our phenotypic groups, and <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> genetic subpopulations using 32 phenotypic traits. (ADM/OSAT are accessions with a high proportion of admixture with <italic>O. sativa</italic>) <bold>(B)</bold> Graphical genotypes display the local ancestry of each <italic>ORSC</italic> individual as determined by RFMix trained on six <italic>ORSC</italic> subpopulations (W1-W6) and five <italic>O. sativa</italic> subpopulations: <italic>indica</italic> (IND), <italic>aus</italic>, (AUS), <italic>temperate japonica</italic> (TEJ), <italic>tropical japonica</italic> (TRJ), and <italic>aromatic</italic> (ARO). The accessions are arranged according to global genotypic ancestry (left) and phenotypic group assignment (right). Each individual&#x2019;s genotype is represented by two adjoining rows corresponding to phased haplotypes with colors indicating RFMix population assignments of chromosome segments <bold>(C)</bold> Proportion of the <italic>ORSC</italic> subpopulations (left) and <italic>O. sativa</italic> subpopulations (right) within each of the three phenotypic classes.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-787703-g003.tif"/>
</fig>
<p>Surveying the individual introgressions across the genome, there were no obvious patterns identifying particular genomic regions underlying the phenotypic group differentiation (<xref ref-type="fig" rid="F3">Figure 3B</xref>). This result was not surprising, given that these groups are distinguished by multiple polygenic traits. We do see that those accessions from the P2/P3 group were more likely to harbor <italic>O. sativa</italic> introgressions (<xref ref-type="fig" rid="F3">Figures 3B,C</xref>) and the median portion of the genome coming from <italic>O. sativa</italic> is also larger in P2/P3 than in P1 or P4 accessions (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figures 14B</xref>, <xref ref-type="supplementary-material" rid="SM2">15</xref>). However, P1 and to a lesser extent P4 harbor a few accessions with significant contributions from <italic>O. sativa</italic> or <italic>ORSC</italic> subpopulations typically associated with a different phenotypic group, suggesting that similar phenotypic syndromes can be achieved through divergent genotypes.</p>
<p>Most <italic>O. sativa</italic> introgressions come from the <italic>indica</italic> subpopulation (<xref ref-type="fig" rid="F3">Figure 3</xref> and <xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 14B</xref>). Only accessions belonging to the P2/P3 group show appreciable contributions from <italic>O. sativa japonica</italic>. Introgressions from <italic>aus</italic> are relatively more prevalent in accessions from both P2/P3 and P4.</p>
</sec>
<sec id="S3.SS5">
<title>Genomic Analysis of Merged <italic>ORSC</italic> Datasets and Geographic Distribution</title>
<p>To compare the population structure of the collection of <italic>ORSC</italic> analyzed by <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> (<italic>n</italic> = 286) with that generated by <xref ref-type="bibr" rid="B24">Huang et al. (2012)</xref> and subsequently analyzed by <xref ref-type="bibr" rid="B57">Wang et al. (2017)</xref> (<italic>n</italic> = 435), a merged SNP dataset was constructed and analyzed using fastSTRUCTURE (<xref ref-type="bibr" rid="B44">Raj et al., 2014</xref>). The combined dataset contained 55,213 SNPs shared between them. Structure analysis of the merged data resulted in six subpopulations that match nearly perfectly with the six subpopulations reported by <xref ref-type="bibr" rid="B57">Wang et al. (2017)</xref>, (<xref ref-type="fig" rid="F4">Figure 4A</xref>). The six subpopulations, designated W1 to W6 by <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> have a clear connection to those designated <italic>Or-A</italic> to <italic>Or-F</italic> by <xref ref-type="bibr" rid="B57">Wang et al. (2017)</xref> as illustrated by shared merged groups. The main differences are due to the thresholds used to classify accessions designated as <italic>admixed</italic>, and the different size and composition of the two collections. Where <xref ref-type="bibr" rid="B57">Wang et al. (2017)</xref> used a threshold of &#x003E; 80% ancestry to classify accessions as belonging to one of the wild groups, <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> used a threshold of &#x003E;60%. As a result, roughly half of the accessions designated as W1 by <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> were identified as <italic>admixed</italic> in the combined analysis. It is also noteworthy that <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> had a smaller proportion of wild accessions from China (W6), and a larger proportion from Papua, New Guinea (W3) and from Nepal (W5), which impacted the subpopulation structure that emerged in the analysis. Nonetheless, there is a clear correspondence between genetic groups in the two studies, with W1 and W3 (predominantly <italic>O. rufipogon</italic> as described by <xref ref-type="bibr" rid="B29">Kim et al., 2016</xref>) corresponding to <italic>Or-A</italic> (<xref ref-type="bibr" rid="B57">Wang et al., 2017</xref>), W6 (Chinese <italic>O. rufipogon</italic>) to <italic>Or-B</italic>, and W2 (predominantly <italic>O. nivara</italic>) to <italic>Or-C</italic>. <italic>W4</italic> (<italic>aus</italic>-like wild ancestor as described by <xref ref-type="bibr" rid="B29">Kim et al., 2016</xref>) is split between <italic>Or-D</italic> (wild) and <italic>Or-E</italic> (feral), and interestingly, the accessions clustered as <italic>Or-E</italic> by <xref ref-type="bibr" rid="B57">Wang et al. (2017)</xref> were recognized as a separate group by <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> when a higher <italic>K</italic> value (<italic>K</italic> = 8) was used in the analysis.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption><p>Comparison of <italic>ORSC</italic> genotypic subpopulations reported by <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> based on 286 <italic>ORSC</italic> accessions to those reported by <xref ref-type="bibr" rid="B57">Wang et al. (2017)</xref> and <xref ref-type="bibr" rid="B24">Huang et al. (2012)</xref> based on 446 <italic>ORSC</italic> accessions. <bold>(A)</bold> Population structure assignments in a merged SNP dataset consisting of accessions from two <italic>ORSC</italic> collections. On the left side are the accessions from <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref>, on the right are accessions from the analysis by <xref ref-type="bibr" rid="B57">Wang et al. (2017)</xref>. In the center is a merged set with group assignments at <italic>K</italic> = 7. Ribbons connect the group assignments of the merged accessions with their group assignments in the respective <italic>ORSC</italic> collections <bold>(B)</bold> Comparison of the three subpopulations originally identified by <xref ref-type="bibr" rid="B24">Huang et al. (2012)</xref> to the six subpopulations and admixed group identified by <xref ref-type="bibr" rid="B57">Wang et al. (2017)</xref> when reanalyzing the same genotypic data. The inclusion of <italic>aus</italic>-like (<italic>Or-E</italic>) and <italic>indica</italic>-like (<italic>Or-F</italic>) accessions in <italic>Or-I</italic> and <italic>japonica</italic>-like (<italic>Or-B</italic>) accessions in <italic>Or-III</italic> as reported by <xref ref-type="bibr" rid="B24">Huang et al. (2012)</xref> is confirmed.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-13-787703-g004.tif"/>
</fig>
<p>The groups identified by <xref ref-type="bibr" rid="B57">Wang et al. (2017)</xref> also show reasonable correspondence to the original three groups proposed by <xref ref-type="bibr" rid="B24">Huang et al. (2012)</xref>, (<xref ref-type="fig" rid="F4">Figure 4B</xref>), though <xref ref-type="bibr" rid="B24">Huang et al. (2012)</xref> did not explicitly call out the large proportion of <italic>ORSC</italic> accessions that are highly admixed with <italic>O. sativa</italic> as both <xref ref-type="bibr" rid="B57">Wang et al. (2017)</xref> and <xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> did. Given the different compositions of the two germplasm collections, this merged genomic analysis allows us to interpret our findings about the three phenotypic groups in a larger context.</p>
<p>There is support for geographic subpopulation structure whereby accessions collected from the Malay Archipelago (Malaysia, Indonesia, Philippines, Papua New Guinea), which represents the southeastern extreme of the geographic distribution of the <italic>ORSC</italic>, are phenotypically classified as P1 in this study; they are also predominantly classified as <italic>O. rufipogon</italic> in the IRGC (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 16</xref>). Accessions from India, Nepal, and Sri Lanka, representing the northwestern extreme of the geographic distribution, are predominantly classified as P4 in this study and <italic>O. nivara</italic> in the IRGC. Accessions collected from China were genetically classified as <italic>O. rufipogon</italic> with substantial admixture from <italic>O. sativa</italic> (<xref ref-type="bibr" rid="B29">Kim et al., 2016</xref>), but they were phenotypically more similar to annual <italic>O. nivara</italic> in this study. Accessions collected across the coastal regions of South and Southeast Asia are recognized phenotypically as P2/P3 and are typically the most highly admixed with <italic>O. sativa</italic>. A great diversity of phenotypic variation is found in mainland Southeast Asia where <italic>ORSC</italic> accessions manifesting both P1 and P4 phenotypic syndromes co-exist, often sympatrically or in contiguous environments. These geographical distributions of <italic>O. rufipogon</italic> and <italic>O. nivara</italic> are clearly shown in <xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 16B</xref> where only the locations of the &#x201C;core&#x201D; accessions, those for which the species, phenotypic group and genotypic subpopulation agreed, are shown. The fact that cultivated forms of rice, predominantly representing the <italic>indica</italic> subpopulation, overlap with wild rice habitats in many parts of Asia creates endless possibilities for cross fertilization and the emergence of novel phenotypes. In addition to their inherent value and interest, these naturally occurring interspecific populations represent valuable, ecologically viable pools of variation for plant breeders interested in identifying novel forms of climate resilience for crop improvement.</p>
</sec>
</sec>
<sec id="S4" sec-type="discussion">
<title>Discussion</title>
<sec id="S4.SS1">
<title>Trait Significance</title>
<p>The traits measured in this study emphasize highly heritable, easily measured morphological and physiological traits that can be readily observed in populations of crop wild relatives grown under controlled conditions. Many of the same traits are evaluated by genebanks and widely used for identification, tracking and management of <italic>ex situ</italic> germplasm collections. Of the traits measured here, four were significantly associated with phenotypic groups and with life history; these included days to heading, plant height, percent filled spikelets per panicle (seeds/panicle), and spikelet/seed size (related to grain weight). These traits provided information that was valuable for differentiating annual and perennial life habits and were similar to traits evaluated by <xref ref-type="bibr" rid="B20">Grillo et al. (2009)</xref> in two <italic>O. rufipogon</italic>/<italic>O. nivara</italic> mapping populations. We also measured several traits that were potentially informative about mating habit, including anther length, style length and stigma length, but only anther length discriminated the groups based on both Hotelling <italic>T</italic><sup>2</sup> value and correlation estimates (<xref ref-type="table" rid="T1">Table 1</xref>). These traits related to life history and mating habit support our understanding of the biological significance of our phenotypic groups.</p>
</sec>
<sec id="S4.SS2">
<title>Traits That Differentiate Phenotypic Groups</title>
<p>The single most informative trait for distinguishing <italic>ORSC</italic> phenotypic groups based on the Hotelling <italic>T</italic><sup>2</sup> values across all three datasets was seed width (HULGRWD) (<italic>T</italic><sup>2</sup> value = 1.15) measured at Dale Bumpers, which served as a proxy for spikelet width (SPKWD) measured at IRRI, and seed volume, (<italic>T</italic><sup>2</sup> value = 1.70) calculated from seed width and length. Spikelet width and seed width were highly correlated (<italic>r</italic> = 0.62) and both width and volume were informative for distinguishing groups. Seed traits were not measured at Cornell and therefore could not be used for further confirmation.</p>
<p>Comparing this analysis to the principle component analysis (PCA) of phenotypic data collected at IRRI on a different set of 116 <italic>ORSC</italic> accessions, <xref ref-type="bibr" rid="B1">Banaticla-Hilario et al. (2013a)</xref> also reported SPKWD as one of the most important characters separating <italic>O. nivara</italic> and <italic>O. rufipogon</italic> accessions. The other important traits included anther length (ANTLT), days to heading (DTHD) and culm length (CULT). Of note, spikelet fertility was not included in the PCA because it was highly correlated with anther length and spikelet width, leading to the conclusion that anther length and spikelet width serve as a proxy for fertility when plants are grown under controlled screenhouse conditions. Our data on DTHD collected at IRRI differentiated P1 from the other groups. The trend is the same in the Cornell and the Dale Bumpers data, but with more overlap between groups. This may be a consequence of evaluating DTHD during the short days of fall at higher latitudes. Across our three environments, unfilled grain number (UNFILGRNB) in the IRRI and Dale Bumpers data and seed weight (Seed_S2) collected from bagged and unbagged panicles at Cornell provided useful proxies for spikelet fertility. Lastly, culm length (CULT) which was measured from the soil surface to the panicle base at IRRI and plant height (PTHT) measured from the soil surface to the tip (end) of the panicle at Dale Bumpers and Cornell were prominent differentiators across all three studies. This comparison confirms that many traits distinguishing the phenotypic groups can be identified across different environments and panels of <italic>ORSC</italic> accessions. These observations are confirmed by earlier studies summarized in <xref ref-type="supplementary-material" rid="TS1">Supplementary Table 1</xref> in which <italic>O. nivara</italic> was characterized by having wider seeds, higher seed production (fertility), shorter culm length (plant height), and earlier flowering (due to photoperiod insensitivity) compared to <italic>O. rufipogon</italic> which was characterized by relatively narrow seeds, low seed production, longer culm length (taller plant) and later flowering (due to photoperiod sensitivity) (<xref ref-type="bibr" rid="B38">Morishima et al., 1961</xref>; <xref ref-type="bibr" rid="B43">Oka and Morishima, 1967</xref>; <xref ref-type="bibr" rid="B3">Barbier, 1989a</xref>).</p>
<p>Considering the high information content of SPKWD (spikelet/seed width) as a distinguishing feature of the phenotypic groups identified in this study and its utility for differentiating the <italic>O. rufipogon</italic> and <italic>O. nivara</italic> species, we recommend including seed width and length in the suite of traits collected on <italic>ORSC</italic> accessions in genebanks around the world. With the availability of scanners and appropriate software, this would be an inexpensive investment that would help improve the classification of <italic>ORSC</italic> accessions, making collections of these crop wild relatives more valuable to users.</p>
</sec>
<sec id="S4.SS3">
<title>&#x201C;Core&#x201D; <italic>Oryza rufipogon</italic> and <italic>Oryza nivara</italic> Accessions and Admixture With <italic>Oryza sativa</italic></title>
<p>Of the <italic>ORSC</italic> accessions phenotyped in this study, approximately 41% (&#x223C;98 accessions, including the 49 extracted for use as training sets) had less than 5% introgression from <italic>O. sativa</italic> and can be considered non-admixed and wild. Most of these (<italic>n</italic> = 91) clustered in either P1 (<italic>n</italic> = 48) where most were classified as <italic>O. rufipogon</italic> or in P4 (<italic>n</italic> = 43) where most were classified as <italic>O. nivara</italic>, while a few (<italic>n</italic> = 7) clustered in P2/P3.</p>
<p>Based on analyses of both phenotypic and genotypic data using different model parameters, we identified 42 &#x201C;core&#x201D; accessions that are consistently classified as perennial <italic>O. rufipogon</italic> and 25 &#x201C;core&#x201D; accessions consistently classified as annual <italic>O. nivara</italic>. These &#x201C;core&#x201D; accessions all have &#x003C;20% admixture from <italic>O. sativa</italic> and intersect with the group carrying &#x003C;5% <italic>O. sativa</italic> introgression. This collection of 67 accessions provides a useful subset of <italic>ORSC</italic> materials for studying the genetic basis of the phenotypic divergence that distinguishes the two ecotypes of wild rice in Asia.</p>
<p><italic>ORSC</italic> accessions that are highly admixed with <italic>O. sativa</italic> are also of interest. In this study, 25% of accessions were classified as admixed because they had &#x003E;40% of the genome introgressed from <italic>O. sativa.</italic> In the previous study by <xref ref-type="bibr" rid="B57">Wang et al. (2017)</xref> using a different panel of <italic>ORSC</italic> accessions and a different threshold for determining admixture, 42% of wild rice samples were reported to be substantially admixed with &#x003E;20% of the genome showing introgression from <italic>O. sativa.</italic> In both studies, <italic>indica</italic> and <italic>aus</italic> introgressions accounted for the largest proportion, and <italic>japonica</italic> introgressions were rare. Admixed accessions found in P1 and P4 in this study had an average of 19.8% and 19.0% of their genomes comprised of introgressions from <italic>O. sativa</italic>, respectively, while levels of admixture in P2/P3 averaged 52.6%; indeed, two P2/P3 accessions carried &#x223C;85% <italic>O. sativa</italic> DNA (<xref ref-type="fig" rid="F3">Figure 3C</xref> and <xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 15</xref>). Nonetheless, the genome-wide level of admixture alone, is not a good predictor of phenotype, given that 13 accessions classified phenotypically into P1 carry &#x003E; 40% <italic>O. sativa</italic> DNA, and 9 accessions classified phenotypically as P4 carry &#x003E; 40% <italic>O. sativa</italic> DNA. We therefore infer that it is the particular distribution of <italic>O. sativa</italic> introgressions across the genome and the subpopulation ancestry of each introgression, in combination with the <italic>ORSC</italic> genetic background that determines the phenotypic outcome. Our evaluation of this <italic>ORSC</italic> collection provides the foundation for future research to examine the genotype-phenotype relationships for specific traits of interest.</p>
</sec>
<sec id="S4.SS4">
<title>SINE-Codes and Chloroplast Markers as Predictors of Genotypic, Species and Phenotypic Groups</title>
<p>In rice, SINE codes have been associated with life history (annual, intermediate or perennial growth habit) and found useful for distinguishing <italic>O. nivara</italic> from <italic>O. rufipogon</italic> (<xref ref-type="bibr" rid="B40">Motohashi et al., 1997</xref>; <xref ref-type="bibr" rid="B12">Cheng et al., 2003</xref>; <xref ref-type="bibr" rid="B11">Chen et al., 2004</xref>; <xref ref-type="bibr" rid="B59">Xu et al., 2007</xref>). <xref ref-type="bibr" rid="B28">Kim (2016)</xref> developed a SINE code and used it to classify most of the <italic>ORSC</italic> accessions used in this study (<xref ref-type="supplementary-material" rid="TS2">Supplementary Table 2</xref>). Using the SINE code, 74.5% of the accessions classified as P1 were identified as perennial (compared to 72.5% of accessions classified as <italic>O. rufipogon</italic>), and 62.5% of accessions classified as P4 were identified as annual or intermediate (compared to 61.7% classified as <italic>O. nivara</italic>). In P2/P3, 47.4% of accessions were classified as perennial and 52.6% as annual or intermediate using the SINE code (compared to 23.3% classified as <italic>O. rufipogon</italic>, 25.6% as <italic>O. nivara</italic> in P2/P3 with the remainder identified as <italic>Oryza</italic> spp.). Thus, the SINE-based classifications approximated the species designations in these groups, though they were not entirely concordant.</p>
<p><xref ref-type="bibr" rid="B29">Kim et al. (2016)</xref> further analyzed 286 <italic>ORSC</italic> accessions for variation at 25 polymorphic sites in the chloroplast genome and identified unique chloroplast haplotypes associated with three wild subpopulation groups located at the geographical extremes of the species range. These may be of interest for investigating refuge populations of <italic>ORSC</italic> and elucidating phylogenetic relationships among wild populations of <italic>Oryza</italic> but chloroplast haplotype analysis alone was not robust enough to be useful as a predictor of genetic subpopulation, life habit or species in the <italic>ORSC</italic>. Nevertheless, the chloroplast data provided clear evidence of gene flow between annual and perennial populations, as well as between <italic>ORSC</italic> and <italic>O. sativa</italic> populations, and confirmed that it occurs through both pollen dissemination and seed dispersal (likely facilitated by human migration). Taken together, these data suggest that the population structure of the <italic>ORSC</italic> is the result of complex evolutionary pathways that intersect and loop back upon each other due to highly permeable &#x2018;species&#x2019; boundaries and, as documented in this study, highly plastic phenotypic variation. Information about SINE-code and chloroplast haplotype variation (as determined by <xref ref-type="bibr" rid="B29">Kim et al., 2016</xref>) corresponding to each accession is summarized in <xref ref-type="supplementary-material" rid="TS2">Supplementary Table 2</xref>. These features, along with geographical and ecological information about where each accession was collected, disease and insect resistance, grain quality and additional use-data provided by the IRGC database add value to these genetic resources and enhance our understanding of where they come from and how they might be used in the future.</p>
</sec>
<sec id="S4.SS5">
<title>Emergent Phenotypes</title>
<p>The P2/P3 group is characterized by 43 accessions with trait combinations that form a coherent group, despite the fact that they do not match either of the <italic>a priori</italic> taxonomic species descriptions. This group is comprised of 29 (67%) admixed <italic>ORSC</italic> accessions with &#x003E; 40% of the genome derived from <italic>O. sativa</italic> but it also contains accessions classified as annual <italic>O. nivara</italic> or perennial <italic>O. rufipogon</italic>. The fact that the P2/P3 accessions were collected in diverse ecological and geographic regions, especially from coastal regions of China, Southeast and South Asia (<xref ref-type="supplementary-material" rid="SM2">Supplementary Figure 16</xref>), and yet share a common suite of emergent phenotypes suggests that particular combinations of traits may evolve repeatedly and independently when annual, perennial and intermediate forms of common wild rice come together as sympatric swarms in unsupervised settings. The patterns of admixture observed in this group suggest that diverse <italic>ORSC</italic> populations have hybridized among themselves and with local populations of <italic>O. sativa</italic>, giving rise to forms of trait variation not observed in the parental populations. In rice, several lines of evidence suggest that annual and intermediate forms have emerged repeatedly from genetically and geographically diverse populations of perennial ancestors in response to changing patterns of temperature, rainfall and CO<sub>2</sub> (<xref ref-type="bibr" rid="B20">Grillo et al., 2009</xref>; <xref ref-type="bibr" rid="B1">Banaticla-Hilario et al., 2013a</xref>,<xref ref-type="bibr" rid="B2">b</xref>; <xref ref-type="bibr" rid="B32">Liu et al., 2015</xref>). The <italic>ORSC</italic> populations examined here provide opportunities to deepen our understanding of incipient speciation and to identify mechanisms by which annual life forms may evolve from perennial ancestors in response to changes in the environment. They also provide material for examining what kinds of selection pressure(s) are associated with a shift toward the annual, early-flowering, seed-bearing habit or conversely, serve to maintain the perennial, stoloniferous, late-flowering and vegetatively vigorous habit. During times of climate change, a strategy of balancing selection to maintain components of annual, intermediate and perennial life forms may be favored by evolution to allow for the emergence of new resilience mechanisms. We have a lot to learn from the existence of these dynamic <italic>ORSC</italic> populations. They have survived many waves of climate change in the past and are likely to hold secrets about survival strategies for the future.</p>
</sec>
</sec>
<sec id="S5" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>The original contributions presented in this study are included in the article/<xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref>, further inquiries can be directed to the corresponding author/s.</p>
</sec>
<sec id="S6">
<title>Author Contributions</title>
<p>KM, MEN, MB-H, GCE, HJK, JJ, SH, JK, and LH phenotyped the plants. YS managed the data. AG, JE, GCE, HJK, JJ, and SRM analyzed the data. GCE, AG, JE, and SRM wrote the manuscript. SRM conceptualized the project. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="conf1" sec-type="COI-statement">
<title>Conflict of Interest</title>
<p>AG is employed by Bayesic Research LLC. The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="pudiscl1" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<sec id="S7" sec-type="funding-information">
<title>Funding</title>
<p>This study was funded in part by the National Science Foundation with grants to SRM and GCE (NSF-PGRP nos. 0606461 and 1026555). Mention of a trademark or proprietary product does not constitute a guarantee or warranty of the product by the US Department of Agriculture and does not imply its approval to the exclusion of other products that also can be suitable. The USDA is an equal opportunity provider and employer.</p>
</sec>
<ack><p>The excellent technical assistance of Daniel E. Wood in collecting the phenotypic data and the assistance of Teresa A. Hancock with organizing and conducting the preliminary analyses of the data collected at the Dale Bumpers National Rice Research Center, is acknowledged. The assistance of Aaron K. Jackson with the manuscript references is acknowledged.</p>
</ack>
<sec id="S9" sec-type="supplementary-material">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2022.787703/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2022.787703/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.pdf" id="TS1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_2.xlsx" id="TS2" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_3.xlsx" id="TS3" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_4.xlsx" id="TS4" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_5.xlsx" id="TS5" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_6.xlsx" id="TS6" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_7.xlsx" id="TS7" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_1.ZIP" id="SM1" mimetype="application/zip" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_2.pdf" id="SM2" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Banaticla-Hilario</surname> <given-names>M. C. N.</given-names></name> <name><surname>Sosef</surname> <given-names>M. S. M.</given-names></name> <name><surname>McNally</surname> <given-names>K. L.</given-names></name> <name><surname>Hamilton</surname> <given-names>N. R. S.</given-names></name> <name><surname>van den Berg</surname> <given-names>R. G.</given-names></name></person-group> (<year>2013a</year>). <article-title>Ecogeographic variation in the morphology of two Asian wild rice species, <italic>Oryza nivara</italic> and <italic>Oryza rufipogon</italic>.</article-title> <source><italic>Int. J. Plant Sci.</italic></source> <volume>174</volume> <fpage>896</fpage>&#x2013;<lpage>909</lpage>. <pub-id pub-id-type="doi">10.1086/670370</pub-id></citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Banaticla-Hilario</surname> <given-names>M. C.</given-names></name> <name><surname>van den Berg</surname> <given-names>R. G.</given-names></name> <name><surname>Hamilton</surname> <given-names>N. R.</given-names></name> <name><surname>McNally</surname> <given-names>K. L.</given-names></name></person-group> (<year>2013b</year>). <article-title>Local differentiation amidst extensive allele sharing in <italic>Oryza nivara</italic> and <italic>O. rufipogon</italic>.</article-title> <source><italic>Ecol. Evol.</italic></source> <volume>3</volume> <fpage>3047</fpage>&#x2013;<lpage>3062</lpage>. <pub-id pub-id-type="doi">10.1002/ece3.689</pub-id> <pub-id pub-id-type="pmid">24101993</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barbier</surname> <given-names>P.</given-names></name></person-group> (<year>1989a</year>). <article-title>Genetic variation and ecotypic differentiation in the wild rice species <italic>Oryza rufipogon</italic>. I. Population differentiation in life-history traits and isozymic loci.</article-title> <source><italic>Jpn. J. Genet.</italic></source> <volume>64</volume> <fpage>259</fpage>&#x2013;<lpage>271</lpage>. <pub-id pub-id-type="doi">10.1266/jjg.64.259</pub-id></citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barbier</surname> <given-names>P.</given-names></name></person-group> (<year>1989b</year>). <article-title>Genetic variation and ecotypic differentiation in the wild rice species <italic>Oryza rufipogon</italic> II. Influence of the mating system and life-history traits on the genetic structure of populations.</article-title> <source><italic>Jpn. J. Genet.</italic></source> <volume>64</volume> <fpage>273</fpage>&#x2013;<lpage>285</lpage>. <pub-id pub-id-type="doi">10.1266/jjg.64.273</pub-id></citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barbier</surname> <given-names>P.</given-names></name> <name><surname>Morishima</surname> <given-names>H.</given-names></name> <name><surname>Ishihama</surname> <given-names>A.</given-names></name></person-group> (<year>1991</year>). <article-title>Phylogenetic relationships of annual and perennial wild rice: probing by direct DNA sequencing.</article-title> <source><italic>Theor. Appl. Genet.</italic></source> <volume>81</volume> <fpage>693</fpage>&#x2013;<lpage>702</lpage>. <pub-id pub-id-type="doi">10.1007/bf00226739</pub-id> <pub-id pub-id-type="pmid">24221388</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><collab>Bioversity International, IRRI, and Africa Rice Center</collab> (<year>2007</year>). <source><italic>Descriptors for Wild and Cultivated Rice (Oryza spp.).</italic></source> <publisher-loc>Rome</publisher-loc>: <publisher-name>Bioversity International</publisher-name>.</citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Browning</surname> <given-names>B. L.</given-names></name> <name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Browning</surname> <given-names>S. R.</given-names></name></person-group> (<year>2018</year>). <article-title>A one-penny imputed genome from next-generation reference panels.</article-title> <source><italic>Am. J. Hum. Genet.</italic></source> <volume>103</volume> <fpage>338</fpage>&#x2013;<lpage>348</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2018.07.015</pub-id> <pub-id pub-id-type="pmid">30100085</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cai</surname> <given-names>H. W.</given-names></name> <name><surname>Wang</surname> <given-names>X. K.</given-names></name> <name><surname>Morishima</surname> <given-names>H.</given-names></name></person-group> (<year>2004</year>). <article-title>Comparison of population genetic structures of common wild rice (<italic>Oryza rufipogon</italic> Griff.), as revealed by analyses of quantitative traits, allozymes, and RFLPs.</article-title> <source><italic>Heredity (Edinb.)</italic></source> <volume>92</volume> <fpage>409</fpage>&#x2013;<lpage>417</lpage>. <pub-id pub-id-type="doi">10.1038/sj.hdy.6800435</pub-id> <pub-id pub-id-type="pmid">14997180</pub-id></citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chang</surname> <given-names>T. T.</given-names></name></person-group> (<year>1976</year>). <article-title>The origin, evolution, cultivation, dissemination, and diversification of Asian and African rices.</article-title> <source><italic>Euphytica</italic></source> <volume>25</volume> <fpage>425</fpage>&#x2013;<lpage>441</lpage>. <pub-id pub-id-type="doi">10.1007/BF00041576</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chatterjee</surname> <given-names>D.</given-names></name></person-group> (<year>1948</year>). <article-title>A modified key and enumeration of the species of <italic>Oryza</italic> Linn.</article-title> <source><italic>Indian J. Agric. Sci.</italic></source> <volume>18</volume> <fpage>185</fpage>&#x2013;<lpage>192</lpage>.</citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>L. J.</given-names></name> <name><surname>Lee</surname> <given-names>D. S.</given-names></name> <name><surname>Song</surname> <given-names>Z. P.</given-names></name> <name><surname>Suh</surname> <given-names>H. S.</given-names></name> <name><surname>Lu</surname> <given-names>B. R.</given-names></name></person-group> (<year>2004</year>). <article-title>Gene flow from cultivated rice (<italic>Oryza sativa</italic>) to its weedy and wild relatives.</article-title> <source><italic>Ann. Bot.</italic></source> <volume>93</volume> <fpage>67</fpage>&#x2013;<lpage>73</lpage>. <pub-id pub-id-type="doi">10.1093/aob/mch006</pub-id> <pub-id pub-id-type="pmid">14602665</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cheng</surname> <given-names>C.</given-names></name> <name><surname>Motohashi</surname> <given-names>R.</given-names></name> <name><surname>Tsuchimoto</surname> <given-names>S.</given-names></name> <name><surname>Fukuta</surname> <given-names>Y.</given-names></name> <name><surname>Ohtsubo</surname> <given-names>H.</given-names></name> <name><surname>Ohtsubo</surname> <given-names>E.</given-names></name></person-group> (<year>2003</year>). <article-title>Polyphyletic origin of cultivated rice: based on the interspersion pattern of SINEs.</article-title> <source><italic>Mol. Biol. Evol.</italic></source> <volume>20</volume> <fpage>67</fpage>&#x2013;<lpage>75</lpage>. <pub-id pub-id-type="doi">10.1093/molbev/msg004</pub-id> <pub-id pub-id-type="pmid">12519908</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Choi</surname> <given-names>J. Y.</given-names></name> <name><surname>Purugganan</surname> <given-names>M. D.</given-names></name></person-group> (<year>2018</year>). <article-title>Multiple origin but single domestication led to <italic>Oryza sativa</italic>.</article-title> <source><italic>G3</italic></source> <volume>8</volume> <fpage>797</fpage>&#x2013;<lpage>803</lpage>. <pub-id pub-id-type="doi">10.1534/g3.117.300334</pub-id> <pub-id pub-id-type="pmid">29301862</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Civ&#x00E1;n</surname> <given-names>P.</given-names></name> <name><surname>Craig</surname> <given-names>H.</given-names></name> <name><surname>Cox</surname> <given-names>C. J.</given-names></name> <name><surname>Brown</surname> <given-names>T. A.</given-names></name></person-group> (<year>2015</year>). <article-title>Three geographically separate domestications of Asian rice.</article-title> <source><italic>Nat. Plants</italic></source> <volume>1</volume>:<issue>15164</issue>. <pub-id pub-id-type="doi">10.1038/nplants.2015.164</pub-id> <pub-id pub-id-type="pmid">27251535</pub-id></citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cooper</surname> <given-names>L.</given-names></name> <name><surname>Meier</surname> <given-names>A.</given-names></name> <name><surname>Laporte</surname> <given-names>M. A.</given-names></name> <name><surname>Elser</surname> <given-names>J. L.</given-names></name> <name><surname>Mungall</surname> <given-names>C.</given-names></name> <name><surname>Sinn</surname> <given-names>B. T.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>The Planteome database: an integrated resource for reference ontologies, plant genomics and phenomics.</article-title> <source><italic>Nucleic Acids Res.</italic></source> <volume>46</volume> <fpage>D1168</fpage>&#x2013;<lpage>D1180</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkx1152</pub-id> <pub-id pub-id-type="pmid">29186578</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Danecek</surname> <given-names>P.</given-names></name> <name><surname>Bonfield</surname> <given-names>J. K.</given-names></name> <name><surname>Liddle</surname> <given-names>J.</given-names></name> <name><surname>Marshall</surname> <given-names>J.</given-names></name> <name><surname>Ohan</surname> <given-names>V.</given-names></name> <name><surname>Pollard</surname> <given-names>M. O.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>Twelve years of SAMtools and BCFtools.</article-title> <source><italic>Gigascience</italic></source> <volume>10</volume>:<issue>giab008</issue>. <pub-id pub-id-type="doi">10.1093/gigascience/giab008</pub-id> <pub-id pub-id-type="pmid">33590861</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gelman</surname> <given-names>A.</given-names></name> <name><surname>Carlin</surname> <given-names>J. B.</given-names></name> <name><surname>Stern</surname> <given-names>H. S.</given-names></name> <name><surname>Rubin</surname> <given-names>D. B.</given-names></name></person-group> (<year>2004</year>). <source><italic>Bayesian Data Analysis.</italic></source> <publisher-loc>Boca Raton, FL</publisher-loc>: <publisher-name>CRC Press</publisher-name>.</citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Greenberg</surname> <given-names>A. J.</given-names></name> <name><surname>Hackett</surname> <given-names>S. R.</given-names></name> <name><surname>Harshman</surname> <given-names>L. G.</given-names></name> <name><surname>Clark</surname> <given-names>A. G.</given-names></name></person-group> (<year>2010</year>). <article-title>A hierarchical Bayesian model for a novel sparse partial diallel crossing design.</article-title> <source><italic>Genetics</italic></source> <volume>185</volume> <fpage>361</fpage>&#x2013;<lpage>373</lpage>. <pub-id pub-id-type="doi">10.1534/genetics.110.115055</pub-id> <pub-id pub-id-type="pmid">20157001</pub-id></citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Greenberg</surname> <given-names>A. J.</given-names></name> <name><surname>Hackett</surname> <given-names>S. R.</given-names></name> <name><surname>Harshman</surname> <given-names>L. G.</given-names></name> <name><surname>Clark</surname> <given-names>A. G.</given-names></name></person-group> (<year>2011</year>). <article-title>Environmental and genetic perturbations reveal different networks of metabolic regulation.</article-title> <source><italic>Mol. Syst. Biol.</italic></source> <volume>7</volume>:<issue>563</issue>. <pub-id pub-id-type="doi">10.1038/msb.2011.96</pub-id> <pub-id pub-id-type="pmid">22186737</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grillo</surname> <given-names>M. A.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name> <name><surname>Fowlkes</surname> <given-names>A. M.</given-names></name> <name><surname>Briggeman</surname> <given-names>T. M.</given-names></name> <name><surname>Zhou</surname> <given-names>A.</given-names></name> <name><surname>Schemske</surname> <given-names>D. W.</given-names></name><etal/></person-group> (<year>2009</year>). <article-title>Genetic architecture for the adaptive origin of annual wild rice, <italic>Oryza nivara</italic>.</article-title> <source><italic>Evolution</italic></source> <volume>63</volume> <fpage>870</fpage>&#x2013;<lpage>883</lpage>. <pub-id pub-id-type="doi">10.1111/j.1558-5646.2008.00602.x</pub-id> <pub-id pub-id-type="pmid">19236476</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hill</surname> <given-names>R.</given-names></name></person-group> (<year>2010</year>). <article-title>The cultivation of perennial rice, an early phase in Southeast Asian agriculture?</article-title> <source><italic>J. Hist. Geogr.</italic></source> <volume>36</volume> <fpage>215</fpage>&#x2013;<lpage>223</lpage>. <pub-id pub-id-type="doi">10.1016/j.jhg.2009.09.001</pub-id></citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hoffman</surname> <given-names>G. E.</given-names></name></person-group> (<year>2013</year>). <article-title>Correcting for population structure and kinship using the linear mixed model: theory and extensions.</article-title> <source><italic>PLoS One</italic></source> <volume>8</volume>:<issue>e75707</issue>. <pub-id pub-id-type="doi">10.1371/journal.pone.0075707</pub-id> <pub-id pub-id-type="pmid">24204578</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hotelling</surname> <given-names>H.</given-names></name></person-group> (<year>1931</year>). <article-title>The generalization of student&#x2019;s ratio.</article-title> <source><italic>Ann. Math. Stat.</italic></source> <volume>2</volume> <fpage>360</fpage>&#x2013;<lpage>378,319</lpage>.</citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>X.</given-names></name> <name><surname>Kurata</surname> <given-names>N.</given-names></name> <name><surname>Wei</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>Z.-X.</given-names></name> <name><surname>Wang</surname> <given-names>A.</given-names></name> <name><surname>Zhao</surname> <given-names>Q.</given-names></name><etal/></person-group> (<year>2012</year>). <article-title>A map of rice genome variation reveals the origin of cultivated rice.</article-title> <source><italic>Nature</italic></source> <volume>490</volume> <fpage>497</fpage>&#x2013;<lpage>501</lpage>. <pub-id pub-id-type="doi">10.1038/nature11532</pub-id> <pub-id pub-id-type="pmid">23034647</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jung</surname> <given-names>J. K. H.</given-names></name></person-group> (<year>2016</year>). <source><italic>Scientific Capacity Building: Enhancing our Understanding of Wild Rice Germplasm and Human Resource Development</italic>. Doctoral dissertation</source>. <publisher-loc>Ithaca, NY</publisher-loc>: <publisher-name>Cornell University</publisher-name>.</citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kang</surname> <given-names>H. M.</given-names></name> <name><surname>Sul</surname> <given-names>J. H.</given-names></name> <name><surname>Service</surname> <given-names>S. K.</given-names></name> <name><surname>Zaitlen</surname> <given-names>N. A.</given-names></name> <name><surname>Kong</surname> <given-names>S.-Y.</given-names></name> <name><surname>Freimer</surname> <given-names>N. B.</given-names></name><etal/></person-group> (<year>2010</year>). <article-title>Variance component model to account for sample structure in genome-wide association studies.</article-title> <source><italic>Nat. Genet.</italic></source> <volume>42</volume> <fpage>348</fpage>&#x2013;<lpage>354</lpage>. <pub-id pub-id-type="doi">10.1038/ng.548</pub-id> <pub-id pub-id-type="pmid">20208533</pub-id></citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kennedy</surname> <given-names>A. B. W.</given-names></name> <name><surname>Sankey</surname> <given-names>H. R.</given-names></name></person-group> (<year>1898</year>). <article-title>The thermal efficiency of steam engines. Report of the committee appointed to the council upon the subject of the definition of a standard or standards of thermal efficiency for steam engines: with an introductory note.</article-title> <source><italic>Minutes Proc. Inst. Civil Eng.</italic></source> <volume>134</volume> <fpage>278</fpage>&#x2013;<lpage>312</lpage>. <pub-id pub-id-type="doi">10.1680/imotp.1898.19100</pub-id> <pub-id pub-id-type="pmid">26962031</pub-id></citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>H.</given-names></name></person-group> (<year>2016</year>). <source><italic>Genetic Characterization of the O. rufipogon Species Complex and Construction of Interspecific Pre-Breeding Resources for the Rice Improvement</italic>. Doctoral dissertation</source>. <publisher-loc>Ithaca, NY</publisher-loc>: <publisher-name>Cornell University</publisher-name>.</citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>H.</given-names></name> <name><surname>Jung</surname> <given-names>J.</given-names></name> <name><surname>Singh</surname> <given-names>N.</given-names></name> <name><surname>Greenberg</surname> <given-names>A.</given-names></name> <name><surname>Doyle</surname> <given-names>J. J.</given-names></name> <name><surname>Tyagi</surname> <given-names>W.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>Population dynamics among six major groups of the <italic>Oryza rufipogon</italic> species complex, wild relative of cultivated Asian rice.</article-title> <source><italic>Rice (N. Y.)</italic></source> <volume>9</volume>:<issue>56</issue>. <pub-id pub-id-type="doi">10.1186/s12284-016-0119-0</pub-id> <pub-id pub-id-type="pmid">27730519</pub-id></citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kurata</surname> <given-names>N.</given-names></name> <name><surname>Yamazaki</surname> <given-names>Y.</given-names></name></person-group> (<year>2006</year>). <article-title>Oryzabase. An integrated biological and genome information database for rice.</article-title> <source><italic>Plant Physiol.</italic></source> <volume>140</volume> <fpage>12</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1104/pp.105.063008</pub-id> <pub-id pub-id-type="pmid">16403737</pub-id></citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kuroda</surname> <given-names>Y.</given-names></name> <name><surname>Sato</surname> <given-names>Y.-I.</given-names></name> <name><surname>Bounphanousay</surname> <given-names>C.</given-names></name> <name><surname>Kono</surname> <given-names>Y.</given-names></name> <name><surname>Tanaka</surname> <given-names>K.</given-names></name></person-group> (<year>2006</year>). <article-title>Genetic structure of three <italic>Oryza</italic> AA genome species (<italic>O. rufipogon</italic>, <italic>O. nivara</italic> and <italic>O. sativa</italic>) as assessed by SSR analysis on the Vientiane Plain of Laos.</article-title> <source><italic>Conserv. Genet.</italic></source> <volume>8</volume> <fpage>149</fpage>&#x2013;<lpage>158</lpage>. <pub-id pub-id-type="doi">10.1007/s10592-006-9156-3</pub-id></citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>R.</given-names></name> <name><surname>Zheng</surname> <given-names>X. M.</given-names></name> <name><surname>Zhou</surname> <given-names>L.</given-names></name> <name><surname>Zhou</surname> <given-names>H. F.</given-names></name> <name><surname>Ge</surname> <given-names>S.</given-names></name></person-group> (<year>2015</year>). <article-title>Population genetic structure of <italic>Oryza rufipogon</italic> and <italic>Oryza nivara</italic>: implications for the origin of <italic>O. nivara</italic>.</article-title> <source><italic>Mol. Ecol.</italic></source> <volume>24</volume> <fpage>5211</fpage>&#x2013;<lpage>5228</lpage>. <pub-id pub-id-type="doi">10.1111/mec.13375</pub-id> <pub-id pub-id-type="pmid">26340227</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Londo</surname> <given-names>J. P.</given-names></name> <name><surname>Chiang</surname> <given-names>Y.-C.</given-names></name> <name><surname>Hung</surname> <given-names>K.-H.</given-names></name> <name><surname>Chiang</surname> <given-names>T.-Y.</given-names></name> <name><surname>Schaal</surname> <given-names>B. A.</given-names></name></person-group> (<year>2006</year>). <article-title>Phylogeography of Asian wild rice, <italic>Oryza rufipogon</italic>, reveals multiple independent domestications of cultivated rice, <italic>Oryza sativa</italic>.</article-title> <source><italic>Proc. Natl. Acad. Sci. U.S.A.</italic></source> <volume>103</volume> <fpage>9578</fpage>&#x2013;<lpage>9583</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.0603152103</pub-id> <pub-id pub-id-type="pmid">16766658</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lynch</surname> <given-names>M.</given-names></name> <name><surname>Walsh</surname> <given-names>B.</given-names></name></person-group> (<year>1998</year>). <source><italic>Genetics and Analysis of Quantitative Traits.</italic></source> <publisher-loc>Sunderland, MA</publisher-loc>: <publisher-name>Sinauer Associates, Inc</publisher-name>.</citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Maples</surname> <given-names>B. K.</given-names></name> <name><surname>Gravel</surname> <given-names>S.</given-names></name> <name><surname>Kenny</surname> <given-names>E. E.</given-names></name> <name><surname>Bustamante</surname> <given-names>C. D.</given-names></name></person-group> (<year>2013</year>). <article-title>RFMix: a discriminative modeling approach for rapid and robust local-ancestry inference.</article-title> <source><italic>Am. J. Hum. Genet.</italic></source> <volume>93</volume> <fpage>278</fpage>&#x2013;<lpage>288</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2013.06.020</pub-id> <pub-id pub-id-type="pmid">23910464</pub-id></citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>McGrory</surname> <given-names>C. A.</given-names></name> <name><surname>Titterington</surname> <given-names>D. M.</given-names></name></person-group> (<year>2007</year>). <article-title>Variational approximations in Bayesian model selection for finite mixture distributions.</article-title> <source><italic>Comput. Stat. Data Anal.</italic></source> <volume>51</volume> <fpage>5352</fpage>&#x2013;<lpage>5367</lpage>. <pub-id pub-id-type="doi">10.1016/j.csda.2006.07.020</pub-id></citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Morishima</surname> <given-names>H.</given-names></name></person-group> (<year>2001</year>). &#x201C;<article-title>Evolution and domestication of rice</article-title>,&#x201D; in <source><italic>Proceedings of the 4th International Rice Genetics Symposium Rice Genetics IV</italic></source>, <role>eds</role> <person-group person-group-type="editor"><name><surname>Khush</surname> <given-names>G. S.</given-names></name> <name><surname>Brar</surname> <given-names>D. S.</given-names></name> <name><surname>Hardy</surname> <given-names>B.</given-names></name></person-group> (<publisher-loc>Los Ba&#x00F1;os, PH</publisher-loc>: <publisher-name>International Rice Research Institute</publisher-name>), <fpage>63</fpage>&#x2013;<lpage>77</lpage>. <pub-id pub-id-type="doi">10.1142/9789812814296_0004</pub-id></citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Morishima</surname> <given-names>H.</given-names></name> <name><surname>Oka</surname> <given-names>H.</given-names></name> <name><surname>Chang</surname> <given-names>W.</given-names></name></person-group> (<year>1961</year>). <article-title>Directions of differentiation in populations of wild rice, <italic>Oryza perennis</italic> and <italic>O. sativa F. spontanea</italic>.</article-title> <source><italic>Evolution (N. Y.)</italic></source> <volume>15</volume> <fpage>326</fpage>&#x2013;<lpage>339</lpage>. <pub-id pub-id-type="doi">10.1111/j.1558-5646.1961.tb03158.x</pub-id></citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Morishima</surname> <given-names>H.</given-names></name> <name><surname>Sano</surname> <given-names>Y.</given-names></name> <name><surname>Oka</surname> <given-names>H. I.</given-names></name></person-group> (<year>1984</year>). <article-title>Differentiation of perennial and annual types due to habitat conditions in the wild rice <italic>Oryza perennis</italic>.</article-title> <source><italic>Plant Syst. Evol.</italic></source> <volume>144</volume> <fpage>119</fpage>&#x2013;<lpage>135</lpage>. <pub-id pub-id-type="doi">10.1007/BF00986670</pub-id></citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Motohashi</surname> <given-names>R.</given-names></name> <name><surname>Mochizuki</surname> <given-names>K.</given-names></name> <name><surname>Ohtsubo</surname> <given-names>H.</given-names></name> <name><surname>Ohtsubo</surname> <given-names>E.</given-names></name></person-group> (<year>1997</year>). <article-title>Structures and distribution of p-SINE1 members in rice genomes.</article-title> <source><italic>Theor. Appl. Genet.</italic></source> <volume>95</volume> <fpage>359</fpage>&#x2013;<lpage>368</lpage>. <pub-id pub-id-type="doi">10.1007/s001220050571</pub-id></citation></ref>
<ref id="B41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oka</surname> <given-names>H.</given-names></name></person-group> (<year>1974</year>). <article-title>Analysis of genes controlling F<sub>1</sub> sterility in rice by the use of isogenic lines.</article-title> <source><italic>Genetics</italic></source> <volume>77</volume> <fpage>521</fpage>&#x2013;<lpage>534</lpage>. <pub-id pub-id-type="doi">10.1093/genetics/77.3.521</pub-id> <pub-id pub-id-type="pmid">17248657</pub-id></citation></ref>
<ref id="B42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oka</surname> <given-names>H. I.</given-names></name></person-group> (<year>1988</year>). <source><italic>Origin of Cultivated Rice.</italic></source> <publisher-loc>Tokyo</publisher-loc>: <publisher-name>Japan Scientific Societies</publisher-name>.</citation></ref>
<ref id="B43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oka</surname> <given-names>H. I.</given-names></name> <name><surname>Morishima</surname> <given-names>H.</given-names></name></person-group> (<year>1967</year>). <article-title>Variations in the breeding system of a wild rice, <italic>Oryza perennis</italic>.</article-title> <source><italic>Evolution</italic></source> <volume>21</volume> <fpage>249</fpage>&#x2013;<lpage>258</lpage>. <pub-id pub-id-type="doi">10.1111/j.1558-5646.1967.tb00153.x</pub-id> <pub-id pub-id-type="pmid">28556132</pub-id></citation></ref>
<ref id="B44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Raj</surname> <given-names>A.</given-names></name> <name><surname>Stephens</surname> <given-names>M.</given-names></name> <name><surname>Pritchard</surname> <given-names>J. K.</given-names></name></person-group> (<year>2014</year>). <article-title>fastSTRUCTURE: variational inference of population structure in large SNP data sets.</article-title> <source><italic>Genetics</italic></source> <volume>197</volume> <fpage>573</fpage>&#x2013;<lpage>589</lpage>. <pub-id pub-id-type="doi">10.1534/genetics.114.164350</pub-id> <pub-id pub-id-type="pmid">24700103</pub-id></citation></ref>
<ref id="B45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Robert</surname> <given-names>C. P.</given-names></name></person-group> (<year>1996</year>). &#x201C;<article-title>Mixtures of distributions: inference and estimation</article-title>,&#x201D; in <source><italic>Markov Chain Monte Carlo in Practice</italic></source>, <role>eds</role> <person-group person-group-type="editor"><name><surname>Gilks</surname> <given-names>W. R.</given-names></name> <name><surname>Richardson</surname> <given-names>S.</given-names></name> <name><surname>Spiegelhalter</surname> <given-names>D. J.</given-names></name></person-group> (<publisher-loc>Boca Raton, FL</publisher-loc>: <publisher-name>Chapman and Hall/CRC</publisher-name>), <fpage>441</fpage>&#x2013;<lpage>464</lpage>.</citation></ref>
<ref id="B46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Roschevicz</surname> <given-names>R. J.</given-names></name></person-group> (<year>1931</year>). <article-title>A contribution to the knowledge of rice.</article-title> <source><italic>Appl. Bot. Genet. Pl. Breed. Bull.</italic></source> <volume>27</volume> <fpage>3</fpage>&#x2013;<lpage>133</lpage>.</citation></ref>
<ref id="B47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Samal</surname> <given-names>R.</given-names></name> <name><surname>Roy</surname> <given-names>P. S.</given-names></name> <name><surname>Sahoo</surname> <given-names>A.</given-names></name> <name><surname>Kar</surname> <given-names>M. K.</given-names></name> <name><surname>Patra</surname> <given-names>B. C.</given-names></name> <name><surname>Marndi</surname> <given-names>B. C.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>Morphological and molecular dissection of wild rices from eastern India suggests distinct speciation between <italic>O. rufipogon</italic> and <italic>O. nivara</italic> populations.</article-title> <source><italic>Sci. Rep.</italic></source> <volume>8</volume>:<issue>2773</issue>. <pub-id pub-id-type="doi">10.1038/s41598-018-20693-7</pub-id> <pub-id pub-id-type="pmid">29426872</pub-id></citation></ref>
<ref id="B48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sano</surname> <given-names>Y.</given-names></name> <name><surname>Morishima</surname> <given-names>H.</given-names></name> <name><surname>Oka</surname> <given-names>H.-I.</given-names></name></person-group> (<year>1980</year>). <article-title>Intermediate perennial-annual populations of <italic>Oryza perennis</italic> found in Thailand and their evolutionary significance.</article-title> <source><italic>Bot. Mag.</italic></source> <volume>93</volume> <fpage>291</fpage>&#x2013;<lpage>305</lpage>. <pub-id pub-id-type="doi">10.1007/BF02488735</pub-id></citation></ref>
<ref id="B49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sharma</surname> <given-names>S.</given-names></name></person-group> (<year>2003</year>). &#x201C;<article-title>Species of genus <italic>Oryza</italic> and their interrelationships</article-title>,&#x201D; in <source><italic>Monograph on Genus Oryza</italic></source>, <person-group person-group-type="editor"><name><surname>Nanda</surname> <given-names>J. S.</given-names></name> <name><surname>Sharma</surname> <given-names>S. D.</given-names></name></person-group> (<publisher-loc>Enfield, NH</publisher-loc>: <publisher-name>Science Publishers</publisher-name>).</citation></ref>
<ref id="B50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sharma</surname> <given-names>S.</given-names></name> <name><surname>Shastry</surname> <given-names>S.</given-names></name></person-group> (<year>1965</year>). <article-title>Taxonomic studies in Genus <italic>Oryza</italic> I. Asiatic types of sativa complex.</article-title> <source><italic>Indian J. Genet. Plant Breed.</italic></source> <volume>25</volume> <fpage>245</fpage>&#x2013;<lpage>259</lpage>.</citation></ref>
<ref id="B51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>VanRaden</surname> <given-names>P. M.</given-names></name></person-group> (<year>2008</year>). <article-title>Efficient methods to compute genomic predictions.</article-title> <source><italic>J. Dairy Sci.</italic></source> <volume>91</volume> <fpage>4414</fpage>&#x2013;<lpage>4423</lpage>. <pub-id pub-id-type="doi">10.3168/jds.2007-0980</pub-id> <pub-id pub-id-type="pmid">18946147</pub-id></citation></ref>
<ref id="B52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaughan</surname> <given-names>D. A.</given-names></name></person-group> (<year>1989</year>). <article-title>The genus <italic>Oryza</italic> L.: current status of taxonomy.</article-title> <source><italic>IRRI Res. Pap. Ser.</italic></source> <volume>138</volume> <fpage>1</fpage>&#x2013;<lpage>21</lpage>.</citation></ref>
<ref id="B53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaughan</surname> <given-names>D. A.</given-names></name></person-group> (<year>1994</year>). <source><italic>The Wild Relatives of Rice: A Genetic Resources Handbook.</italic></source> <publisher-loc>Los Banos, PH</publisher-loc>: <publisher-name>International Rice Research Institute</publisher-name>.</citation></ref>
<ref id="B54"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaughan</surname> <given-names>D. A.</given-names></name> <name><surname>Lu</surname> <given-names>B.-R.</given-names></name> <name><surname>Tomooka</surname> <given-names>N.</given-names></name></person-group> (<year>2008</year>). <article-title>Was Asian rice (<italic>Oryza sativa</italic>) domesticated more than once?</article-title> <source><italic>Rice</italic></source> <volume>1</volume> <fpage>16</fpage>&#x2013;<lpage>24</lpage>. <pub-id pub-id-type="doi">10.1007/s12284-008-9000-0</pub-id></citation></ref>
<ref id="B55"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaughan</surname> <given-names>D. A.</given-names></name> <name><surname>Morishima</surname> <given-names>H.</given-names></name> <name><surname>Kadowaki</surname> <given-names>K.</given-names></name></person-group> (<year>2003</year>). <article-title>Diversity in the <italic>Oryza</italic> genus.</article-title> <source><italic>Curr. Opin. Plant Biol.</italic></source> <volume>6</volume> <fpage>139</fpage>&#x2013;<lpage>146</lpage>. <pub-id pub-id-type="doi">10.1016/s1369-5266(03)00009-8</pub-id></citation></ref>
<ref id="B56"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaughan</surname> <given-names>L. K.</given-names></name> <name><surname>Ottis</surname> <given-names>B. V.</given-names></name> <name><surname>Prazak-Havey</surname> <given-names>A. M.</given-names></name> <name><surname>Bormans</surname> <given-names>C. A.</given-names></name> <name><surname>Sneller</surname> <given-names>C.</given-names></name> <name><surname>Chandler</surname> <given-names>J. M.</given-names></name><etal/></person-group> (<year>2001</year>). <article-title>Is all red rice found in commercial rice really <italic>Oryza sativa</italic>?</article-title> <source><italic>Weed Sci.</italic></source> <volume>49</volume> <fpage>468</fpage>&#x2013;<lpage>476</lpage>.</citation></ref>
<ref id="B57"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Vieira</surname> <given-names>F. G.</given-names></name> <name><surname>Crawford</surname> <given-names>J. E.</given-names></name> <name><surname>Chu</surname> <given-names>C.</given-names></name> <name><surname>Nielsen</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <article-title>Asian wild rice is a hybrid swarm with extensive gene flow and feralization from domesticated rice.</article-title> <source><italic>Genome Res.</italic></source> <volume>27</volume> <fpage>1029</fpage>&#x2013;<lpage>1038</lpage>. <pub-id pub-id-type="doi">10.1101/gr.204800.116</pub-id> <pub-id pub-id-type="pmid">28385712</pub-id></citation></ref>
<ref id="B58"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Xu</surname> <given-names>X.</given-names></name> <name><surname>Vieira</surname> <given-names>F. G.</given-names></name> <name><surname>Xiao</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>The power of inbreeding: NGS-based GWAS of rice reveals convergent evolution during rice domestication.</article-title> <source><italic>Mol. Plant</italic></source> <volume>9</volume> <fpage>975</fpage>&#x2013;<lpage>985</lpage>. <pub-id pub-id-type="doi">10.1016/j.molp.2016.04.018</pub-id> <pub-id pub-id-type="pmid">27179918</pub-id></citation></ref>
<ref id="B59"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>J. H.</given-names></name> <name><surname>Cheng</surname> <given-names>C.</given-names></name> <name><surname>Tsuchimoto</surname> <given-names>S.</given-names></name> <name><surname>Ohtsubo</surname> <given-names>H.</given-names></name> <name><surname>Ohtsubo</surname> <given-names>E.</given-names></name></person-group> (<year>2007</year>). <article-title>Phylogenetic analysis of <italic>Oryza rufipogon</italic> strains and their relations to <italic>Oryza sativa</italic> strains by insertion polymorphism of rice SINEs.</article-title> <source><italic>Genes Genet. Syst.</italic></source> <volume>82</volume> <fpage>217</fpage>&#x2013;<lpage>229</lpage>. <pub-id pub-id-type="doi">10.1266/ggs.82.217</pub-id> <pub-id pub-id-type="pmid">17660692</pub-id></citation></ref>
<ref id="B60"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yamazaki</surname> <given-names>Y.</given-names></name> <name><surname>Sakaniwa</surname> <given-names>S.</given-names></name> <name><surname>Tsuchiya</surname> <given-names>R.</given-names></name> <name><surname>Nonomura</surname> <given-names>K.-I.</given-names></name> <name><surname>Kurata</surname> <given-names>N.</given-names></name></person-group> (<year>2010</year>). <article-title>Oryzabase: an integrated information resource for rice science.</article-title> <source><italic>Breed. Sci.</italic></source> <volume>60</volume> <fpage>544</fpage>&#x2013;<lpage>548</lpage>. <pub-id pub-id-type="doi">10.1270/jsbbs.60.544</pub-id> <pub-id pub-id-type="pmid">26081539</pub-id></citation></ref>
<ref id="B61"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>X. M.</given-names></name> <name><surname>Ge</surname> <given-names>S.</given-names></name></person-group> (<year>2010</year>). <article-title>Ecological divergence in the presence of gene flow in two closely related <italic>Oryza</italic> species (<italic>Oryza rufipogon</italic> and <italic>O. nivara</italic>).</article-title> <source><italic>Mol. Ecol.</italic></source> <volume>19</volume> <fpage>2439</fpage>&#x2013;<lpage>2454</lpage>. <pub-id pub-id-type="doi">10.1111/j.1365-294x.2010.04674.x</pub-id> <pub-id pub-id-type="pmid">20653085</pub-id></citation></ref>
</ref-list>
<fn-group>
<fn id="footnote1">
<label>1</label>
<p><ext-link ext-link-type="uri" xlink:href="https://gringlobal.irri.org/gringlobal/search">https://gringlobal.irri.org/gringlobal/search</ext-link></p></fn>
<fn id="footnote2">
<label>2</label>
<p><ext-link ext-link-type="uri" xlink:href="https://shigen.nig.ac.jp/rice/oryzabase/strain/wildCore/list">https://shigen.nig.ac.jp/rice/oryzabase/strain/wildCore/list</ext-link></p></fn>
<fn id="footnote3">
<label>3</label>
<p><ext-link ext-link-type="uri" xlink:href="https://github.com/tonymugen/MuGen">https://github.com/tonymugen/MuGen</ext-link></p></fn>
<fn id="footnote4">
<label>4</label>
<p><ext-link ext-link-type="uri" xlink:href="https://github.com/tonymugen/MuGaMix">https://github.com/tonymugen/MuGaMix</ext-link></p></fn>
<fn id="footnote5">
<label>5</label>
<p><ext-link ext-link-type="uri" xlink:href="https://github.com/davidsjoberg/ggsankey">https://github.com/davidsjoberg/ggsankey</ext-link></p></fn>
</fn-group>
</back>
</article>