<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Insect Sci.</journal-id>
<journal-title>Frontiers in Insect Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Insect Sci.</abbrev-journal-title>
<issn pub-type="epub">2673-8600</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/finsc.2024.1509942</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Insect Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Bayesian Optimization of insect trap distribution for pest monitoring efficiency in agroecosystems</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Yanchenko</surname>
<given-names>Eric</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2866788"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chappell</surname>
<given-names>Thomas M.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/748296"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Huseth</surname>
<given-names>Anders S.</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1113010"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Global Connectivity Program, Akita International University</institution>, <addr-line>Akita</addr-line>, <country>Japan</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Plant Pathology and Microbiology, Texas A&amp;M University, College</institution>, <addr-line>Station, TX</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Entomology and Plant Pathology and North Carolina Plant Science Initiative, North Carolina State University</institution>, <addr-line>Raleigh, NC</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Andrea Sciarretta, University of Molise, Italy</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Rachele Nieri, University of Trento, Italy</p>
<p>Petros T. Damos, Ministry of Education, Research and Religious Affairs, Greece</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Eric Yanchenko, <email xlink:href="mailto:eyanchenko@aiu.ac.jp">eyanchenko@aiu.ac.jp</email>; Anders S. Huseth, <email xlink:href="mailto:ashuseth@ncsu.edu">ashuseth@ncsu.edu</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>22</day>
<month>01</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>4</volume>
<elocation-id>1509942</elocation-id>
<history>
<date date-type="received">
<day>11</day>
<month>10</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>12</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Yanchenko, Chappell and Huseth</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Yanchenko, Chappell and Huseth</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Insect trap networks targeting agricultural pests are commonplace but seldom optimized to improve precision or efficiency. Trap site selection is often driven by user convenience or predetermined trap densities relative to sensitive host crop abundance in the landscape. Monitoring for invasive pests often requires expedient decisions based on dispersal potential and ecology to inform trap placement. Optimization of trap networks using contemporary analytical approaches can help users determine the distribution of traps as information accumulates and priorities change. In this study, a Bayesian optimization (BO) algorithm was used to learn more about the optimal distribution of a fine-scale trap network targeting <italic>Helicoverpa zea</italic> (Boddie), a significant agricultural pest across North America. Four years of pheromone trap monitoring was conducted at the same 21 locations distributed across ~7,000 square kilometers in a five-county area in North Carolina, USA. Three years of data were used to train a BO model with a fourth year designated for testing. For any quantity of trap locations, the approach identified those that provide the most information, allowing optimization of trapping efficiency given either a constraint on the number of locations, or a set precision required for pest density estimation. Results suggest that BO is a powerful approach to enable optimized trap placement decisions by practitioners given finite resources and time.</p>
</abstract>
<kwd-group>
<kwd>
<italic>Helicoverpa zea</italic>
</kwd>
<kwd>sampling efficiency</kwd>
<kwd>adaptive sampling</kwd>
<kwd>eco-efficiency</kwd>
<kwd>integrated pest management</kwd>
</kwd-group>
<contract-num rid="cn001">2023-51181-41157, 2017-70006-27205</contract-num>
<contract-sponsor id="cn001">National Institute of Food and Agriculture<named-content content-type="fundref-id">10.13039/100005825</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">North Carolina Cotton Producers Association<named-content content-type="fundref-id">10.13039/100007323</named-content>
</contract-sponsor>
<counts>
<fig-count count="6"/>
<table-count count="0"/>
<equation-count count="11"/>
<ref-count count="33"/>
<page-count count="11"/>
<word-count count="5793"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Insect Economics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Efficient management of agricultural pest insects requires accurate and timely information about the relative abundance of populations from plant to landscape scales. Understanding where populations occur is also important because many insect pests have uneven distributions within fields and across agroecosystems. For decades, Integrated Pest Management (IPM) practitioners have used pest trap networks as one tool to monitor pest activity and inform growers about infestation risk during the growing season. Pest density estimation using traps informs grower management decisions when used in combination with traditional crop scouting, but collecting this information is costly and rarely optimized for specific cropping systems. Examples of management thresholds using traps exist, but as pest management practices change or monitoring tools evolve (e.g., trap technology, pheromone blend), the relevance of these decision tools decreases. Improving analytical approaches through contemporary modeling will make trap networks more adaptable to enable better management decisions. This improvement addresses a central principle of IPM, which seeks to maximize sustainability of pest management through diversified management practices in the context of agroecosystems (<xref ref-type="bibr" rid="B1">1</xref>).</p>
<p>A central challenge to the continued refinement and adoption of trap network information arises from the translation of pest activity to direct negative impacts on crops. Recent literature has explored trap network optimization for agricultural or medical pest populations in simulated landscapes that include parameters to include varying ecological complexity and stakeholder goals (<xref ref-type="bibr" rid="B2">2</xref>&#x2013;<xref ref-type="bibr" rid="B6">6</xref>). A general theme of this work is that pest populations can be viewed as processes and monitored using simulation techniques developed to detect changes or abnormalities. However, pest populations in agroecosystems are often aggregated in space in addition to being temporally dynamic. Contagion is also a feature of wild insect populations that can challenge assumptions involved in risk assessment. This results in a need to monitor pest populations through time, but it also necessitates attention to the spatial distribution of trap network sites. Techniques useful for generating expectations of pest densities at unsampled geospatial locations exist. For example, Gaussian process regression or Kriging approaches can be useful to combine spatial with temporal aspects of sampling optimization and may increase efficacy of trap networks if users are willing to reconfigure distributions based on stakeholder goals (i.e., detection of large populations). Although these studies can accommodate complex simulations and multiple objectives, they rarely use the abundance of observational data being collected in real agricultural systems to better understand how the efficiency of existing networks can be improved.</p>
<p>Here we develop an application of Bayesian Optimization (BO) (<xref ref-type="bibr" rid="B7">7</xref>) to select from a finite number of trapping locations involved in monitoring an agricultural pest lepidopteran insect, <italic>Helicoverpa zea</italic> Boddie. Bayesian Optimization has become a popular tool for optimizing complicated objective functions and has found success in many domains, particularly environmental monitoring and sensor selection. Practical examples of Bayesian Optimization approaches include selecting the optimal weather sensors from a predefined set to predict the maximum rainfall at unobserved locations (<xref ref-type="bibr" rid="B8">8</xref>), selecting sensors to monitor ozone concentrations (<xref ref-type="bibr" rid="B9">9</xref>), and selecting sensors for temperature monitoring in a given area (<xref ref-type="bibr" rid="B10">10</xref>). In our study, there preexists a certain number of locations from which adult <italic>H. zea</italic> abundance data have been collected using pheromone-baited Texas Hartstack traps (<xref ref-type="bibr" rid="B11">11</xref>). We hypothesize that the information yield of different trap locations varies and the number of traps in a network can be optimized to a smaller subset of high-value locations. In other words, for any predetermined precision requirement, trapping effort can be minimized, and for any predetermined amount of available effort, precision can be maximized. Through this application, we optimize the subset of locations at which to trap to maximize information return (coverage, precision) while minimizing costs. Important additional benefits arise from this approach, for example improvements to the expedient distribution of traps when monitoring for novel or invasive pests, or to the identification of geospatial areas or place-time combinations that are inconsistent with their surroundings in ways that expedite research into the system&#x2019;s biology.</p>
<p>Bayesian Optimization is used to address a combinatorial problem in which the goal is to choose an optimal <inline-formula>
<mml:math display="inline" id="im1">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> locations from a possible set <inline-formula>
<mml:math display="inline" id="im2">
<mml:mi>L</mml:mi>
</mml:math>
</inline-formula>, meaning there are <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>K</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> possible sets of locations that can be chosen. Problems in combinatorics inherently require efficiency to address because of the expansive search spaces they present to brute force approaches: a realistic example of 50 potential locations from which to choose a subset of 20 to receive traps results in more than 10<sup>13</sup> combinations, which is not practical to search, even if the datum at each trap site is simple and affordable to obtain. However, the necessity of efficiency is increased in this use case because the quantity to be optimized is complicated and expensive to calculate. Bayesian Optimization is thus a well-suited approach to this problem, because it can approximate a complicated loss function with a simple statistical model. In our approach, a model is fit with Bayesian regression and then maximized with an acquisition function. We evaluate the original loss function with the optimum from the surrogate model, add it to our data set, and then refit the model. A relatively simple surrogate model is more efficient to optimize in comparison to the complicated original objective function.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Pest ecology and trap network</title>
<p>In the southeastern U.S., <italic>H. zea</italic> completes several generations that cycle through non-crop weeds and then crops that include corn (<italic>Zea mays</italic> L.), upland cotton (<italic>Gossypium hirsutum</italic> L.), soybean (<italic>Glycine max</italic> (L.) Merr.), and several other minor crops (<xref ref-type="bibr" rid="B12">12</xref>&#x2013;<xref ref-type="bibr" rid="B15">15</xref>). Larvae are the economically important life stage because they consume vegetative and reproductive plant structures that result in crop yield loss (e.g., cotton bolls, soybean pods, and corn kernels). As the season progresses, populations grow when high-quality crop hosts become abundant in the landscape, with some of the largest seasonal populations originating from corn (<xref ref-type="bibr" rid="B15">15</xref>). The adult flight originating from corn often coincides with blooming cotton and soybean in early August.</p>
<p>To provide an early warning system for <italic>H. zea</italic> flights, university extension researchers maintain trap networks to monitor <italic>H. zea</italic> adults across the region (<xref ref-type="bibr" rid="B16">16</xref>). Resulting trap data is used as an early warning system to incentivize intensive in-field <italic>H. zea</italic> egg and larval monitoring. Importantly, growers often infer <italic>H. zea</italic> activity for their locations based on a small number of traps distributed across large geographic regions. For example, the North Carolina trap network has included 22 unique black light trap locations in 2024 that span approximately 36,000 km<sup>2</sup> of land area from central to eastern North Carolina. Traps in this network are concentrated in major agricultural regions with the goal of providing growers with timely pest activity information (data available at: <ext-link ext-link-type="uri" xlink:href="https://www.ces.ncsu.edu/trap-data/">https://www.ces.ncsu.edu/trap-data/</ext-link>). These networks provide useful <italic>H. zea</italic> activity information each year but may not account for population variation across spatially and temporally heterogenous agricultural landscapes (<xref ref-type="bibr" rid="B17">17</xref>, <xref ref-type="bibr" rid="B18">18</xref>).</p>
<p>This study documented variation in <italic>H. zea</italic> adult activity in an experimental trap network distributed across 21 locations in a major row crop production landscape from 2020-2023. The trap network spans approximately 700 km<sup>2</sup> of agricultural land in five North Carolina counties (i.e. Northampton, Halifax, Nash, Edgecombe, and Wilson). The fine-scale trap network monitored adult male <italic>H. zea</italic> using commercially available pheromone lures (PHEROCON<sup>&#xae;</sup> CEW, Tr&#xe9;c&#xe9; Inc., Adair, OK) and fabricated metal Texas Hartstack traps (<xref ref-type="bibr" rid="B11">11</xref>). We collaborated with row crop growers to select 21 trap locations that were distributed across the study extent and located in open areas adjacent to agricultural fields. Each year, traps were installed at the same location late in July and monitored weekly for 6-11 weeks during the peak flight period (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>). Captured moths from individual traps were returned to the laboratory and counted each week.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Sampling locations distributed across an intensive row crop production landscape in five North Carolina counties. Numbers correspond to trap ID used in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="finsc-04-1509942-g001.tif"/>
</fig>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Trap data</title>
<p>Our dataset includes <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>705</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> observations, <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>21</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> locations, <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> years (2020-2023). We have 83 site-year combinations (site 3 did not participate in 2020) and 6 - 11 weeks of observations per year. In 2020 there were 6 weeks of trapping, 2021 had 7 weeks, 2022 had 10 weeks, and 2023 had 11 weeks. In 2023, Trap 13 was missing an observation for week 2, so for this observation we imputed the average value of all other sites for this same week. Let <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> be the set of all sampling locations. Our response variable of interest is the cumulative pest count for each week. In particular, if <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the pest count (in hundreds of pests) for location <inline-formula>
<mml:math display="inline" id="im9">
<mml:mi>s</mml:mi>
</mml:math>
</inline-formula> in year <inline-formula>
<mml:math display="inline" id="im10">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula> at week <inline-formula>
<mml:math display="inline" id="im11">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula>, then our response variable, <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, is the <italic>cumulative</italic> pest count for location <inline-formula>
<mml:math display="inline" id="im13">
<mml:mi>s</mml:mi>
</mml:math>
</inline-formula> in year <inline-formula>
<mml:math display="inline" id="im14">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula> at week <inline-formula>
<mml:math display="inline" id="im15">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula>, i.e.,</p>
<disp-formula>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>x</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Note that by definition, <inline-formula>
<mml:math display="inline" id="im16">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Logistic growth model</title>
<p>The logistic growth curve represents biological population growth by including both exponential increase and a density-limiting carrying capacity. It is defined as</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>for all <inline-formula>
<mml:math display="inline" id="im17">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im18">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula> represents the week of the observation. The model is defined by three parameters. <inline-formula>
<mml:math display="inline" id="im19">
<mml:mi>&#x3b2;</mml:mi>
</mml:math>
</inline-formula> is supremum of the function, and here represents the maximum number of cumulative pests. A larger value of <inline-formula>
<mml:math display="inline" id="im20">
<mml:mi>&#x3b2;</mml:mi>
</mml:math>
</inline-formula> means a greater number of pests. The growth parameter, <inline-formula>
<mml:math display="inline" id="im21">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula>, represents how quickly the pest population is increasing, and is inversely correlated with growth rate. If <inline-formula>
<mml:math display="inline" id="im22">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula> is large, then the growth is slower, while a small <inline-formula>
<mml:math display="inline" id="im23">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula> indicates fast growth. Finally, <inline-formula>
<mml:math display="inline" id="im24">
<mml:mi>&#x3b3;</mml:mi>
</mml:math>
</inline-formula> is the midpoint and represents where the function attains half of its supremum, i.e., <inline-formula>
<mml:math display="inline" id="im25">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo stretchy="false">/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula>
<mml:math display="inline" id="im26">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>larger <inline-formula>
<mml:math display="inline" id="im27">
<mml:mi>&#x3b3;</mml:mi>
</mml:math>
</inline-formula> means that the midpoint is later in the growing season, and vice-versa for a smaller <inline-formula>
<mml:math display="inline" id="im28">
<mml:mi>&#x3b3;</mml:mi>
</mml:math>
</inline-formula>.</p>
<p>As formulated in <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>, the logistic growth curve is the same at each trapping location. Because spatial dependencies among trapping locations are expected, <inline-formula>
<mml:math display="inline" id="im29">
<mml:mi>&#x3b2;</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im30">
<mml:mi>&#x3b3;</mml:mi>
</mml:math>
</inline-formula> depend on location. Mathematically, the model becomes</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>~</mml:mo>
<mml:mi>N</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where</p>
<disp-formula>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The asymptote parameter, <inline-formula>
<mml:math display="inline" id="im31">
<mml:mi>&#x3b2;</mml:mi>
</mml:math>
</inline-formula>, has been replaced with a spatially-varying parameter, <inline-formula>
<mml:math display="inline" id="im32">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Similarly, the midpoint <inline-formula>
<mml:math display="inline" id="im33">
<mml:mi>&#x3b3;</mml:mi>
</mml:math>
</inline-formula>, is now modeled spatially with <inline-formula>
<mml:math display="inline" id="im34">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Spatial terms are modeled as a Gaussian Process with exponential correlation, i.e.,</p>
<disp-formula>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>~</mml:mo>
<mml:mi>N</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>&#x3a3;</mml:mi>
<mml:mi>&#x3b2;</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>and</p>
<p>
<inline-formula>
<mml:math display="inline" id="im35">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>~</mml:mo>
<mml:mi>N</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>&#x3a3;</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>,</p>
<p>where <inline-formula>
<mml:math display="inline" id="im36">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3a3;</mml:mi>
<mml:mi>&#x3b2;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im37">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3a3;</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent exponential correlation structures, a special case of Matern correlation. If <inline-formula>
<mml:math display="inline" id="im38">
<mml:mi>&#x3a3;</mml:mi>
</mml:math>
</inline-formula> is an exponential correlation matrix, then</p>
<disp-formula>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3a3;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>&#x3c1;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im39">
<mml:mi>&#x3c1;</mml:mi>
</mml:math>
</inline-formula> is the spatial range parameter, and <inline-formula>
<mml:math display="inline" id="im40">
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the Euclidean distance between trapping locations. In this correlation function, points are more correlated if they are closer together, and the correlation decreases exponentially fast with distance. This model was then fit using all locations and all years (before 2023) via Bayesian inference. Please see the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Materials</bold>
</xref> for full prior specification.</p>
<p>There are several noteworthy points about this model. First, the model has spatial variation, but is the same for each year, i.e., there is no temporal component. There were not enough years of data to accurately learn a parameter differentiating years, and the model thus overfit on year when such a parameter was included. Next, the model has spatial variation in the supremum and midpoint, but not in the growth parameter. Additional models were tested with spatial variation in <inline-formula>
<mml:math display="inline" id="im41">
<mml:mi>k</mml:mi>
</mml:math>
</inline-formula>, but there was very little variation among sites. Thus, the growth parameter was fixed across space. Additionally, both <inline-formula>
<mml:math display="inline" id="im42">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im43">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> have exponential correlation structures, but are allowed to have different spatial range parameters, <inline-formula>
<mml:math display="inline" id="im44">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mi>&#x3b2;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im45">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mi>&#x3b3;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Finally, covariates were not included in the model (e.g., landscape composition), because we found that models including candidate covariates tended to overfit to the data and did not provide good out-of-sample prediction. While we might reasonably assume that covariates such as landscape composition weather, etc. do affect the pest population, the current model implicitly accounts for such factors in the spatial processes. Moreover, the focus of this paper is selecting the optimal trapping locations, not necessarily the reason that these sites were chosen. We consider this latter question an important avenue for future work.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Optimal sampling locations</title>
<p>After empirically establishing the model form, we proceeded to determine how to select the optimal trapping location sets. We seek a small number of trapping locations, <inline-formula>
<mml:math display="inline" id="im46">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula>, with <inline-formula>
<mml:math display="inline" id="im47">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>&lt;</mml:mo>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, such that if we fit a model on these <inline-formula>
<mml:math display="inline" id="im48">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> locations, then the model can most accurately predict at the <inline-formula>
<mml:math display="inline" id="im49">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> unobserved sites. For now, we consider <inline-formula>
<mml:math display="inline" id="im50">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> as known and fixed but we will generalize this in the experimental results.</p>
<p>We seek to minimize the mean log error (MLE) between the real and predicted cumulative counts at the unobserved locations in 2023. Let <inline-formula>
<mml:math display="inline" id="im51">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>i</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>i</mml:mi>
<mml:mi>K</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> be a set of <inline-formula>
<mml:math display="inline" id="im52">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> locations where <inline-formula>
<mml:math display="inline" id="im53">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2286;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and let <inline-formula>
<mml:math display="inline" id="im54">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> be the predicted value at site <inline-formula>
<mml:math display="inline" id="im55">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2209;</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in year <inline-formula>
<mml:math display="inline" id="im56">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>2023</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> for week <inline-formula>
<mml:math display="inline" id="im57">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mn>11</mml:mn>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, using the locations in <inline-formula>
<mml:math display="inline" id="im58">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to fit the logistic growth model. Thus, <inline-formula>
<mml:math display="inline" id="im59">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> are out-of-sample predictions. Then our loss function is the MLE on the sites we did not train on for 2023, defined as</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>11</mml:mn>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2209;</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>11</mml:mn>
</mml:mrow>
</mml:munderover>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where log() is the natural logarithm. Optimization thus seeks</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>S</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In this case, the optimum is the set of sampling locations, <inline-formula>
<mml:math display="inline" id="im60">
<mml:mover accent="true">
<mml:mi>S</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula>, such that, if the model is trained on this set of locations from 2020-2023, and then used to predict trap data from other locations in 2023, MLE will be minimized for predictions of these out-of-sample points. 2023 was chosen as the testing data subset to simulate forecasting of future data using past data. MLE was chosen for our loss function due to the following observation. The observed cumulative counts vary in orders of magnitude, i.e., the first week may have tens of counts while the cumulative sum at the end of the season can be in the thousands. By taking the logarithm of the difference between the observed and fitted values, it ensures that all errors are on the same scale, such that each component of the sum in (3) contributes approximately equally to the loss function. Please see the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Materials</bold>
</xref> for further discussion on the loss-function as well as potential other options.</p>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Bayesian optimization</title>
<p>The loss function in <xref ref-type="disp-formula" rid="eq3">Equation 3</xref> is complex enough to complicate optimization in practice. Indeed, it is not easy to minimize this loss function directly, primarily because it is a combinatorial optimization problem with <inline-formula>
<mml:math display="inline" id="im61">
<mml:mi>L</mml:mi>
</mml:math>
</inline-formula> choose <inline-formula>
<mml:math display="inline" id="im62">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> many solutions. Because fitting the logistic growth curve model takes a nontrivial amount of time, a brute force approach is not feasible. Thus, we employ Bayesian Optimization (BO) to approximate the optimal trapping locations <inline-formula>
<mml:math display="inline" id="im63">
<mml:mover accent="true">
<mml:mi>S</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula>. There are three main steps to a BO schema: the objective function, statistical model, and acquisition function, after Yanchenko (<xref ref-type="bibr" rid="B19">19</xref>). Please see <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> for an overview of the modeling workflow.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Overview of data pipeline and algorithm structure.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="finsc-04-1509942-g002.tif"/>
</fig>
<sec id="s2_5_1">
<label>2.5.1</label>
<title>Objective function</title>
<p>The first step is to evaluate the objective function at some initial points. While it is too costly to evaluate the loss function at every possible sampling set <inline-formula>
<mml:math display="inline" id="im64">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, we can evaluate it at <inline-formula>
<mml:math display="inline" id="im65">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> points in a reasonable amount of time, if <inline-formula>
<mml:math display="inline" id="im66">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is small. Let <inline-formula>
<mml:math display="inline" id="im67">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mi>L</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> be such that <inline-formula>
<mml:math display="inline" id="im68">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> if <inline-formula>
<mml:math display="inline" id="im69">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and 0 otherwise. Since we can only choose <inline-formula>
<mml:math display="inline" id="im70">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> locations, we impose the constraint that <inline-formula>
<mml:math display="inline" id="im71">
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>L</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Then we define <inline-formula>
<mml:math display="inline" id="im72">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> as the MSE for all sites <inline-formula>
<mml:math display="inline" id="im73">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2209;</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and for years 2020-2022, i.e.,</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mover accent="true">
<mml:mi>n</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:mfrac>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>2020</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2022</mml:mn>
</mml:mrow>
</mml:munderover>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2209;</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>x</mml:mi>
</mml:munder>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im74">
<mml:mover accent="true">
<mml:mi>n</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> is the number of terms in the sum. We stress that our goal is the sampling locations <inline-formula>
<mml:math display="inline" id="im75">
<mml:mover accent="true">
<mml:mi>S</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> which minimize the MSE at all locations <inline-formula>
<mml:math display="inline" id="im76">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2209;</mml:mo>
<mml:mover accent="true">
<mml:mi>S</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> for the year 2023. To train the BO algorithm, however, we cannot use the 2023 data, so our choice of optimal sampling points can only depend on the data before 2023. For our initial evaluation, we randomly sample <inline-formula>
<mml:math display="inline" id="im77">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> locations to construct each <inline-formula>
<mml:math display="inline" id="im78">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and define <inline-formula>
<mml:math display="inline" id="im79">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>x</mml:mi>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:msup><mml:mo stretchy="false">)</mml:mo>
<mml:mi>T</mml:mi></mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. For each <inline-formula>
<mml:math display="inline" id="im80">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, we fit the logistic growth model in <xref ref-type="disp-formula" rid="eq2">Equation 2</xref> and evaluate the loss function, <inline-formula>
<mml:math display="inline" id="im81">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> using <xref ref-type="disp-formula" rid="eq5">Equation 5</xref>.</p>
</sec>
<sec id="s2_5_2">
<label>2.5.2</label>
<title>Statistical model</title>
<p>Next, we need a model for the objective function in <xref ref-type="disp-formula" rid="eq5">Equation 5</xref>. Let <inline-formula>
<mml:math display="inline" id="im82">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> be a surrogate model for <inline-formula>
<mml:math display="inline" id="im83">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> parameterized by <bold>
<italic>b</italic>
</bold>. We assume that the surrogate model is linear in its parameters,</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>L</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>for <inline-formula>
<mml:math display="inline" id="im84">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. Since <inline-formula>
<mml:math display="inline" id="im85">
<mml:mi>L</mml:mi>
</mml:math>
</inline-formula> is small, we use a Bayesian multiple linear regression to fit the model in <xref ref-type="disp-formula" rid="eq6">Equation 6</xref> with standard non-informative priors. The coefficient <inline-formula>
<mml:math display="inline" id="im86">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the marginal contribution of site <inline-formula>
<mml:math display="inline" id="im87">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to the overall MLE. So if <inline-formula>
<mml:math display="inline" id="im88">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is large and positive, including <inline-formula>
<mml:math display="inline" id="im89">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in <inline-formula>
<mml:math display="inline" id="im90">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> will likely lead to a large MLE on our hold-out predictions. Conversely, if <inline-formula>
<mml:math display="inline" id="im91">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is negative, then the MLE will be smaller when <inline-formula>
<mml:math display="inline" id="im92">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is used to train the model. The interpretation of <inline-formula>
<mml:math display="inline" id="im93">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>'</mml:mo>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the increase in MLE if we used site <inline-formula>
<mml:math display="inline" id="im94">
<mml:mi>j</mml:mi>
</mml:math>
</inline-formula> to train the model instead of site <inline-formula>
<mml:math display="inline" id="im95">
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>'</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, with all else being equal. In this sense, the coefficient <inline-formula>
<mml:math display="inline" id="im96">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> yields a sense of the relative increase/decrease to the testing MLE if site <inline-formula>
<mml:math display="inline" id="im97">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is used to train the model.</p>
</sec>
<sec id="s2_5_3">
<label>2.5.3</label>
<title>Acquisition function</title>
<p>Lastly, we need to optimize our surrogate model <inline-formula>
<mml:math display="inline" id="im98">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Recall that each coefficient <inline-formula>
<mml:math display="inline" id="im99">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the marginal contribution to the MSE when site <inline-formula>
<mml:math display="inline" id="im100">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is used to train the model. So, if <inline-formula>
<mml:math display="inline" id="im101">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is large in absolute value and negative, we can expect that training on this site will lead to a lower MSE. Therefore, selecting the <inline-formula>
<mml:math display="inline" id="im102">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> sites with lowest <inline-formula>
<mml:math display="inline" id="im103">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> will minimize <inline-formula>
<mml:math display="inline" id="im104">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. For given <inline-formula>
<mml:math display="inline" id="im105">
<mml:mover accent="true">
<mml:mi>b</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:math>
</inline-formula>, the optimal sampling locations are <inline-formula>
<mml:math display="inline" id="im106">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mi>x</mml:mi>
</mml:munder>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mover accent="true">
<mml:mi>b</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow> </mml:math>
</inline-formula>where <inline-formula>
<mml:math display="inline" id="im107">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">&#x2dc;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> if <inline-formula>
<mml:math display="inline" id="im108">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>b</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>b</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> where <inline-formula>
<mml:math display="inline" id="im109">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>b</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the <inline-formula>
<mml:math display="inline" id="im110">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> order statistic of <inline-formula>
<mml:math display="inline" id="im111">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>b</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>b</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>L</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. After solving for <inline-formula>
<mml:math display="inline" id="im112">
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula>, we evaluate <inline-formula>
<mml:math display="inline" id="im113">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and then re-fit (6) by appending <inline-formula>
<mml:math display="inline" id="im114">
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im115">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#x2dc;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula>
<mml:math display="inline" id="im116">
<mml:mi>X</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im117">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, respectively. This step is repeated <inline-formula>
<mml:math display="inline" id="im118">
<mml:mi>B</mml:mi>
</mml:math>
</inline-formula> times, and the optimal seed set is <inline-formula>
<mml:math display="inline" id="im119">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>*</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:munder>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where the minimum is found over the <inline-formula>
<mml:math display="inline" id="im120">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> sampled points. Then <inline-formula>
<mml:math display="inline" id="im121">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>*</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> corresponds to our approximate optimum of (3) and therefore (5) as well.</p>
</sec>
</sec>
<sec id="s2_6">
<label>2.6</label>
<title>Experiment</title>
<p>We now apply the BO method from Section 2.4 to our <italic>H. zea</italic> trap data. For a fixed <inline-formula>
<mml:math display="inline" id="im122">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula>, we follow the method outline in Section 2.4 to obtain our approximate optimum <inline-formula>
<mml:math display="inline" id="im123">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>*</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. We then use the trap locations <inline-formula>
<mml:math display="inline" id="im124">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>*</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> corresponding to <inline-formula>
<mml:math display="inline" id="im125">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>*</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> to fit a logistic growth model based on the data from 2023 and compute the MLE on the hold-out sites for 2023, i.e. <inline-formula>
<mml:math display="inline" id="im126">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>*</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. We do this for <inline-formula>
<mml:math display="inline" id="im127">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>5</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>6</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mn>14</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. For comparison, we also randomly select <inline-formula>
<mml:math display="inline" id="im128">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> trap locations and compute the out-of-sample MSE for 2023, which serves as a baseline method. Both the BO algorithm and random sampling are repeated for 50 Monte Carlo (MC) samples, and the average MLE is reported. We use the standard Kriging estimate to predict cumulative counts at new locations, and all Bayesian models are fit using Stan in R (<xref ref-type="bibr" rid="B20">20</xref>).</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results and discussion</title>
<sec id="s3_1">
<label>3.1</label>
<title>MLE results</title>
<p>The main results show that the proposed BO approach greatly outperforms randomly sampling locations with a lower MLE for all <inline-formula>
<mml:math display="inline" id="im129">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>). Additionally, the MLE monotonically decreases as the number of training sites increases for the BO algorithm. We plot the proportion of MC samples for which a trap location was chosen in the optimal set for <inline-formula>
<mml:math display="inline" id="im130">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>5</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4A, B</bold>
</xref>), we also plot the proportion of samples a trap was chosen in the optimal set averaged over all <inline-formula>
<mml:math display="inline" id="im133">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>). The central traps are chosen for the optimal set more often when <inline-formula>
<mml:math display="inline" id="im134">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, whereas the spread of optimal sites for <inline-formula>
<mml:math display="inline" id="im135">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> is biased toward the south. When the results are averaged over all <inline-formula>
<mml:math display="inline" id="im136">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula>, the southern and central locations have a higher proportion of MC samples for which they are in the optimal set.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Mean log error for hold-out 2023 locations for Bayesian Optimization (black line) and random sampling locations (gray line) against number of training sites. Averaged over 50 Monte Carlo samples.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="finsc-04-1509942-g003.tif"/>
</fig>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Proportion of Monte Carlo samples a trap was chosen in the optimal set for <inline-formula>
<mml:math display="inline" id="im131">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> <bold>(A)</bold> and <inline-formula>
<mml:math display="inline" id="im132">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> <bold>(B)</bold>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="finsc-04-1509942-g004.tif"/>
</fig>
<p>While <xref ref-type="fig" rid="f4">
<bold>Figures&#xa0;4</bold>
</xref>, <xref ref-type="fig" rid="f5">
<bold>5</bold>
</xref> show the proportion of MC samples for which a site was chosen in the optimal set, <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref> considers the binary inclusion/exclusion of sites in the optimal set. For each value of <inline-formula>
<mml:math display="inline" id="im140">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula>, the proportion of samples for which each site was chosen in the optimal set was calculated. Then the <inline-formula>
<mml:math display="inline" id="im141">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> sites with the largest proportion are chosen as the optimal sites. This is represented in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref> as a black tile if the trap has one of the <inline-formula>
<mml:math display="inline" id="im142">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> largest proportions, and a white tile otherwise. In case of ties, both sites have a black tile. Traps 1, 5 and 14 are included in the optimal site for every value of <inline-formula>
<mml:math display="inline" id="im143">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula>. Conversely, traps 13, 15, 16, 18, and 21 are never selected. The plot also shows a substantial amount of nesting in the optimal sites, i.e., trap <italic>i</italic> is selected in the optimal set for <inline-formula>
<mml:math display="inline" id="im144">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula>, then it is also selected in the optimal set for <inline-formula>
<mml:math display="inline" id="im145">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Proportion of times a site was chosen in the optimal set averaged over all <inline-formula>
<mml:math display="inline" id="im137">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="finsc-04-1509942-g005.tif"/>
</fig>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Inclusion/exclusion of traps in optimal set for each value of <inline-formula>
<mml:math display="inline" id="im138">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula>. A black square indicates that trap was selected in the optimal set for the particular value of <inline-formula>
<mml:math display="inline" id="im139">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula>. In case of tie, both sites are included.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="finsc-04-1509942-g006.tif"/>
</fig>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Model extension</title>
<p>As an extension of this work, we could explicitly include money, time, distance, etc. into our loss function if we wanted these to factor into our optimal choice (e.g., <xref ref-type="bibr" rid="B9">9</xref>). For example,</p>
<disp-formula>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>11</mml:mn>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2209;</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>11</mml:mn>
</mml:mrow>
</mml:munderover>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im146">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the cost to sample at locations <inline-formula>
<mml:math display="inline" id="im147">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im148">
<mml:mi>&#x3bb;</mml:mi>
</mml:math>
</inline-formula> is a tuning parameter which controls how much we weigh fit (small <inline-formula>
<mml:math display="inline" id="im149">
<mml:mi>&#x3bb;</mml:mi>
</mml:math>
</inline-formula>) versus cost (large <inline-formula>
<mml:math display="inline" id="im150">
<mml:mi>&#x3bb;</mml:mi>
</mml:math>
</inline-formula>) in the loss function.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Model performance and interpretation</title>
<p>The general implication of these empirical findings is that optimization enriches the information provided per unit trap in a multi-site network. First, the magnitude of the MLE associated with an optimized set is appreciably smaller than that for random sampling. For example, a MLE of 1.2 means that the predicted count was within 230 units of the true cumulative count value, on average. In the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Materials</bold>
</xref>, we also compare the results of the BO algorithm with a brute-force computation for <inline-formula>
<mml:math display="inline" id="im151">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and find that the BO results are close to the global optimum with similar trap combinations detected by each method (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1</bold>
</xref>). This establishes the utility of the method for efficiently selecting trapping locations.</p>
<p>The BO algorithm was run 50 times and the average MLE was recorded. Ideally, the algorithm would be run once and the global optimum would be found, such that running multiple MC iterations would be unnecessary. The necessity of running the BO algorithm multiple times could be because the algorithm is finding a local optimum and/or because the objective function surface is relatively &#x201c;flat.&#x201d; This would mean that several different sets of <inline-formula>
<mml:math display="inline" id="im152">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> trap locations yield comparable MLEs. While these considerations and the computational expense of running the BO algorithm multiple times are relevant, running the algorithm multiple times also has advantages. The proportion of MC samples that a trap is included in the optimal set yields a continuous measure of importance as opposed to a binary result. This allows for some uncertainty quantification in the optimal sites and a richer understanding of which sites to choose for maximum precision in pest density estimation, and perhaps which sites to choose for research to understand why they are inconsistent or otherwise different from the optimal set.</p>
<p>In general, the optimal trapping locations appear in the southern and central locations. This could be interpreted one of two ways. It could mean that knowing the number of <italic>H. zea</italic> at these locations is very informative for knowing the number of moths at other locations. On the other hand, it could mean that these locations are very difficult to predict so we should trap at these sites to ensure an accurate value. For example, trap 1 could be included in every optimal set because knowing its value helps predict at the other sites, or because predicting the number of <italic>H. zea</italic> at trap 1 is extremely difficult. In the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Materials</bold>
</xref>, we looked for systematic reasons why some trapping locations were chosen more than others. We found a moderate relationship showing that traps with a high probability of being included in the optimal set tended to have larger amounts of corn and soy planted within a 1 km radius. These are initial findings, however, and require further study.</p>
<p>Together, these steps toward optimization are important, as the costs of monitoring <italic>H. zea</italic> are immediately clear. Establishing which traps in the network provide the most return on investment would be a significant step toward sustainable monitoring approaches that continue to provide useful information to agricultural stakeholders. In the future, this approach would also apply to invasive species monitoring, which is a common approach when outbreaks occur. The persistent threat of pests with similar ecology as <italic>H. zea</italic> (i.e., <italic>Helicoverpa armigera</italic>) is a major focus of regulatory agencies tasked with responding to incipient threats of invasive species in the United States (e.g., USDA-APHIS). Ecological niche modeling suggests that <italic>H. armigera</italic> could establish in the southern U.S. and would pose a significant threat to agriculture across broad geographies (<xref ref-type="bibr" rid="B21">21</xref>). Effective monitoring strategies would be a first step toward effective management of a widespread invasive outbreak. The BO approach is one tool which could be adapted to model effective monitoring networks that are informed by the current distribution and activity of a closely related species.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Extending BO modeling to understand pest ecological drivers and sampling efficiency</title>
<p>These results suggest that understanding these contrasting motivating factors will help to assess which locations are informative and/or what about these locations make them informative. Future studies focused on assessing location-wise characteristics from the perspective of <italic>H. zea</italic> could generate new context for the optimized dataset. For example, overwintering temperatures limit <italic>H. zea</italic> survival during the pupal stage (<xref ref-type="bibr" rid="B22">22</xref>) which contributes to regional trends in population abundance over time (<xref ref-type="bibr" rid="B16">16</xref>). Moreover, soil conditions which vary at much smaller spatial scales are also linked to overwintering survival which affects <italic>H. zea</italic> population size in the spring (<xref ref-type="bibr" rid="B23">23</xref>, <xref ref-type="bibr" rid="B24">24</xref>). During the growing season, the abundance of highly suitable host plants for <italic>H. zea</italic> development across multiple generations may also explain variation among traps across the relatively small geographic region where we established our trap network (<xref ref-type="bibr" rid="B12">12</xref>&#x2013;<xref ref-type="bibr" rid="B14">14</xref>, <xref ref-type="bibr" rid="B18">18</xref>). Although these studies independently assessed specific aspects of <italic>H. zea</italic> ecology, they are unable to account for multiple variables influencing these insects in agricultural systems. Absent an optimized network, it is difficult to learn what factors at key trap sites are responsible for those sites&#x2019; providing information that can positively affect grower management decisions. Next steps connecting optimized networks to relevant ecological and environmental drivers would further refine our understanding about <italic>H. zea</italic> population dynamics in the context of realistic agroecosystems. In short, the improvement of observational efficiency results in the improvement of any study or application that relies on precision estimates of the observed process. Optimizing the provision of evidence used to describe biological systems, to make system management decisions, or to discover novel or incipient events in those system is a valuable step to take early in a study or management program.</p>
<p>This study builds on a deep body of literature focused on improving sampling optimization for insect pests in agriculture. Entomologists apply numerous sampling techniques for the purpose of estimating arthropod population densities, sometimes adapting from methods used in engineering (<xref ref-type="bibr" rid="B25">25</xref>) or analysis of data from clinical trials (<xref ref-type="bibr" rid="B26">26</xref>, <xref ref-type="bibr" rid="B27">27</xref>). A commonly applied sampling technique in entomological pest management is sequential sampling, first developed in the 1940&#x2019;s (<xref ref-type="bibr" rid="B28">28</xref>), for the purpose of increasing the efficiency of estimation for values important to industrial processes. The primary motivation for the development of sequential sampling techniques was efficiency: the potential to accelerate estimation or increase precision per sample, while minimizing wasted sampling effort. In pest management, this translates to iterative sampling and interim analysis based on observations. Samplers use pre-established stopping conditions to determine when sufficient pest density or crop damage information has been collected to make a management decision. This level of tolerated pest activity is informed by the economic thresholds, which is a pest density or other damage metric above which <italic>not</italic> managing the pest is expected to result in net economic loss. For example, the threshold at which the value of the expected yield loss to <italic>H. zea</italic> larval herbivory exceeds the cost of management. Effectively, this effort becomes a statistical hypothesis, and sequential sampling proceeds until the hypothesis is rejected or fails to be rejected under pre-established tolerances for Type I and Type II error rates. While some important developments have extended the rationale of increasing sampling efficiency for decision support (e.g. <xref ref-type="bibr" rid="B25">25</xref>, <xref ref-type="bibr" rid="B29">29</xref>&#x2013;<xref ref-type="bibr" rid="B31">31</xref>, <xref ref-type="bibr" rid="B32">32</xref>), efficiency gains through innovative statistical approaches continue to represent a major opportunity for improving the environmental compatibility and economy of agricultural pest management. The BO algorithm developed in this study enhances sampling in a way parallel to existing practices of sequential sampling or interim analysis, but adds a substantial amount of value to data in the process by showing the aforementioned nesting of high-value sites within larger optimal sets, by providing a measure of uncertainty around the importance of individual sites, and by including correlation structure that allows nonrandom associations between sites to be exploited for optimization. Similar approaches could be developed for other pest species and geographic regions but will require initial investment to collect activity data for model development and validation.</p>
</sec>
</sec>
<sec id="s4" sec-type="conclusions">
<label>4</label>
<title>Conclusions and future directions</title>
<p>In this study, BO provided useful insight into the optimization of a trap network targeting a key agricultural pest. Our results showed that we can achieve <italic>H. zea</italic> monitoring objectives without sampling at every trapping location. Moreover, the BO approach greatly outperforms the baseline method of random sampling. These empirical results provide several key takeaways for practitioners and stakeholders. First, if budgets limit the number of traps in a study area, then our BO procedure yields the optimal locations for these traps, while initial monitoring still may be required to validate the experimental set-up. Second, our methodology can also suggest how many traps to use, as in prospective power analysis. For example, if there is some pre-determined error tolerance or pre-specified desired coverage level, then the BO procedure can be used, not only choose the locations of the traps, but also how many to deploy. Indeed, the results in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref> show a clear monotonically decreasing error as the number of traps increases, so this must be chosen carefully. Being able to afford just one more trap in the best location may lead to high payoff. Collecting data from all 21 traps in the existing network currently requires a ~700 km trip for two entomologists for 10 weeks. Because time and distance are considerable, we also discussed an extension to the BO analysis that includes flexibility to optimize networks based on user-defined criteria. Finally, once the number and location of the traps are chosen, further analysis can be performed to understand <italic>why</italic> these sites were selected. This question is slightly beyond the scope of this work, so we only took a first step toward this end looking at landscape information which aligns with previous work suggesting that the abundance of soybean in the landscape is an important predictor of <italic>H. zea</italic> abundance in this region (<xref ref-type="bibr" rid="B33">33</xref>). The proposed statistics framework, however, facilitates principled study of this question.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The datasets presented in this article are not readily available because the information includes specific location information. The raw data supporting the conclusions of this article will be made available by the authors upon reasonable request. Requests to access the datasets should be directed to <email xlink:href="mailto:ashuseth@ncsu.edu">ashuseth@ncsu.edu</email>.</p>
</sec>
<sec id="s6" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The manuscript presents research on animals that do not require ethical approval for their study.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>EY: Formal analysis, Methodology, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing, Conceptualization, Investigation. TC: Methodology, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing, Formal analysis. AH: Conceptualization, Methodology, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing, Data curation, Funding acquisition, Investigation, Project administration, Resources, Supervision, Visualization.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. Parts of this work is supported by the North Carolina Cotton Producers Association, Specialty Crop Research Initiative program grant 2023-51181-41157, Crop Protection and Pest Management competitive grant 2017-70006-27205, and Biotechnology Risk Assessment Grant 2020-33522-32272, from the U.S. Department of Agriculture&#x2019;s National Institute of Food and Agriculture.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We thank Seth Dorman, Lindsey Christianson, Renee Ackerman, Bailey Allison, and Gabbie Frech for servicing the trap network. We thank cooperating growers for providing access to trap locations over the duration of this project.</p>
</ack>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec id="s10" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12" sec-type="disclaimer">
<title>Author disclaimer</title>
<p>Any opinions, findings, conclusions, or recommendations expressed in this publication are those of the author(s) and should not be construed to represent any official USDA or U.S. Government determination or policy.</p>
</sec>
<sec id="s13" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/finsc.2024.1509942/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/finsc.2024.1509942/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.pdf" id="SM1" mimetype="application/pdf"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Getz</surname> <given-names>WM</given-names>
</name>
<name>
<surname>Gutierrez</surname> <given-names>AP</given-names>
</name>
</person-group>. <article-title>A perspective on systems analysis in crop production and insect pest management</article-title>. <source>Ann Rev Entomol</source>. (<year>1982</year>) <volume>27</volume>:<page-range>447&#x2013;66</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1146/annurev.en.27.010182.002311</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Manoukis</surname> <given-names>NC</given-names>
</name>
<name>
<surname>Hall</surname> <given-names>B</given-names>
</name>
<name>
<surname>Geib</surname> <given-names>SM</given-names>
</name>
</person-group>. <article-title>A computer model of insect traps in a landscape</article-title>. <source>Sci Rep</source>. (<year>2014</year>) <volume>4</volume>:<fpage>7015</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/srep07015</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guimapi</surname> <given-names>RA</given-names>
</name>
<name>
<surname>Mohamed</surname> <given-names>SA</given-names>
</name>
<name>
<surname>Ekesi</surname> <given-names>S</given-names>
</name>
<name>
<surname>Biber-Freudenberger</surname> <given-names>L</given-names>
</name>
<name>
<surname>Borgemeister</surname> <given-names>C</given-names>
</name>
<name>
<surname>Tonnang</surname> <given-names>HE</given-names>
</name>
</person-group>. <article-title>Optimizing spatial positioning of traps in the context of integrated pest management</article-title>. <source>Ecol Complex</source>. (<year>2020</year>) <volume>41</volume>:<fpage>100808</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecocom.2019.100808</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Caton</surname> <given-names>BP</given-names>
</name>
<name>
<surname>Manoukis</surname> <given-names>NC</given-names>
</name>
<name>
<surname>Pallipparambil</surname> <given-names>GR</given-names>
</name>
</person-group>. <article-title>Simulation-based evaluation of two insect trapping grids for delimitation surveys</article-title>. <source>Sci Rep</source>. (<year>2022</year>) <volume>12</volume>:<fpage>11089</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-022-14958-5</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Caton</surname> <given-names>BP</given-names>
</name>
<name>
<surname>Fang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Pallipparambil</surname> <given-names>GR</given-names>
</name>
<name>
<surname>Manoukis</surname> <given-names>NC</given-names>
</name>
</person-group>. <article-title>Transect-based trapping for area-wide delimitation of insects</article-title>. <source>J Econ Entomol</source>. (<year>2023</year>) <volume>116</volume>:<page-range>1002&#x2013;16</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/jee/toad059</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Duan</surname> <given-names>S</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>B</given-names>
</name>
<name>
<surname>Adam</surname> <given-names>B</given-names>
</name>
<name>
<surname>He</surname> <given-names>Z</given-names>
</name>
</person-group>. <article-title>Intelligent pest trap monitoring under uncertainty in food industry</article-title>. <source>Swarm Evol Comput</source>. (<year>2024</year>) <volume>86</volume>:<fpage>101465</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.swevo.2023.101465</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Frazier</surname> <given-names>PI</given-names>
</name>
</person-group>. <article-title>A tutorial on Bayesian optimization</article-title>. <source>arXiv preprint arXiv:1807.02811</source>. (<year>2018</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1807.02811</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Garnett</surname> <given-names>R</given-names>
</name>
<name>
<surname>Osborne</surname> <given-names>MA</given-names>
</name>
<name>
<surname>Roberts</surname> <given-names>SJ</given-names>
</name>
</person-group>. (<year>2010</year>). <article-title>Bayesian optimization for sensor set selection</article-title>, in: <conf-name>Proceedings of the 9th ACM/IEEE international conference on information processing in sensor networks</conf-name>, (<publisher-loc>Stockholm, Sweden</publisher-loc>). pp. <page-range>209&#x2013;19</page-range>.</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Marchant</surname> <given-names>R</given-names>
</name>
<name>
<surname>Ramos</surname> <given-names>F</given-names>
</name>
</person-group>. (<year>2012</year>). <article-title>Bayesian optimisation for intelligent environmental monitoring</article-title>, in: <conf-name>2012 IEEE/RSJ international conference on intelligent robots and systems</conf-name>, . pp. <page-range>2242&#x2013;9</page-range>, <publisher-loc>Vilamoura-Algarve, Portugal</publisher-loc>: <publisher-name>IEEE</publisher-name>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/IROS.2012.6385653</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Srinivas</surname> <given-names>N</given-names>
</name>
<name>
<surname>Krause</surname> <given-names>A</given-names>
</name>
<name>
<surname>Kakade</surname> <given-names>SM</given-names>
</name>
<name>
<surname>Seeger</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Gaussian process optimization in the bandit setting: No regret and experimental design</article-title>. <source>arXiv preprint arXiv:0912.3995</source>. (<year>2009</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.0912.3995</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hartstack</surname> <given-names>AW</given-names>
</name>
<name>
<surname>Witz</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Buck</surname> <given-names>DR</given-names>
</name>
</person-group>. <article-title>Moth traps for the tobacco budworm</article-title>. <source>J Econ Entomol</source>. (<year>1979</year>) <volume>72</volume>:<page-range>519&#x2013;22</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/jee/72.4.519</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Neunzig</surname> <given-names>HH</given-names>
</name>
</person-group>. <article-title>Wild host plants of the corn earworm and the tobacco budworm in eastern North Carolina</article-title>. <source>J Econ Entomol</source>. (<year>1963</year>) <volume>56</volume>:<page-range>135&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/jee/56.2.135</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Neunzig</surname> <given-names>HH</given-names>
</name>
</person-group>. (<year>1969</year>). <article-title>Biology of the tobacco budworm and the corn earworm in North Carolina</article-title>, in: <conf-name>North Carolina Agricultural Experiment Station. Tech. Bull. No. 196</conf-name>, <conf-loc>Raleigh, NC</conf-loc>.</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Johnson</surname> <given-names>MW</given-names>
</name>
<name>
<surname>Stinner</surname> <given-names>RE</given-names>
</name>
<name>
<surname>Rabb</surname> <given-names>RL</given-names>
</name>
</person-group>. <article-title>Ovipositional response of <italic>Heliothis zea</italic> (Boddie) to its major hosts in North Carolina</article-title>. <source>Environ Entomol</source>. (<year>1975</year>) <volume>4</volume>:<page-range>291&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/ee/4.2.291</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jackson</surname> <given-names>RE</given-names>
</name>
<name>
<surname>Bradley</surname> <given-names>JR</given-names>
</name>
<name>
<surname>Van Duyn</surname> <given-names>JOHN</given-names>
</name>
<name>
<surname>Leonard</surname> <given-names>BR</given-names>
</name>
<name>
<surname>Allen</surname> <given-names>KC</given-names>
</name>
<name>
<surname>Luttrell</surname> <given-names>R</given-names>
</name>
<etal/>
</person-group>. <article-title>Regional assessment of <italic>Helicoverpa zea</italic> populations on cotton and non-cotton crop hosts</article-title>. <source>Entomol Exp Appl</source>. (<year>2008</year>) <volume>126</volume>:<fpage>89</fpage>&#x2013;<lpage>106</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/j.1570-7458.2007.00653.x</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lawton</surname> <given-names>D</given-names>
</name>
<name>
<surname>Huseth</surname> <given-names>AS</given-names>
</name>
<name>
<surname>Kennedy</surname> <given-names>GG</given-names>
</name>
<name>
<surname>Morey</surname> <given-names>AC</given-names>
</name>
<name>
<surname>Hutchison</surname> <given-names>WD</given-names>
</name>
<name>
<surname>Reisig</surname> <given-names>DD</given-names>
</name>
<etal/>
</person-group>. <article-title>Pest population dynamics are related to a continental overwintering gradient</article-title>. <source>P. Natl Acad Sci U S A</source>. (<year>2022</year>) <volume>119</volume>:<elocation-id>e2203230119</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.2203230119</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Huseth</surname> <given-names>AS</given-names>
</name>
<name>
<surname>Reisig</surname> <given-names>DD</given-names>
</name>
<name>
<surname>Hutchison</surname> <given-names>WD</given-names>
</name>
</person-group>. <article-title>Linking corn earworm populations and management to landscapes across North America</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Brewer</surname> <given-names>MJ</given-names>
</name>
<name>
<surname>Hein</surname> <given-names>GL</given-names>
</name>
</person-group>, editors. <source>Arthropod Management and Landscape Considerations in Large-scale Agroecosystems</source>. <publisher-name>CABI</publisher-name>, <publisher-loc>Wallingford, Great Britain</publisher-loc> (<year>2024</year>). p. <fpage>187</fpage>&#x2013;<lpage>208</lpage>.</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kennedy</surname> <given-names>GG</given-names>
</name>
<name>
<surname>Storer</surname> <given-names>NP</given-names>
</name>
</person-group>. <article-title>Life systems of polyphagous arthropod pests in temporally unstable cropping systems</article-title>. <source>Ann Rev Entomol</source>. (<year>2000</year>) <volume>45</volume>:<page-range>467&#x2013;93</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1146/annurev.ento.45.1.467</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yanchenko</surname> <given-names>E</given-names>
</name>
</person-group>. <article-title>BOPIM: Bayesian Optimization for influence maximization on temporal networks</article-title>. <source>arXiv preprint arXiv:2308.047</source>. (<year>2023</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2308.04700</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="web">
<person-group person-group-type="author">
<collab>Stan Development Team</collab>
</person-group>. <source>RStan: the R interface to Stan. R package version 2.32.6</source> (<year>2024</year>). Available online at: <uri xlink:href="https://mc-stan.org/">https://mc-stan.org/</uri> (Accessed <access-date>March 1, 2024</access-date>).</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haile</surname> <given-names>F</given-names>
</name>
<name>
<surname>Nowatzki</surname> <given-names>T</given-names>
</name>
<name>
<surname>Storer</surname> <given-names>N</given-names>
</name>
</person-group>. <article-title>Overview of pest status, potential risk, and management considerations of <italic>Helicoverpa armigera</italic> (Lepidoptera: Noctuidae) for US soybean production</article-title>. <source>J Integrat. Pest Manage</source>. (<year>2021</year>) <volume>12</volume>:<fpage>1</fpage>&#x2013;<lpage>10</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/jipm/pmaa030</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Morey</surname> <given-names>AC</given-names>
</name>
<name>
<surname>Hutchison</surname> <given-names>WD</given-names>
</name>
<name>
<surname>Venette</surname> <given-names>RC</given-names>
</name>
<name>
<surname>Burkness</surname> <given-names>EC</given-names>
</name>
</person-group>. <article-title>Cold hardiness of <italic>Helicoverpa zea</italic> (Lepidoptera: Noctuidae) pupae</article-title>. <source>Environ Entomol</source>. (<year>2012</year>) <volume>41</volume>:<page-range>172&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1603/EN11026</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dillard</surname> <given-names>D</given-names>
</name>
<name>
<surname>Reisig</surname> <given-names>DD</given-names>
</name>
<name>
<surname>Reay-Jones</surname> <given-names>FP</given-names>
</name>
</person-group>. <article-title>
<italic>Helicoverpa zea</italic> (Lepidoptera: Noctuidae) in-season and overwintering pupation response to soil type</article-title>. <source>Environ Entomol</source>. (<year>2023</year>) <volume>52</volume>:<fpage>67</fpage>&#x2013;<lpage>73</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/ee/nvac106</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dillard</surname> <given-names>D</given-names>
</name>
<name>
<surname>Reisig</surname> <given-names>DD</given-names>
</name>
<name>
<surname>Schug</surname> <given-names>HT</given-names>
</name>
<name>
<surname>Burrack</surname> <given-names>HJ</given-names>
</name>
</person-group>. <article-title>Moisture and soil type are primary drivers of <italic>Helicoverpa zea</italic> (Lepidoptera: Noctuidae) pupation</article-title>. <source>Environ Entomol</source>. (<year>2023</year>) <volume>52</volume>:<page-range>847&#x2013;52</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/ee/nvad074</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Binns</surname> <given-names>MR</given-names>
</name>
</person-group>. <article-title>Sequential sampling for classifying pest status</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Pedigo</surname> <given-names>LP</given-names>
</name>
<name>
<surname>Buntin</surname> <given-names>GD</given-names>
</name>
</person-group>, editors. <source>Handbook of sampling methods for arthropods in agriculture</source>. <publisher-name>CRC Press</publisher-name>, <publisher-loc>Boca Raton, FL</publisher-loc> (<year>1994</year>). p. <page-range>137&#x2013;74</page-range>.</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>O&#x2019;Brien</surname> <given-names>PC</given-names>
</name>
<name>
<surname>Fleming</surname> <given-names>TR</given-names>
</name>
</person-group>. <article-title>A multiple testing procedure for clinical trials</article-title>. <source>Biometrics</source>. (<year>1979</year>) <volume>35</volume>:<page-range>549&#x2013;56</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2307/2530245</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Demets</surname> <given-names>DL</given-names>
</name>
<name>
<surname>Lan</surname> <given-names>KG</given-names>
</name>
</person-group>. <article-title>Interim analysis: the alpha spending function approach</article-title>. <source>Stat Med</source>. (<year>1994</year>) <volume>13</volume>:<page-range>1341&#x2013;52</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/sim.4780131308</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wallis</surname> <given-names>WA</given-names>
</name>
</person-group>. <article-title>The statistical research group 1942&#x2013;1945</article-title>. <source>J Am Stat Assoc</source>. (<year>1980</year>) <volume>75</volume>:<page-range>320&#x2013;30</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/01621459.1980.10477469</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shelton</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Theunissen</surname> <given-names>J</given-names>
</name>
<name>
<surname>Hoy</surname> <given-names>CW</given-names>
</name>
</person-group>. <article-title>Efficiency of variable-intensity and sequential sampling for insect control decisions in cole crops in the Netherlands</article-title>. <source>Entomol Exp Appl</source>. (<year>1994</year>) <volume>70</volume>:<page-range>209&#x2013;15</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/j.1570-7458.1994.tb00749.x</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Taylor</surname> <given-names>LR</given-names>
</name>
</person-group>. <article-title>A natural law for the spatial disposition of insects</article-title>. <source>Proc 12th Int Congr. Entomol</source>. (<year>1965</year>) <volume>12</volume>:<page-range>396&#x2013;7</page-range>.</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wilson</surname> <given-names>LT</given-names>
</name>
<name>
<surname>Room</surname> <given-names>PM</given-names>
</name>
</person-group>. <article-title>Clumping patterns of fruit and arthropods in cotton, with implications for binomial sampling</article-title>. <source>Environ Entomol</source>. (<year>1983</year>) <volume>12</volume>:<page-range>50&#x2013;4</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/ee/12.1.50</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kogan</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Integrated pest management: historical perspectives and contemporary developments</article-title>. <source>Ann Rev Entomol</source>. (<year>1998</year>) <volume>43</volume>:<page-range>243&#x2013;70</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1146/annurev.ento.43.1.243</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dorman</surname> <given-names>SJ</given-names>
</name>
<name>
<surname>Hopperstad</surname> <given-names>KA</given-names>
</name>
<name>
<surname>Reich</surname> <given-names>BJ</given-names>
</name>
<name>
<surname>Kennedy</surname> <given-names>G</given-names>
</name>
<name>
<surname>Huseth</surname> <given-names>AS</given-names>
</name>
</person-group>. <article-title>Soybeans as a non-Bt refuge for <italic>Helicoverpa zea</italic> in maize-cotton agroecosystems</article-title>. <source>Agr. Ecosyt. Environ</source>. (<year>2021</year>) <volume>322</volume>:<fpage>107642</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.agee.2021.107642</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>