<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Nutr.</journal-id>
<journal-title>Frontiers in Nutrition</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Nutr.</abbrev-journal-title>
<issn pub-type="epub">2296-861X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnut.2025.1610363</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Nutrition</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>ByteTrack: a deep learning approach for bite count and bite rate detection using meal videos in children</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Bhat</surname>
<given-names>Yashaswini Rajendra</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3110183/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Keller</surname>
<given-names>Kathleen L.</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/977071/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Brick</surname>
<given-names>Timothy R.</given-names>
</name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/10467/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Pearce</surname>
<given-names>Alaina L.</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/736174/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Nutritional Sciences, Pennsylvania State University</institution>, <addr-line>University Park, PA</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Food Science, Pennsylvania State University</institution>, <addr-line>University Park, PA</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Human Development and Family Studies, Pennsylvania State University</institution>, <addr-line>University Park, PA</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>Institute of Computational and Data Sciences, Pennsylvania State University</institution>, <addr-line>University Park, PA</addr-line>, <country>United States</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0002">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1142799/overview">Colby Vorland</ext-link>, Indiana University, United States</p>
</fn>
<fn fn-type="edited-by" id="fn0003">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/159363/overview">Megan A. McCrory</ext-link>, Boston University, United States</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3039406/overview">Jiangpeng He</ext-link>, Massachusetts Institute of Technology, United States</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Alaina L. Pearce, <email>azp271@psu.edu</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>10</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1610363</elocation-id>
<history>
<date date-type="received">
<day>11</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Bhat, Keller, Brick and Pearce.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Bhat, Keller, Brick and Pearce</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Assessing eating behaviors such as eating rate can shed light on risk for overconsumption and obesity. Current approaches either use sensors that disrupt natural eating or rely on labor-intensive video coding, which limits scalability.</p>
</sec>
<sec>
<title>Methods</title>
<p>We developed ByteTrack, a deep learning system for automated bite count and bite-rate detection from video-recorded child meals. The dataset comprised 1,440 minutes from 242 videos of 94 children (ages 7&#x2013;9 years) consuming four meals, spaced one week apart, with identical foods served in varying amounts. ByteTrack operates in two stages: (1) face detection via a hybrid Faster R-CNN and YOLOv7 pipeline, and (2) bite classification using an EfficientNet convolutional neural network combined with a long short-term memory (LSTM) recurrent network. The model was designed to handle blur, low light, camera shake, and occlusions (hands or utensils blocking the mouth). Performance was compared with manual observational coding.</p>
</sec>
<sec>
<title>Results</title>
<p>On a test set of 51 videos, ByteTrack achieved an average precision of 79.4%, recall of 67.9%, and F1 score of 70.6%. Agreement with the gold-standard coding, assessed by intraclass correlation coefficient, averaged 0.66 (range 0.16&#x2013;0.99), with lower reliability in videos with extensive movement or occlusions.</p>
</sec>
<sec>
<title>Discussion</title>
<p>This pilot study demonstrates the feasibility of a scalable, automated tool for bite detection in children&#x2019;s meals. While results were promising, performance decreased when faces were partially blocked or motion was high. Future work will focus on improving robustness across diverse populations and recording conditions.</p>
</sec>
<sec>
<title>Clinical trial registration</title>
<p><uri xlink:href="https://clinicaltrials.gov/study/NCT03341247">https://clinicaltrials.gov/study/NCT03341247</uri>, identifier NCT03341247.</p>
</sec>
</abstract>
<kwd-group>
<kwd>bite detection</kwd>
<kwd>neural networks</kwd>
<kwd>eating behaviors</kwd>
<kwd>childhood obesity</kwd>
<kwd>dietary assessment</kwd>
<kwd>automation</kwd>
</kwd-group>
<counts>
<fig-count count="7"/>
<table-count count="2"/>
<equation-count count="6"/>
<ref-count count="81"/>
<page-count count="17"/>
<word-count count="11717"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Nutrition Methodology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec2">
<label>1</label>
<title>Introduction</title>
<p>Behaviors exhibited during a bout of eating (e.g., bites, chews, eating rate, bite-size, etc.) are collectively known as &#x201C;meal microstructure.&#x201D; Meal microstructure can be assessed to understand individual differences in eating patterns (<xref ref-type="bibr" rid="ref1">1</xref>), the effects of food properties (<xref ref-type="bibr" rid="ref1 ref2 ref3">1&#x2013;3</xref>), and mechanisms of disordered eating and obesity (<xref ref-type="bibr" rid="ref4 ref5 ref6 ref7 ref8">4&#x2013;8</xref>). In pediatric populations, these insights are especially valuable for understanding obesity risk, as behaviors like larger bites and faster eating have been linked to greater food consumption and obesity (<xref ref-type="bibr" rid="ref5">5</xref>, <xref ref-type="bibr" rid="ref8">8</xref>). Characterizing meal microstructure can provide insights into pediatric obesity risk, which could potentially lead to novel interventions to reduce this global pandemic (<xref ref-type="bibr" rid="ref9">9</xref>). Observational studies have informed interventions targeting eating speed in children, with promising results. For example, an intervention (<xref ref-type="bibr" rid="ref10">10</xref>) aimed at slowing child eating rates through educational materials and timers resulted in slower parent-reported eating rates and lower BMI gain over 8 weeks compared to the control group, suggesting interventions on meal microstructure hold promise for weight gain prevention in youth (<xref ref-type="bibr" rid="ref11">11</xref>). Although these results are promising, research on meal microstructure may be held back by the expense and difficulty of reliably coding eating episodes. Developing methods like ByteTrack could help streamline measurement and improve the scalability and sustainability of this approach.</p>
<p>Several approaches have been developed to measure meal microstructure in humans. Currently, the gold standard for bite and microstructure analysis is manual observational coding (<xref ref-type="bibr" rid="ref12">12</xref>), where researchers manually review videos and annotate bite timestamps through observation. Although observational coding is highly accurate and reliable (<xref ref-type="bibr" rid="ref13">13</xref>), it is time-consuming, labor-intensive, and costly, making it less scalable and efficient compared to automatic bite detection systems. To address those limitations, wearable devices that use various sensor modalities such as acoustic sensors and accelerometers have been designed to record meal microstructure in adults (<xref ref-type="bibr" rid="ref14">14</xref>). However, wearable sensor-based bite detection relies on predefined motion thresholds, which can lead to false positives (misidentifying hand movements like drinking or gesturing as bites), struggle with utensil variability (difficulty adapting to different eating methods such as chopsticks, spoons, or eating by hand), and face challenges in different contextual settings (<xref ref-type="bibr" rid="ref15">15</xref>, <xref ref-type="bibr" rid="ref16">16</xref>).</p>
<p>To address challenges with wearable devices, several groups have developed automated approaches for bite counting from video, which are more adaptable. One method uses facial landmarks to define bites based on criteria like hand proximity or mouth opening (<xref ref-type="bibr" rid="ref17 ref18 ref19">17&#x2013;19</xref>). While effective in controlled environments, this approach is prone to false positives from non-eating behaviors such as gestures, talking, or facial expressions. Optical flow approaches (<xref ref-type="bibr" rid="ref20">20</xref>, <xref ref-type="bibr" rid="ref21">21</xref>), which track motion between consecutive frames, also face significant limitations in reliably distinguishing between eating actions (i.e., bites) and other dynamic movements such as fidgeting, gesturing, or speaking. These challenges are especially pronounced in children (<xref ref-type="bibr" rid="ref22">22</xref>, <xref ref-type="bibr" rid="ref23">23</xref>), who often engage in frequent hand-to-face movements or fidgeting, but similar issues can also occur in adults during social interactions.</p>
<p>Deep learning-based approaches (e.g., Convolutional Neural Networks or CNNs) have demonstrated stronger performance in bite detection (<xref ref-type="bibr" rid="ref18">18</xref>, <xref ref-type="bibr" rid="ref20">20</xref>, <xref ref-type="bibr" rid="ref24">24</xref>) compared to facial landmark and optical flow-based models. However, these methods have primarily been tested under ideal conditions with high-quality video recordings of eating events, often involving adult participants. Real-world applications present a broader range of challenges, including varied lighting and higher variability in movement patterns, which are common across age groups. Development of deep learning approaches to automate bite detection would advance the field by making models more resistant to non-eating movements.</p>
<p>The purpose of this paper is to present ByteTrack, a deep learning model designed to detect bites and calculate eating speed in pediatric populations. To ensure its robustness in addressing challenges specific to pediatric samples, ByteTrack was directly trained on video recordings of children eating meals. It integrates advanced machine learning techniques, including Convolutional Neural Networks (CNNs) and Long Short-Term Memory-Recurrent Neural Networks (LSTM-RNNs). The main objectives of this paper are: (1) to develop a deep-learning-based bite detection system to automatically identify bites in video data recorded from children&#x2019;s laboratory meals; (2) to evaluate ByteTrack&#x2019;s accuracy and reliability by testing it on a designated video dataset (test set); and (3) to compare ByteTrack&#x2019;s performance with manual (gold standard) annotations to assess its practical utility for capturing key measures of children&#x2019;s eating behavior such as bite count, bite rate, and meal duration, along with correlations between measured intake with predicted bite counts.</p>
</sec>
<sec sec-type="methods" id="sec3">
<label>2</label>
<title>Methods</title>
<sec id="sec4">
<label>2.1</label>
<title>Data collection</title>
<sec id="sec5">
<label>2.1.1</label>
<title>Study design and participants</title>
<p>ByteTrack was trained on 242 videos (4,770&#x202F;min) of laboratory meals in 94 children aged 7&#x2013;9 years. Children consumed 4 laboratory meals with approximately 1-week between each meal. Video data came from the Food and Brain study (<ext-link xlink:href="https://ClinicalTrials.gov" ext-link-type="uri">ClinicalTrials.gov</ext-link>, NCT03341247), a prospective investigation examining neural and cognitive risk factors for the development of obesity in middle childhood. The study employed a longitudinal family risk design, with children attending six baseline visits and one follow-up visit conducted 1 year after the first baseline visit. Data related to neurocognitive and longitudinal outcomes are reported elsewhere (<xref ref-type="bibr" rid="ref25 ref26 ref27 ref28 ref29 ref30 ref31">25&#x2013;31</xref>). Methods used to collect anthropometric and demographic data can be found in earlier studies (<xref ref-type="bibr" rid="ref26">26</xref>, <xref ref-type="bibr" rid="ref27">27</xref>, <xref ref-type="bibr" rid="ref32">32</xref>). Families were recruited for the study based on maternal weight status: low (maternal BMI&#x202F;&#x003C;&#x202F;25) versus high (maternal BMI&#x202F;&#x2265;&#x202F;30) risk for obesity. All children were below the 90th BMI-for-age percentile at baseline. The study received approval from the Pennsylvania State University Institutional Review Board. Parental consent and child assent were obtained at the first visit, and families received modest monetary compensation for completing each study visit. The final sample included 94 subjects who had a total of 345 viable meal videos (<xref ref-type="fig" rid="fig1">Figure 1</xref> for reference to viability of data). Participant inclusion and exclusion with relevant CONSORT diagram are detailed in earlier publications (<xref ref-type="bibr" rid="ref25">25</xref>). Demographic data for the sample can be found in <xref ref-type="table" rid="tab1">Table 1</xref>. <xref rid="SM1" ref-type="supplementary-material">Supplementary Table 1</xref> provides detailed information on subject-by-subject meal videos considered for training and testing, including subject IDs and meal sessions, along with details on exclusions and inclusions.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Flowchart of video dataset consideration for ByteTrack model development.</p>
</caption>
<graphic xlink:href="fnut-12-1610363-g001.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Flowchart displaying the evaluation and exclusion process for videos. Out of 368 portion size meals, 10 were excluded due to corrupted data, leaving 358 for manual annotation. Thirteen more were excluded for ByteTrack development issues, resulting in 345 viable videos. Exclusions include cropped layouts, camera angles, visibility issues, and missing labeling.</alt-text>
</graphic>
</fig>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Demographics of the sample.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Characteristic</th>
<th align="center" valign="top" colspan="3">Total Included (<italic>n</italic>&#x202F;=&#x202F;94 children)</th>
</tr>
<tr>
<th align="left" valign="top">Categorical variables</th>
<th/>
<th/>
<th>n (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="13"/>
<td align="center" valign="top" colspan="3">Sex</td>
</tr>
<tr>
<td/>
<td align="center" valign="top">Male</td>
<td align="center" valign="top">49 (52.1)</td>
</tr>
<tr>
<td/>
<td align="center" valign="top">Female</td>
<td align="center" valign="top">45 (47.9)</td>
</tr>
<tr>
<td align="center" valign="top" colspan="3">Race</td>
</tr>
<tr>
<td/>
<td align="center" valign="top">White</td>
<td align="center" valign="top">91 (96.8)</td>
</tr>
<tr>
<td/>
<td align="center" valign="top">Non-white<sup>#</sup></td>
<td align="center" valign="top">3 (3.2)</td>
</tr>
<tr>
<td align="center" valign="top" colspan="3">Parental education<sup>$</sup></td>
</tr>
<tr>
<td/>
<td align="center" valign="top">&#x003C;Bachelor&#x2019;s degree (&#x003C;16&#x202F;years)</td>
<td align="center" valign="top">18(19.1)</td>
</tr>
<tr>
<td/>
<td align="center" valign="top">Bachelor&#x2019;s degree (16&#x202F;years)</td>
<td align="center" valign="top">42 (44.7)</td>
</tr>
<tr>
<td/>
<td align="center" valign="top">&#x003E;Bachelor&#x2019;s degree (&#x003E;16&#x202F;years)</td>
<td align="center" valign="top">34 (36.2)</td>
</tr>
<tr>
<td align="center" valign="top" colspan="3">Parental income<sup>&#x2020;</sup></td>
</tr>
<tr>
<td/>
<td align="center" valign="top">&#x003C;$51,000</td>
<td align="center" valign="top">12 (12.8)</td>
</tr>
<tr>
<td/>
<td align="center" valign="top">$51,000&#x2013;$100,000</td>
<td align="center" valign="middle">45 (47.9)</td>
</tr>
<tr>
<td/>
<td/>
<td align="center" valign="top">&#x003E;$100,000</td>
<td align="center" valign="top">34 (36.2)</td>
</tr>
</tbody>
</table>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Continuous variables</th>
<th align="center" valign="top"></th>
<th align="center" valign="top">Mean (SD), [min, max]</th>
<th/>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="2"/>
<td align="center" valign="top">Age (years) at baseline</td>
<td align="center" valign="top">7.9 (0.6), [7.0, 9.0]</td>
<td/>
</tr>
<tr>
<td align="center" valign="top">BMI percentile at baseline</td>
<td align="center" valign="top">47.9 (24.3), [3.9, 89.5]</td>
<td/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><sup>#</sup>Asian: <italic>n</italic> =&#x202F;3; American Indian/Alaskan Native, Black/African American, Hawaiian/Pacific Islander: all <italic>n</italic> =&#x202F;0; All participants reported not being Hispanic/Latino. <sup>&#x2020;</sup>Parent income missing values (<italic>n</italic>&#x202F;=&#x202F;3). <sup>$</sup>Parental Education: &#x003C; Bachelor&#x2019;s Degree (&#x003C;16&#x202F;years): High School/GED, Associate&#x2019;s Degree, Technical or vocational school; &#x003E; Bachelor&#x2019;s Degree (&#x003E;16&#x202F;years): Master&#x2019;s Degree, Ph.D.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="sec6">
<label>2.1.2</label>
<title>Laboratory meals</title>
<p>During visits 2&#x2013;5, children were served meals of identical foods that varied by amount served (i.e., portion size), according to the aims of the parent study. The protocols for these meal sessions along with foods served have been previously described (<xref ref-type="bibr" rid="ref25">25</xref>). In brief, meals consisted of foods that are common to United States children (i.e., macaroni and cheese, chicken nuggets, grapes, and broccoli). The reference portion (smallest) consisted of the usual serving sizes for this age group, and the subsequent three portions increased the amount served for each item by ~33%. The order in which children received the meals was randomly assigned and counter- balanced. Meal sessions were separated by at least a week.</p>
<p>Children had up to 30&#x202F;min to eat <italic>ad libitum</italic> until comfortably full, while being read a non-food related story. Initially, these stories were read by a research assistant (<italic>n</italic>&#x202F;=&#x202F;62). However, due to COVID-19 safety protocol changes, the remaining 32 children were either read a book by a parent (<italic>n</italic>&#x202F;=&#x202F;16) or listened to a computerized audiobook (<italic>n</italic>&#x202F;=&#x202F;16; Audible by Amazon, Newark, NJ). Intake was calculated by subtracting post-weight from pre-weight for each food. Intake was converted to kilocalories using Nutrition Facts panel Information or an online database.<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref></p>
</sec>
<sec id="sec7">
<label>2.1.3</label>
<title>Video recording</title>
<p>Each eating event (meal session) was video recorded at 30 frames per second using an Axis M3004-V network camera. The camera was positioned outside the line of sight of children during the meal session. While parents were informed about the recordings, children were not. If a child noticed the camera or asked about its purpose, the research assistant explained that the camera was for safety purposes, aiming to reduce any observer effect. <xref ref-type="fig" rid="fig2">Figure 2</xref> provides a schematic of the eating environment for the two dining rooms used over the visits, along with examples of a meal session.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Observation room layouts for meal videos. Children&#x2019;s meal intake was recorded in one of two observation rooms (A or B). Cameras were wall-mounted to record children&#x2019;s meal intake.</p>
</caption>
<graphic xlink:href="fnut-12-1610363-g002.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Layout and images depict an observational study setup. Top left shows a diagram of the room with labeled areas, including a plate, bowls of grapes and broccoli, ketchup, cutlery, water, and participants' positions. Top right displays a researcher and child seated at a table with similar food items. Bottom left mimics top left with a slightly different angle. Bottom right mirrors top right with another pair of a researcher and child. The focus is on the experimental setup for observing interactions.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="sec8">
<label>2.2</label>
<title>Model building</title>
<p>The ByteTrack pipeline for detecting bites consists of two major parts. The first part focuses on detecting and tracking faces in the video, to ensure that the system concentrates on the target child and ignored irrelevant objects or other individuals. The purpose of this is to reduce noise and prepare clean data for the second part of the pipeline. In the second part, the identified faces from the first step are analyzed to classify their movements and determine whether a child was taking a bite or performing other actions, such as talking or any irrelevant gestures. To enhance accuracy, a filtering process is applied to refine the results. Together, these steps form a 2-model pipeline to identify bites in videos. Diagrammatic representation of overall bite detection is in <xref ref-type="fig" rid="fig3">Figure 3</xref>. A more detailed flowchart for the system development is in <xref rid="SM1" ref-type="supplementary-material">Supplementary Figure 1</xref>. All model development, deployment, and statistical analyses were conducted using Python version 3.11.7 (<xref ref-type="bibr" rid="ref33">33</xref>). Inter-rater reliability (ICC) was computed using Pingouin (v0.5.4) (<xref ref-type="bibr" rid="ref34">34</xref>), while other statistical analyses, including regression and correlation, were performed using Statsmodels (v0.14.0) (<xref ref-type="bibr" rid="ref35">35</xref>). F1 score calculations and classification metrics were computed using scikit-learn (v1.2.2) (<xref ref-type="bibr" rid="ref36">36</xref>). Model development and computer vision tasks used PyTorch (v2.2.2) (<xref ref-type="bibr" rid="ref37">37</xref>) and OpenCV (v4.5.3) (<xref ref-type="bibr" rid="ref38">38</xref>). All plots and visualizations were generated using Matplotlib (v3.8.4) (<xref ref-type="bibr" rid="ref39">39</xref>).</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Overview of the bite detection pipeline. Meal videos were first converted to a constant frame rate of 30 frames per second (fps) before face detection (Model 1) using YOLOv7, with Faster R-CNN as a fallback. Detected faces were processed through EfficientNet CNN for feature extraction, followed by bite classification using an LSTM-RNN (Model 2). Post-processing techniques, including optical flow validation (Lucas-Kanade method), duplicate detection suppression, and temporal smoothing, were applied to refine predictions. The final output included the peak frame of each detected bite, with timestamps analyzed based on the 30fps conversion.</p>
</caption>
<graphic xlink:href="fnut-12-1610363-g003.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Flowchart illustrating a process to analyze meal videos for bite detection. It starts with meal videos converted to thirty frames per second. Model 1 handles face detection, optionally using YOLOv7 or Faster RCNN as fallbacks. EfficientNet CNN extracts features, followed by Model 2 for bite classification (LSTM-RNN). Optical Flow, duplicate detection suppression, and temporal smoothing refine results. The peak frame of each bite is detected, and timestamps are analyzed using thirty frames per second conversion.</alt-text>
</graphic>
</fig>
<sec id="sec9">
<label>2.2.1</label>
<title>Model 1: face detection</title>
<p>The first part of the model pipeline focuses on detecting and tracking faces in video footage, an essential step for identifying who is present and ensuring the system only analyzes relevant areas. To achieve this, we gathered a dataset of frames extracted from a random subset of videos to train and test the system. Two different approaches are used to detect faces: one that prioritizes speed for quick processing (You Only Look Once or YOLOv7) and another that is designed to handle more challenging situations (Faster Regional Convolutional Neural Network or Faster R-CNN), such as when faces are partially blocked or hard to see. The system is designed with the goal of achieving both efficiency and accuracy in face detection before progressing to Model 2 in the pipeline. Refer back to <xref ref-type="fig" rid="fig3">Figure 3</xref> for an overview of ByteTrack model development.</p>
<sec id="sec10">
<label>2.2.1.1</label>
<title>Step 1: model 1 dataset preparation</title>
<p>To ensure consistent frame rates across all videos for ground truth matching, meal videos (<italic>n</italic>&#x202F;=&#x202F;345) are converted from a variable frame rate to a constant frame rate with .mp4 format and 30 frames per second (fps). This conversion is done using FFmpeg software (<xref ref-type="bibr" rid="ref40">40</xref>) (version 4.3.2) with CUDA acceleration (h264_nvenc codec). Videos are then converted into frames at 6 frames per second to balance temporal resolution and computational efficiency.</p>
<p>A random sample of 52 videos (50 frames per video, 2,600 frames total) that include a diverse range of skin tones and seating positions is randomly selected for the face detection dataset. A 70&#x2013;15-15% subject split is done (<xref ref-type="bibr" rid="ref41">41</xref>, <xref ref-type="bibr" rid="ref42">42</xref>), yielding training (<italic>n</italic>&#x202F;=&#x202F;1,800 frames, 36 subjects), validation (<italic>n</italic>&#x202F;=&#x202F;400 frames, 8 subjects), and test (<italic>n</italic>&#x202F;=&#x202F;400 frames, 8 subjects) sets. These sampled image frames are considered the Model 1 Dataset. To ensure an independent validation sample, no subjects appeared in more than one data set.</p>
<p>To enhance face detection model robustness to real-world variations in conditions (e.g., lighting changes, motion, camera angles, etc.,) the following augmentations are applied to the training dataset (<xref ref-type="bibr" rid="ref43">43</xref>): Y-reflection (mirror), 2-D Gaussian smoothing (blurring), brightness adjustment, orientation change to portrait mode, and rotation (clockwise 10 degrees and anticlockwise 10 degrees). As a result of these adjustments, the total training set increased to include 10,800 images (i.e., frames of videos).</p>
</sec>
<sec id="sec11">
<label>2.2.1.2</label>
<title>Step 2: ground truth for model 1&#x2014;manual face manual labeling</title>
<p>To create a labeled dataset for training, within each image of the Model 1 Dataset, a single researcher (YRB) used bounding boxes to identify the children&#x2019;s faces using the ImageLabeler API [LabelImg (<xref ref-type="bibr" rid="ref44">44</xref>)]. Examples of the labeled images with bounding boxes are shown in <xref ref-type="fig" rid="fig4">Figure 4</xref>. The annotated label files are also augmented or transformed along with corresponding images in the training set. To maintain the aspect ratio (original size: 1,920 &#x00D7; 1,080) and reduce computation time, all original and transformed images are resized to 510 &#x00D7; 300 with padding (<xref ref-type="bibr" rid="ref45">45</xref>) (when needed in augmentations). The Model 1 Dataset is utilized for both Faster R-CNN and YOLOv7 face detection models.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Examples of bounding boxes using LabelIMg used for labeling child faces for face detection.</p>
</caption>
<graphic xlink:href="fnut-12-1610363-g004.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Four images depict a child and a researcher or parent in a small room. The child sits at a table, sometimes with food and a tray. In each image, different activities are noted, such as using an audiobook or reading materials. The room features plain walls and a rug.</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec12">
<label>2.2.1.3</label>
<title>Step 3A: automatic face detection with YOLO V7</title>
<p>YOLOv7 (<xref ref-type="bibr" rid="ref46">46</xref>, <xref ref-type="bibr" rid="ref47">47</xref>) is used to develop a lightweight face detector to reduce computational costs. YOLOv7 is a one-stage object detection model designed for faster inference speeds compared to two-stage models like Faster R-CNN. In YOLOv7, the entire image is processed in a single pass through the network, allowing for faster and real-time object detection with lower computational overhead (<xref ref-type="bibr" rid="ref47">47</xref>, <xref ref-type="bibr" rid="ref48">48</xref>). This makes YOLOv7 particularly suited for tasks that require fast detection in video-based settings.</p>
<p>YOLOv7 provides a general-purpose deep learning network for object identification, but the parameters must be fine-tuned to specifically identify children&#x2019;s faces. To achieve this, the model was trained with Stochastic Gradient Descent (SGD) (<xref ref-type="bibr" rid="ref49">49</xref>), a commonly used optimizer that updates model parameters incrementally on a subset of the training data, using an initial learning rate of 0.001 and exponential decay. The learning rate determines the size of the steps taken by the optimization algorithm (SGD) to adjust the model&#x2019;s parameters during training, where a smaller learning rate makes smaller, more precise steps and a larger learning rate makes bigger, faster adjustments but risks overshooting the optimal solution. To balance early, faster progress with fine-tuning later in training, the learning rate was reduced gradually using an exponential decay (i.e., decreased by a fixed proportion over time), allowing the model to make smaller and more refined updates as training progressed. A batch size of four was used, meaning the model processed four samples at a time before updating its internal parameters. The training was conducted over 100 epochs, or complete passes through the entire dataset. To prevent overfitting, early stopping was applied to halt training if the validation set loss (a measure of model performance on unseen data during training) does not improve after 10 consecutive epochs (i.e., patience&#x202F;=&#x202F;10). No additional hyperparameter tuning or cross-validation was performed beyond the standard training configuration provided by YOLOv7. Training was run on a Dell XPS 15 laptop with a 4GB GPU, 8 CPU cores, and 16GB of RAM. Training time was 6&#x202F;h for YOLOv7 model.</p>
</sec>
<sec id="sec13">
<label>2.2.1.4</label>
<title>Step 3B. Automatic face detection with faster regional CNN</title>
<p>We used transfer learning on a Faster R-CNN model (<xref ref-type="bibr" rid="ref50">50</xref>) with a ResNet-50 backbone (<xref ref-type="bibr" rid="ref51">51</xref>) and Feature Pyramid Network (FPN) (<xref ref-type="bibr" rid="ref52">52</xref>) for child face detection. Faster R-CNN is a two-stage object detection model that first generates regional proposals and then classifies and refines these regions, making it well-suited for tasks requiring accurate localization, such as face detection. The ResNet-50 backbone is a deep convolutional network with 50 layers, and its integration with FPN enhances the model&#x2019;s ability to detect objects at multiple scales, which improves accuracy in detecting faces of varying sizes and positions.</p>
<p>To assess the model&#x2019;s generalizability, we initially conducted 3-fold cross-validation (<xref ref-type="bibr" rid="ref53">53</xref>), where the training set was split into three subsets or folds. Each fold used 7,200 training images and 3,600 validation images. Following cross-validation, we conducted a grid search (<xref ref-type="bibr" rid="ref54">54</xref>) to fine-tune the model&#x2019;s hyperparameters using the full training set, guided by feedback from a separate validation set. The model was trained using a batch size of 8 images with a loss accumulation over eight batches, which increased the batch size to 64 images. We used an initial learning rate of 0.01 with an SGD optimizer with a weight decay of 0.0005, a step size of 2 where the scheduler updated the learning rate every two epochs instead of default. We also used a momentum of 0.9 for the SGD optimizer to smooth parameter updates by incorporating a fraction of the previous update into the current one, where a high momentum (momentum&#x202F;=&#x202F;1) gains all information from the previous step. Early stopping with a patience of three epochs was used based on feedback from the validation set for minimizing validation loss. Training was carried out on a high-performance cluster with 800GB RAM and 32 CPU cores. The training time with selected hyperparameters was 18&#x202F;h and 22&#x202F;min.</p>
</sec>
<sec id="sec14">
<label>2.2.1.5</label>
<title>Step 4: face detection using a combination of YOLOv7 and faster RCNN</title>
<p>We implemented a face detection and tracking system combining YOLOv7 and Faster R-CNN [similar to method used in (<xref ref-type="bibr" rid="ref55">55</xref>)]. Videos were processed at 30 fps, with YOLOv7 handling initial detection. If YOLOv7&#x2019;s confidence score exceeded 0.8, its detected bounding box was used to track the child&#x2019;s face using a Kernelized Correlation Filter (KCF) tracker (<xref ref-type="bibr" rid="ref56">56</xref>), a high-speed tracker which updated every 20 frames to maintain accurate localization and prevent drift.</p>
<p>YOLOv7 served as the primary detector, provided it successfully detected a face with a confidence score of 0.8 or higher. If YOLOv7 failed to detect a face or produced a confidence score below this threshold, Faster R-CNN was used as a fallback. By default, the detections were weighted, with YOLOv7 assigned 80% and Faster R-CNN 20%. However, if YOLOv7&#x2019;s bounding box was significantly smaller&#x2014;less than 30% of the area of the Faster R-CNN detection&#x2014;the weights were adjusted to 60% for YOLOv7 and 40% for Faster R-CNN. This approach leveraged YOLOv7&#x2019;s speed while incorporating the robustness of Faster R-CNN for more reliable detection. Detected face images were resized to 224&#x00D7;224 pixels to prepare images for Model 2, which aimed to detect and classify bites.</p>
</sec>
</sec>
<sec id="sec15">
<label>2.2.2</label>
<title>Model 2: bite classification</title>
<p>For Model 2, we aimed to accurately classify bite events by leveraging deep learning on high-level facial features. This involved training a sequential model (LSTM) using manually annotated bite data while addressing class imbalance and optimizing classification performance. Post-processing techniques were applied to refine detections and minimize false positives, ensuring better accuracy for bite event identification from video data. Refer back to <xref ref-type="fig" rid="fig3">Figure 3</xref> for an overview of ByteTrack model development.</p>
<sec id="sec16">
<label>2.2.2.1</label>
<title>Step 1: ground truth for model 2&#x2014;manual annotation of bites</title>
<p>Manual annotated timestamps were used as ground truth for model training of bite instances. Coding was conducted using Noldus Observer XT v16 (Noldus, 1991). Bites of food, sips of water, and active eating time were coded using an established protocol developed by Pearce and colleagues (<xref ref-type="bibr" rid="ref12">12</xref>, <xref ref-type="bibr" rid="ref57">57</xref>, <xref ref-type="bibr" rid="ref58">58</xref>). All videos were coded by two independent research assistants. The inter-rater reliability for each behavior, calculated using intraclass correlation coefficients [ICC (<xref ref-type="bibr" rid="ref1">1</xref>, <xref ref-type="bibr" rid="ref3">3</xref>) i.e., two-way mixed-effects model for a single measure (<xref ref-type="bibr" rid="ref59">59</xref>)] was excellent for all eating events, ICCs &#x003E;0.98 (<xref ref-type="bibr" rid="ref25">25</xref>).</p>
</sec>
<sec id="sec17">
<label>2.2.2.2</label>
<title>Step 2: pre-processing for bite classification (model 2)</title>
<p>As described previously, detected face images were resized to 224&#x00D7;224 pixels to prepare images for the next steps, which included feature extraction through Efficient Net Convolutional Neural Network (EfficientNet CNN) and bite classification through Long Short-Term Memory Recurrent Neural Network (LSTM-RNN). Bites were tagged with a timestamp in seconds to map each detection to the corresponding video frame.</p>
<p>For bite classification, all videos (345 videos) were split into training, validation, and test sets (70&#x2013;15-15%) (<xref ref-type="bibr" rid="ref41">41</xref>, <xref ref-type="bibr" rid="ref42">42</xref>) while maintaining split consistency with the Model 1 Dataset (from face detection split). An average bite sequence was assessed to be 50 frames through visual inspection (i.e., ~1.7&#x202F;s). Bite sequences were selected as 50 frames with the manually annotated timestamp placed at the center (i.e., 25th frame). Non-bite sequences were selected with a 10-frame buffer between bite and non-bite sequences. An example of bite sequence labeling is shown in <xref ref-type="fig" rid="fig5">Figure 5</xref>.</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>Example of bite labeling; bite center marked in red and whole bite sequence marked in green.</p>
</caption>
<graphic xlink:href="fnut-12-1610363-g005.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Grid of blurry headshots, primarily of a child's profile, with various tracking numbers. Certain images are outlined in green and one in red, indicating specific focus or significance. Black bars obscure parts of each image.</alt-text>
</graphic>
</fig>
<p>Bite sequences with more than 45 valid frames (&#x2264;10% missing data) were retained and padded to retain constant sequence length of 50 frames. Padding involves adding placeholder frames, here all-black frames, to ensure all sequences have a consistent length to facilitate uniform processing and analysis. Masking was applied to ensure that the LSTM ignores padded frames, preventing it from learning patterns from missing or non-informative data (<xref ref-type="bibr" rid="ref60">60</xref>). Any sequences with fewer than 45 frames (&#x003E;10% missing frames) were discarded. The resulting training set contained 13,527 bite sequences (minority class) and 77,653 non-bite sequences (majority class).</p>
<sec id="sec18"><label>2.2.2.2.1</label><title>Addressing class imbalance and loss function</title> <p>Class imbalance is a common challenge in visual classification tasks, including food-related applications (<xref ref-type="bibr" rid="ref61">61</xref>). To address the significant class imbalance between bite (minority class) and non-bite (majority class) events, we implemented a hybrid sampling approach and a custom loss function (error minimization function; detailed in <xref rid="SM1" ref-type="supplementary-material">Supplemental material</xref>). This method combines random undersampling of the majority class (<xref ref-type="bibr" rid="ref62">62</xref>) and Synthetic Minority Over-sampling Technique (SMOTE)-based oversampling of the minority class (<xref ref-type="bibr" rid="ref63">63</xref>) to preserve as much information as possible. This combination approach simultaneously reduces the risk of information loss from extreme undersampling of the majority class and prevents redundancy from excessive oversampling of the minority class. The majority class (non-bites) was undersampled to 3x the size of the minority class (bites), where the undersampling ratio was chosen through grid search. After undersampling, we had 13,537 bites and 40,611 non-bite sequences. Next, the bite class was oversampled to match the non-bite class sample size using SMOTE (<xref ref-type="bibr" rid="ref63">63</xref>). The final Model 2 Dataset had 40,611 bite sequences and 40,611 non-bite sequences.</p></sec>
</sec>
<sec id="sec19">
<label>2.2.2.3</label>
<title>Step 3: bite vs. non-bite classification</title>
<p>We implemented a bite classification model using transfer learning with EfficientNet-CNN (<xref ref-type="bibr" rid="ref64">64</xref>), a lightweight convolutional neural network designed for image recognition. Bite and non-bite frame sequences were passed through the encoding layers of EfficientNet-CNN (but not the classification layer) to transform the images into a set of high-level nonlinear features. These features, along with their corresponding labels and masks, were then fed into an LSTM-RNN (<xref ref-type="bibr" rid="ref65">65</xref>), a commonly used time sequence model that can intake a sequence of images for action detection (i.e., bite classification).</p>
<sec id="sec20"><label>2.2.2.3.1</label><title>Training parameters for LSTM bite classification</title> <p>A bidirectional LSTM, a variation of LSTM that gathers information from both the beginning and end of a sequence, was used for bite detection. The grid search (<xref ref-type="bibr" rid="ref54">54</xref>) identified the optimal hyperparameters as follows: a batch size of 128, a learning rate of 5&#x202F;&#x00D7;&#x202F;10<sup>&#x2212;5</sup>, 40 training epochs, three hidden layers, and hidden size as 256. To reduce overfitting, we applied a dropout rate of 0.4 during training. This reduces dependence on any single feature and helps the model generalize better to unseen data. The model was trained using the Adam optimizer (<xref ref-type="bibr" rid="ref66">66</xref>), selected for its adaptive learning rate capabilities and computational efficiency. Early stopping was employed based on validation F1 score improvement with patience&#x202F;=&#x202F;5.</p><p>Dynamic thresholding was used during training to adjust the model&#x2019;s confidence level for each bite prediction. The model tested different thresholds, adjusting the point at which a prediction is considered correct (i.e., when the model is confident enough to label an event as bite). After testing multiple thresholds, 0.65 was determined to be the optimal value, providing the best balance between minimizing false positives and false negatives in the bite detection task. This ensured the model could detect bite events accurately without over- or under-predicting. Training, validation, and testing were conducted on a high-performance cluster with 800GB RAM and 32 CPU cores. The training time with selected hyperparameters was 26&#x202F;h and 35&#x202F;min.</p></sec>
</sec>
<sec id="sec21">
<label>2.2.2.4</label>
<title>Step 4: automatic bite detection from video</title>
<p><italic>Post hoc</italic> processing was applied to enhance bite detection precision using three techniques: temporal smoothing (<xref ref-type="bibr" rid="ref67">67</xref>), duplicate detection suppression (<xref ref-type="bibr" rid="ref68">68</xref>), and optical flow validation (<xref ref-type="bibr" rid="ref69">69</xref>). Temporal smoothing was achieved by applying a moving average over a 20-frame window to stabilize detection probabilities and reduce noise from transient movements. The purpose of this smoothing was to reduce noise and allow for detection of trends in the data. To prevent overcounting the same bite event, we enforced a 15-frame interval threshold, filtering out additional detections within this period. This threshold ensured that each detection was distinct, allowing for improved accuracy by spacing out events and reducing duplicates. Optical flow validation was employed to further reduce false positives. Using the Lucas-Kanade method (<xref ref-type="bibr" rid="ref69">69</xref>), key points were tracked over a 2-s window post-bite to confirm chewing motion. A small motion threshold of 0.02 ensured that detected events exhibited the typical small, repetitive motion of chewing, filtering out unrelated movements. Bite detection was conducted on a Dell XPS 15 laptop with a 4GB GPU, 8 CPU cores, and 16GB of RAM.</p>
</sec>
</sec>
</sec>
<sec id="sec22">
<label>2.3</label>
<title>Model performance</title>
<p>The ByteTrack pipeline was evaluated based on its accuracy and reliability in detecting bites from video data. Performance was assessed by testing the model on a designated video dataset (test set) and analyzing key metrics. ByteTrack&#x2019;s bite predictions were compared with manual annotations (<italic>n</italic>&#x202F;=&#x202F;51 videos) using both Pearson correlation and simple linear regression. Correlation was used to assess the strength of association, and regression was used to evaluate the linear fit and prediction error for bite count and meal duration. Agreement between methods was further evaluated using ICC and Bland Altman analysis.</p>
<p>Simple linear-regression coefficients (intake ~ bite count) were computed to relate predicted bite counts to measured intake (<italic>n</italic>&#x202F;=&#x202F;50 videos; <italic>n</italic>&#x202F;=&#x202F;1 excluded for missing intake).</p>
<p>The overall ByteTrack pipeline with face detection (Model 1) followed by bite classification (Model 2) were utilized for bite classification and identifying the timestamp at which the bite occurred in the video. For bite timestamping, we selected the frame with the peak detection probability over each event, which was then converted to seconds using a 30 FPS frame rate. To accommodate computational delay, a 10-s margin around each manual timestamp was applied, marking detections within this window as true positives (TP). This margin accounts for annotation variability, temporal smoothing effects that may shift predictions, and the focus on bite count over exact timing. Since bite detection prioritizes detecting the correct number of bites rather than precise frame-level accuracy, this margin ensures a fairer evaluation aligned with real-world use cases. Missed manual bites were labeled false negatives (FN), and extra detections by the model were false positives (FP). Similar to the individual model performance, the same metrics&#x2014;precision, recall, and F1 score were calculated for each video in the test set. The overall precision, recall, and F1 scores were found by taking the arithmetic mean for all 51 videos. Detailed information on video-to-video on performance metrics is available in <xref rid="SM1" ref-type="supplementary-material">Supplementary Table 2</xref>.</p>
<sec id="sec23">
<label>2.3.1</label>
<title>Model performance metrics</title>
<p>Model 1 and Model 2 were tested individually on their respective test sets to calculate common performance metrics, such as precision, recall, and F1 score.</p>
<list list-type="simple">
<list-item>
<p>(1) Precision, which indicates how many detected bites were true and helps assess false positives, was calculated as the proportion of detected bites that were actual events.</p>
</list-item>
</list>
<disp-formula id="E1">
<mml:math id="M1">
<mml:mtext mathvariant="italic">Precision</mml:mtext>
<mml:mo stretchy="true">(</mml:mo>
<mml:mo>%</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext mathvariant="italic">True Positives</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi mathvariant="italic">TP</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">True Positives</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi mathvariant="italic">TP</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:mtext mathvariant="italic">False Positives</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi mathvariant="italic">FP</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mspace width="0em"/>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:math>
</disp-formula>
<list list-type="simple">
<list-item>
<p>(2) Recall assessed the ability to identify all actual bite events by calculating the proportion of true bites correctly detected:</p></list-item>
</list>
<disp-formula id="E2">
<mml:math id="M2">
<mml:mtext mathvariant="italic">Recall</mml:mtext>
<mml:mo stretchy="true">(</mml:mo>
<mml:mo>%</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext mathvariant="italic">True Positives</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi mathvariant="italic">TP</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">True Positives</mml:mtext>
<mml:mspace width="0.25em"/>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi mathvariant="italic">TP</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:mtext mathvariant="italic">False Negatives</mml:mtext>
<mml:mo stretchy="true">(</mml:mo>
<mml:mi mathvariant="italic">FN</mml:mi>
<mml:mo stretchy="true">)</mml:mo>
<mml:mspace width="0em"/>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:math>
</disp-formula>
<list list-type="simple">
<list-item>
<p>(3) F1 score provides a balanced assessment of the model&#x2019;s ability to accurately detect bites while minimizing false detections by calculating the harmonic mean of precision and recall:</p>
</list-item>
</list>
<disp-formula id="E3">
<mml:math id="M3">
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mspace width="0.25em"/>
<mml:mtext mathvariant="italic">score</mml:mtext>
<mml:mo stretchy="true">(</mml:mo>
<mml:mo>%</mml:mo>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext mathvariant="italic">Precision</mml:mtext>
<mml:mo>&#x00D7;</mml:mo>
<mml:mtext mathvariant="italic">Recall</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext mathvariant="italic">Precision</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext mathvariant="italic">Recall</mml:mtext>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:math>
</disp-formula>
</sec>
<sec id="sec24">
<label>2.3.2</label>
<title>Inter-rater reliability</title>
<p>To assess the reliability of automated bite detection, we used the Intraclass Correlation Coefficient (ICC) (<xref ref-type="bibr" rid="ref59">59</xref>), a two-way mixed-effects model for a single measure. This model, appropriate for a fixed set of raters (one of the human raters for ground truth and the automated detection model, ByteTrack), evaluated consistency in bite event identification. Calculating ICC provided a measure of agreement between ByteTrack and human annotations, focusing on consistent detection across repeated measures within each subject. We then averaged ICC values across subjects to assess the overall reliability of the model&#x2019;s performance across the dataset.</p>
</sec>
<sec id="sec25">
<label>2.3.3</label>
<title>Assessment of model eating behavior detection</title>
<p>Multiple metrics were used to assess systematic errors and overall performance of the ByteTrack bite count and meal duration predictions against manual ground truth. Scatterplots were employed to visualize the relationship between modeled and manual metrics and identify trends in overestimation or underestimation and quantitative measures such as Root Mean Square Error (RMSE), percentage RMSE (%RMSE), and error percentage (Error %) capture the magnitude and nature of deviations. Additionally, to understand the relationship of predicted bite count with actual intake (<italic>n</italic>&#x202F;=&#x202F;50, 1 participant excluded due to unavailability of objective intake measure), correlations between predicted bite count and measured energy intake (kcal) and gram intake at meals were calculated. The specific metrics were:</p>
<list list-type="simple">
<list-item>
<p>(1) Slope, which reflects proportional errors, with values &#x003E;1 indicating overestimation and &#x003C;1 indicating underestimation with a 45&#x00B0; line (y&#x202F;=&#x202F;x) representing perfect agreement.</p>
</list-item>
<list-item>
<p>(2) Intercept, which reflects any consistent bias or offset.</p>
</list-item>
<list-item>
<p>(3) R<sup>2</sup>, which provides an overall measure of how well the modeled values explain the variance in manual values.</p>
</list-item>
<list-item>
<p>(4) RMSE, which assesses raw error while maintaining units by calculating the average deviation between predicted and manual metrics (i.e., RMSE tells how far off the model is on average from the true values):</p>
</list-item>
</list>
<disp-formula id="E4">
<mml:math id="M4">
<mml:mtext>RMSE</mml:mtext>
<mml:mo>=</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mo movablelimits="false">&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math id="M5">
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>&#x202F;=&#x202F;manual value, <inline-formula>
<mml:math id="M6">
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>= modeled value, N&#x202F;=&#x202F;total number of observations.</p>
<list list-type="simple">
<list-item>
<p>(5) RMSE%, which allows for comparison between metrics by normalizing RMSE relative to the mean of the manual metrics:</p>
</list-item>
</list>
<disp-formula id="E5">
<mml:math id="M7">
<mml:mtext>RMSE</mml:mtext>
<mml:mo>%</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mtext>RMSE</mml:mtext>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
</mml:mfrac>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math id="M8">
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> = mean of the manual values, calculated as <inline-formula>
<mml:math id="M9">
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mo movablelimits="false">&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula></p>
<list list-type="simple">
<list-item>
<p>(6) Error %, which assesses localized patterns of bias in the model&#x2019;s predictions by calculating the deviation between predicted and actual bite counts across different videos.</p></list-item>
</list>
<disp-formula id="E6">
<mml:math id="M10">
<mml:mtext>Error</mml:mtext>
<mml:mo>%</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="true">)</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mfrac>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math id="M11">
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>&#x202F;=&#x202F;manual value, <inline-formula>
<mml:math id="M12">
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">&#x0302;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula>= modeled value</p>
<p>We conducted a retrospective visual review of the videos to gain a general understanding of where the model performed well or poorly. This visual inspection aimed to estimate potential reasons for mismatches in bite count and meal duration between the model and manual annotations.</p>
</sec>
</sec>
</sec>
<sec sec-type="results" id="sec26">
<label>3</label>
<title>Results</title>
<sec id="sec27">
<label>3.1</label>
<title>Model performance</title>
<sec id="sec28">
<label>3.1.1</label>
<title>Model 1&#x2014;face detection</title>
<p>Both YOLOv7 and Faster RCNN were evaluated using an Intersection over Union (IoU) threshold of 0.5, a standard measure in object detection. An IoU threshold of 0.5 means that a prediction is considered correct if the predicted bounding box overlaps with at least 50% of the actual object&#x2019;s (manually labeled) bounding box. This threshold is widely used because it provides a balanced approach to precision and recall, ensuring predictions are accurate without being overly strict. A threshold lower than 0.5 might allow too many false positives, while a higher threshold could miss valid detections that are not perfectly aligned.</p>
<p>YOLOv7 achieved a precision of 98.12%, recall of 94.35%, and an F1 score of 96.98%. These results demonstrate YOLOv7&#x2019;s ability to detect faces quickly and accurately. YOLOv7 also shows slightly lower recall (i.e., misses some faces). Faster RCNN had a precision of 92.94%, recall of 98.75%, and an F1 score of 95.76%, with higher recall than YOLOv7. We therefore used the faster model YOLOv7 as the primary model with Faster RCNN as a fallback.</p>
<p>This combination of YOLOv7 as primary model with Faster RCNN as fallback, gave us a precision of 99.24%, recall of 98.25%, and F1 score of 98.74% at an IoU&#x202F;=&#x202F;0.5.</p>
</sec>
<sec id="sec29">
<label>3.1.2</label>
<title>Model 2&#x2014;bite classification</title>
<p>The LSTM-RNN model achieved a mean precision of 72.8%, mean recall of 80.9% and an mean F1 score of 76.2% for bite detection across the test dataset (<italic>n</italic>&#x202F;=&#x202F;51 videos). This performance was evaluated on a test set comprising 3,776 bite sequences and 22,140 non-bite sequences in a sequence-to-sequence analysis at a confidence threshold of 0.65. This is a sequence-to-sequence analysis, i.e., measuring performance on chunks of image sequences (images from Model 1), which allows for a controlled evaluation of the model&#x2019;s bite classification ability, independent of continuous video tracking errors, frame inconsistencies, and temporal noise. By focusing on pre-segmented sequences derived from object detection, this approach isolates the LSTM&#x2019;s performance, ensuring that the assessment reflects its ability to recognize temporal patterns without the confounding effects of tracking stability.</p>
</sec>
<sec id="sec30">
<label>3.1.3</label>
<title>ByteTrack performance&#x2014;bite detection</title>
<p>ByteTrack&#x2019;s bite detection performance was evaluated on 51 videos from 42 children, achieving an average precision of 79.4%, recall of 67.9%, and an F1 score of 70.6% with a 10-s margin from ground truth. We see large variability between subjects, with precision ranging from 38.2 to 100%, recall from 17.6 to 93.6%, and F1 score from 26.3 to 91.2%. Post-hoc smoothing likely improved precision by filtering spurious detections but reduced recall by removing some true bites. The confusion matrix from the ByteTrack system on the test set (<italic>n</italic>&#x202F;=&#x202F;51 videos) is shown in <xref ref-type="table" rid="tab2">Table 2</xref>.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Confusion matrix on test set for bite detected in test video data using ByteTrack (<italic>n</italic>&#x202F;=&#x202F;51 videos).</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th colspan="2" rowspan="2"/>
<th align="center" valign="top" colspan="2">Predicted classes</th>
</tr>
<tr>
<th align="center" valign="top">Bite</th>
<th align="center" valign="top">Non-bite</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top" rowspan="2">Actual class</td>
<td align="center" valign="top">Bite</td>
<td align="center" valign="top">5,213 (TP)</td>
<td align="center" valign="top">1,842 (FN)</td>
</tr>
<tr>
<td align="center" valign="top">Non-bite</td>
<td align="center" valign="top">1,653 (FP)</td>
<td align="center" valign="top">Unknown (TN)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>TP, True positives; FP, False positives; TN, True negatives; FN, False negatives.</p>
</table-wrap-foot>
</table-wrap>
<p>Although we did not log inference time per video, ByteTrack typically processed a 30-min video in ~25&#x2013;30&#x202F;min on a Dell XPS 15 laptop (4GB GPU), depending on activity level. In contrast, manual double-coded annotation took ~70&#x2013;80&#x202F;min per video.</p>
</sec>
</sec>
<sec id="sec31">
<label>3.2</label>
<title>ByteTrack performance relative to gold-standard manual annotation</title>
<sec id="sec32">
<label>3.2.1</label>
<title>Inter-rater reliability</title>
<p>The reliability of bite events between manually coded data and ByteTrack or inter-rater reliability, measured using ICC, showed moderate reliability (<xref ref-type="bibr" rid="ref59">59</xref>) with a mean value of 0.66 and a range of 0.24&#x2013;0.99.</p>
</sec>
<sec id="sec33">
<label>3.2.2</label>
<title>Bite count</title>
<p>The scatter plot comparing modeled and manual bite counts shows consistent overestimation by the model (<xref ref-type="fig" rid="fig6">Figure 6A</xref>). Linear regression fitted across all data points produced a slope of 0.79 and an intercept of 56.48, with an R<sup>2</sup> of 0.12 and a Pearson correlation coefficient of <italic>r</italic>&#x202F;=&#x202F;0.35, indicating a weak linear association between modeled and manual counts. The mean of the per-subject RMSE was 61.6 bites and mean per-subject RMSE% of 96.9%. The mean per-subject error% was 72.9%. The model captures the general bite count trend, with predicted counts correlated to ground truth. It overestimates on average and shows high variability across subjects. Children with higher true bite counts are generally ranked higher, despite errors in exact values. A Bland&#x2013;Altman plot depicting the differences in bite counts can be found in <xref rid="SM1" ref-type="supplementary-material">Supplementary Figure 2</xref>. The Bland&#x2013;Altman plot shows that model-predicted bite counts were on average 47.6 bites higher than manual counts, with 95% limits of agreement ranging from &#x2212;197.8 to 112.6 bites (manual&#x2014;model), indicating that differences between the two methods spanned from the model predicting more bites to the manual count exceeding the model. The differences appear to widen with increasing average bite counts.</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>Scatter plots showing the relationship between manual (ground truth) and modeled (predicted by ByteTrack) eating behavior metrics (<italic>n</italic>&#x202F;=&#x202F;51 videos), assessed using both Pearson correlation and simple linear regression. The red line represents ideal agreement (y&#x202F;=&#x202F;x), and the blue line shows the fitted regression line. Each point represents one test video. <bold>(A)</bold> Manual vs. modeled bite count. <bold>(B)</bold> Manual vs. modeled meal duration. RMSE, Root Mean Square Error, measures the average magnitude of prediction error; RMSE%, RMSE expressed as a percentage of the mean manual value; Error%, Average absolute percentage difference between manual and modeled metrics; R<sup>2</sup>, Coefficient of determination from the regression model; r, Pearson correlation coefficient.</p>
</caption>
<graphic xlink:href="fnut-12-1610363-g006.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Scatter plots showing correlations between modeled and manual measurements. A: Modeled Bite Count vs Manual Bite Count with R&#x00B2; = 0.12 and r = 0.35. B: Modeled Meal Duration vs Manual Meal Duration with R&#x00B2; = 0.69 and r = 0.83. Each plot includes a linear regression line (solid blue) and reference line (dashed red).</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec34">
<label>3.2.3</label>
<title>Meal duration</title>
<p>The relationship between manual and model-calculated meal duration is shown in <xref ref-type="fig" rid="fig6">Figure 6B</xref> indicating a moderate positive relationship between modeled and manual durations. A linear regression fitted across all data points produced a slope of 0.64 and an intercept of 2.53 with an R<sup>2</sup> of 0.69, indicating moderate linear association between modeled and manual computed meal duration. The mean of the per-subject RMSE was 4.39&#x202F;min, with a mean per-subject RMSE% of 28.3%. The mean percentage error, derived from the percent error per subject, was &#x2212;16.0%, indicating a systematic underestimation.</p>
</sec>
<sec id="sec35">
<label>3.2.4</label>
<title>ByteTrack performance with real-world eating behavior outcomes</title>
<p>Simple linear regression models were used to assess the ability of ByteTrack to model eating behavior that relates to real-world outcomes such as meal intake. The relationship between meal energy intake (kcal) and gram intake (g) and modeled bite counts is shown in <xref ref-type="fig" rid="fig7">Figures 7A</xref>,<xref ref-type="fig" rid="fig7">B</xref> respectively. The relationships between meal energy and gram intake with the manual annotations are in <xref ref-type="fig" rid="fig7">Figures 7C</xref>,<xref ref-type="fig" rid="fig7">D</xref>. The relationship between modeled bite count and meal intake shows weak but clear trends. The R<sup>2</sup> values (regression coefficient) are low (<italic>R</italic><sup>2</sup>&#x202F;=&#x202F;0.05 for kcal, <italic>R</italic><sup>2</sup>&#x202F;=&#x202F;0.06 for grams), with high variability in how much modeled bite count predicts intake. Both figures show positive slopes (i.e., higher bite count associated with higher intake) between modeled bite counts and measured intake. Substantial inter-individual variability is seen in the plots. Positive slopes in both figures show higher bite counts are generally associated with greater intake. While the associations are weaker than those observed with manual bite counts (<italic>R</italic><sup>2</sup>&#x202F;=&#x202F;0.42 for kcal, <italic>R</italic><sup>2</sup>&#x202F;=&#x202F;0.53 for grams), the trends remain evident, suggesting that modeled bite count captures meaningful intake patterns despite variability across individuals.</p>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>Scatter plots showing the correlation between modeled or predicted bite counts <bold>(A,B)</bold> and manual or ground truth <bold>(C,D)</bold> vs. actual meal intake (grams or kcal; <italic>n</italic>&#x202F;=&#x202F;50 videos; <italic>n</italic>&#x202F;=&#x202F;1 missing measured meal intake). <bold>(A)</bold> Scatter plots showing modeled bite counts vs. actual calculated energy (kcal) intake at meal. <bold>(B)</bold> Scatter plots showing modeled bite counts vs. actual calculated gram intake at meal. <bold>(C)</bold> Scatter plots showing manual bite counts vs. actual calculated energy (kcal) intake at meal. <bold>(D)</bold> Scatter plots showing manual bite counts vs. actual calculated gram intake at meal. R<sup>2</sup>, coefficient of determination from simple linear regression (intake ~ bite count).</p>
</caption>
<graphic xlink:href="fnut-12-1610363-g007.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Four scatter plots labeled A to D. A: Shows the relationship between modeled bite count and total kcal, with a trend line having equation y = 0.71x + 449.92 and R&#x00B2; = 0.05.B: Displays modeled bite count versus total grams, with a trend line y = 0.57x + 371.50 and R&#x00B2; = 0.06.C: Illustrates manual bite count against total kcal, with a trend line y = 4.84x + 180.75 and R&#x00B2; = 0.42.D: Depicts manual bite count versus total grams, with a trend line y = 4.10x + 139.76 and R&#x00B2; = 0.53.</alt-text>
</graphic>
</fig>
</sec>
</sec>
</sec>
<sec sec-type="discussion" id="sec36">
<label>4</label>
<title>Discussion</title>
<p>ByteTrack demonstrated moderate performance, with an average F1 score of 71% and an inter-rater reliability (ICC&#x202F;=&#x202F;0.66) when compared to manually annotated ground truths. To our knowledge, this is the first automated system specifically developed to analyze eating behaviors in children, whose video data presents unique challenges due to frequent movements and occlusions. ByteTrack serves as a proof-of-concept for automated bite detection in children and suggests a promising future for this direction of research.</p>
<p>To support robust bite detection, the first stage (part 1) of the pipeline focused on accurate face localization despite child movement and occlusion. A two-stage detection strategy was used, in which a fast, high-precision YOLOv7 model served as the primary detector, while a higher-recall Faster R-CNN acted as a fallback in cases of missed detections. This design allowed the system to maintain high face detection performance (recall and precision &#x003E;98%), balancing the need for speed with tolerance to the visual variability common in child mealtime videos.</p>
<p>Bite detection (part 2) showed greater variability but moderate performance across subjects (mean F1&#x202F;=&#x202F;71.3%; ICC&#x202F;=&#x202F;0.66). However, total overall bite counts were generally inflated, with over-firing concentrated in the early portion of meals and under-firing during later or longer sessions. Retrospective visual inspection of videos suggests several possible contributors. Rapid, closely spaced bites, often involving brief spoon nibbling, may blur event boundaries and lead to extra detections. As meals progress, children tend to shift focus or play with food, producing more body movement and occlusions that can suppress detections. Together, these factors appear to let the model identify the general timing of bites yet trigger too frequently around true events at the start and too sparingly as eating slows, leading to shorter estimated meal durations.</p>
<p>While the video data used for ByteTrack was collected in controlled laboratory settings, the conditions of recording simulated a more natural mealtime environment in that additional people were present to engage with children (~80% videos in training data and ~82% videos in test set with additional person). This approach contrasts with previous systems developed for adults in tightly controlled settings (<xref ref-type="bibr" rid="ref18">18</xref>, <xref ref-type="bibr" rid="ref20">20</xref>) by accommodating the unique behavioral patterns and interactions typical in a child&#x2019;s meal. However, the majority of a child&#x2019;s food intake at this age takes place at home and school (<xref ref-type="bibr" rid="ref70 ref71 ref72">70&#x2013;72</xref>), therefore future studies are needed to improve the flexibility of ByteTrack to evaluate eating behaviors in these diverse settings.</p>
<p>Traditional assessment methods for eating rate rely on self-reporting, which is often inaccurate due to memory lapses and social desirability bias (<xref ref-type="bibr" rid="ref73">73</xref>, <xref ref-type="bibr" rid="ref74">74</xref>). More objective measurements come from wearable devices and video-based monitoring. Wearable devices, such as bite counter watches (<xref ref-type="bibr" rid="ref75">75</xref>) and sensor-based eyeglasses (<xref ref-type="bibr" rid="ref76">76</xref>), can track bites. But there are limitations with these devices as they require researchers or users to start and stop data collection which can be intrusive to the natural eating process. Video-based monitoring methods like ByteTrack, while also requiring similar manual start/stop, offer a less intrusive approach for measuring meal eating behaviors that aligns with current gold standards of manual observational coding. Accurate, automated, real-world video-based approaches may enable the use of smartphone cameras for passive dietary monitoring in naturalistic settings, such as at home, creating new opportunities for scalable dietary data collection and intervention. Applying home recording methods with ByteTrack for automated bite detection provides a practical solution for capturing meal and snack intake, enabling the estimation of food intake and eating rates in natural settings. However, as these technologies advance, ensuring data privacy and encryption will be critical for secure handling of sensitive information (<xref ref-type="bibr" rid="ref77">77</xref>).</p>
<p>The goal for future iterations of ByteTrack will be to replace or supplement manual observational coding, as it is a highly time and resource-intensive process. In the current study, double-coded manual annotation took approximately 80&#x202F;min per 30-min video (&#x003E;40&#x202F;h total for the dataset), posing challenges for scaling to larger datasets. In contrast, ByteTrack completed the same task in typically 25&#x2013;30&#x202F;min per 30-min video, with minimal human input beyond initiating the script. However, this version of ByteTrack is not yet optimized for real-time bite detection. Human annotators may also introduce variability due to differences in interpretation, fatigue, or experience, which is why double coding is used to ensure reliability by resolving discrepancies between two independent annotations. In contrast, automated coding can apply consistent criteria across all videos, eliminating the need for double coding and improving research efficiency. Once refined, automated approaches like ByteTrack could enhance the ability to study human eating behavior outside the laboratory.</p>
<p>While the ByteTrack model had moderate F1 scores, which demonstrated good alignment with manual annotation, performance variability of the model highlights areas for improvement in future iterations. The ByteTrack system was less accurate when children are rapidly and had pronounced head and hand movements, potentially leading to a higher number of false positives (i.e., mistaking these movements for bites). False negatives occurred when bite motions were occluded, such as when a child&#x2019;s hand, utensils, or other objects blocked the view of their mouth, which led to missed bite events and lower recall in detection accuracy. Additionally, it appears that the model overestimates the bite count during the initial rapid eating phase, when the child is more focused on eating. As the meal progresses and becomes longer, the child may slow down, move around, or lose attention to the food, leading to increased occlusions and missed bite events. This shift could result in the overestimation of bite counts early in the meal and the underestimation of meal duration later on, as fewer bites are detected when eating slows and occlusions become more frequent. This pattern of overestimating bite counts at the start of longer meals and underestimating meal duration in the later stages seems to contribute to overall inaccuracies in both bite count and meal duration estimation. Another limitation of ByteTrack is that there was no explicit modeling to differentiate bites from sips, which could have led to a misclassification of sips as bites. These challenges underscore the need to further refine the ByteTrack model to enhance its robustness in naturalistic eating scenarios.</p>
<p>Despite its limitations, there are several strengths to the current iteration of ByteTrack. As a non-intrusive, video-based system, it provides an alternative to wearable sensors and sets the stage for large-scale, automated detection by reducing reliance on manual annotation. The deep learning architecture used to construct ByteTrack combined state of the art methods (e.g., YOLOv7, Faster R-CNN, and LSTM) to achieve accurate face detection while accounting for the unique movement patterns of children. Furthermore, extending ByteTrack&#x2019;s application beyond the lab to home and school environments, a direction in our ongoing studies, will further validate ByteTrack&#x2019;s performance and enhance its real-world applicability.</p>
<p>Future iterations of ByteTrack will enhance robustness by incorporating diverse training data, including varied lighting conditions, movement patterns, and occlusions (<xref ref-type="bibr" rid="ref78">78</xref>). Action detection in real-world video remains challenging due to the variability in human movement and environmental conditions (e.g., lighting, cluttered backgrounds). Data augmentation techniques, such as occlusion augmentation (e.g., adding synthetic hands, utensils, or objects partially covering the mouth), motion blur to reflect natural head movements, and temporal adjustments like varying frame rates or inserting brief distractions, can help simulate real-world eating scenarios (<xref ref-type="bibr" rid="ref45">45</xref>). Additionally, integrating inter-subject variability by using subject identity as a model feature can improve the system&#x2019;s ability to distinguish bites from non-bites (<xref ref-type="bibr" rid="ref79">79</xref>). Explicitly modeling bite and sip classification separately may improve accuracy and reduce bite overestimation in the current model. Incorporating more data from real-world smartphone videos may further enhance performance and practical utility. Moreover, ByteTrack&#x2019;s bite-count output could also be paired with complementary tools for portion-size estimation and food identification (<xref ref-type="bibr" rid="ref80">80</xref>, <xref ref-type="bibr" rid="ref81">81</xref>) to yield more precise, holistic measures of meal microstructure and dietary intake in future work.</p>
<p>ByteTrack is a proof-of-concept, automatic bite detection framework for easing the time and resources required for manual video annotation. This represents a first step toward scalable, automated bite detection for the measurement of meal-related eating behaviors in children. With additional testing and model improvements, ByteTrack may expand the ability to capture real-time changes in human eating behaviors measured outside the laboratory.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec37">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found at: <ext-link xlink:href="https://github.com/YashuBhat96/ByteTrack" ext-link-type="uri">https://github.com/YashuBhat96/ByteTrack</ext-link> and <ext-link xlink:href="https://osf.io/g6muv/" ext-link-type="uri">https://osf.io/g6muv/</ext-link>.</p>
</sec>
<sec sec-type="ethics-statement" id="sec38">
<title>Ethics statement</title>
<p>The studies involving humans were approved by the Pennsylvania State University IRB. The studies were conducted in accordance with the local legislation and institutional requirements. Written informed consent for participation in this study was provided by the participants&#x2019; legal guardians/next of kin.</p>
</sec>
<sec sec-type="author-contributions" id="sec39">
<title>Author contributions</title>
<p>YRB: Conceptualization, Methodology, Software, Formal analysis, Investigation, Data curation, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. KK: Funding acquisition, Writing &#x2013; review &#x0026; editing, Project administration, Supervision, Conceptualization, Data curation, Resources. TB: Resources, Conceptualization, Methodology, Writing &#x2013; review &#x0026; editing, Project administration, Supervision. AP: Project administration, Supervision, Writing &#x2013; review &#x0026; editing, Conceptualization, Funding acquisition.</p>
</sec>
<sec sec-type="funding-information" id="sec40">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This research was supported in part by a Seed Grant award from the Institute for Computational and Data Sciences at the Pennsylvania State University. Funding support also comes from the National Institutes of Health (NIH) R01DK110060 and R01DK126050 [principal investigator (PI): Kathleen L. Keller]. These methods were supported by the National Center for Advancing Translational Sciences, grant U54 TR002014-05A1. Additional salary supported for Kathleen L. Keller comes from the United States Department of Agriculture (USDA) Hatch Grant PEN04708.</p>
</sec>
<ack>
<p>We thank the children and parents who participated in this study.</p>
</ack>
<sec sec-type="COI-statement" id="sec41">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec42">
<title>Generative AI statement</title>
<p>The authors declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="sec43">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec44">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fnut.2025.1610363/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fnut.2025.1610363/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.DOCX" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn id="fn0001"><p><sup>1</sup><ext-link xlink:href="https://fdc.nal.usda.gov/" ext-link-type="uri">https://fdc.nal.usda.gov/</ext-link></p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="ref1"><label>1.</label><citation citation-type="book"><person-group person-group-type="author"><name><surname>Kissileff</surname><given-names>HR</given-names></name> <name><surname>Thornton</surname><given-names>J</given-names></name></person-group>. <article-title>Facilitation and inhibition in the cumulative food intake curve in man</article-title> In: <source>Changing concepts of the nervous system</source>. <publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Elsevier</publisher-name> (<year>1982</year>)</citation></ref>
<ref id="ref2"><label>2.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bobroff</surname><given-names>EM</given-names></name> <name><surname>Kissileff</surname><given-names>HR</given-names></name></person-group>. <article-title>Effects of changes in palatability on food intake and the cumulative food intake curve in man</article-title>. <source>Appetite</source>. (<year>1986</year>) <volume>7</volume>:<fpage>85</fpage>&#x2013;<lpage>96</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0195-6663(86)80044-7</pub-id>, PMID: <pub-id pub-id-type="pmid">3963801</pub-id></citation></ref>
<ref id="ref3"><label>3.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kissileff</surname><given-names>HR</given-names></name> <name><surname>Klingsberg</surname><given-names>G</given-names></name> <name><surname>Van Itallie</surname><given-names>TB</given-names></name></person-group>. <article-title>Universal eating monitor for continuous recording of solid or liquid consumption in man</article-title>. <source>Am J Phys Regul Integr Comp Phys</source>. (<year>1980</year>) <volume>238</volume>:<fpage>R14</fpage>&#x2013;<lpage>22</lpage>. doi: <pub-id pub-id-type="doi">10.1152/ajpregu.1980.238.1.R14</pub-id>, PMID: <pub-id pub-id-type="pmid">7356043</pub-id></citation></ref>
<ref id="ref4"><label>4.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Langlet</surname><given-names>BS</given-names></name> <name><surname>Anvret</surname><given-names>A</given-names></name> <name><surname>Maramis</surname><given-names>C</given-names></name> <name><surname>Moulos</surname><given-names>I</given-names></name> <name><surname>Papapanagiotou</surname><given-names>V</given-names></name> <name><surname>Diou</surname><given-names>C</given-names></name> <etal/></person-group>. <article-title>Objective measures of eating behaviour in a Swedish high school</article-title>. <source>Behav Inf Technol</source>. (<year>2017</year>) <volume>36</volume>:<fpage>1005</fpage>&#x2013;<lpage>13</lpage>. doi: <pub-id pub-id-type="doi">10.1080/0144929X.2017.1322146</pub-id></citation></ref>
<ref id="ref5"><label>5.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fogel</surname><given-names>A</given-names></name> <name><surname>Goh</surname><given-names>AT</given-names></name> <name><surname>Fries</surname><given-names>LR</given-names></name> <name><surname>Sadananthan</surname><given-names>SA</given-names></name> <name><surname>Velan</surname><given-names>SS</given-names></name> <name><surname>Michael</surname><given-names>N</given-names></name> <etal/></person-group>. <article-title>Faster eating rates are associated with higher energy intakes during an ad libitum meal, higher BMI and greater adiposity among 4 5-year-old children: results from the growing up in Singapore towards healthy outcomes (GUSTO) cohort</article-title>. <source>Br J Nutr</source>. (<year>2017</year>) <volume>117</volume>:<fpage>1042</fpage>&#x2013;<lpage>51</lpage>. doi: <pub-id pub-id-type="doi">10.1017/S0007114517000848</pub-id>, PMID: <pub-id pub-id-type="pmid">28462734</pub-id></citation></ref>
<ref id="ref6"><label>6.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Berkowitz</surname><given-names>RI</given-names></name> <name><surname>Moore</surname><given-names>RH</given-names></name> <name><surname>Faith</surname><given-names>MS</given-names></name> <name><surname>Stallings</surname><given-names>VA</given-names></name> <name><surname>Kral</surname><given-names>TVE</given-names></name> <name><surname>Stunkard</surname><given-names>AJ</given-names></name></person-group>. <article-title>Identification of an obese eating style in 4-year-old children born at high and low risk for obesity</article-title>. <source>Obesity</source>. (<year>2010</year>) <volume>18</volume>:<fpage>505</fpage>&#x2013;<lpage>12</lpage>. doi: <pub-id pub-id-type="doi">10.1038/oby.2009.299</pub-id>, PMID: <pub-id pub-id-type="pmid">19779474</pub-id></citation></ref>
<ref id="ref7"><label>7.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Llewellyn</surname><given-names>CH</given-names></name> <name><surname>van Jaarsveld</surname><given-names>CHM</given-names></name> <name><surname>Boniface</surname><given-names>D</given-names></name> <name><surname>Carnell</surname><given-names>S</given-names></name> <name><surname>Wardle</surname><given-names>J</given-names></name></person-group>. <article-title>Eating rate is a heritable phenotype related to weight in children</article-title>. <source>Am J Clin Nutr</source>. (<year>2008</year>) <volume>88</volume>:<fpage>1560</fpage>&#x2013;<lpage>6</lpage>. doi: <pub-id pub-id-type="doi">10.3945/ajcn.2008.26175</pub-id>, PMID: <pub-id pub-id-type="pmid">19064516</pub-id></citation></ref>
<ref id="ref8"><label>8.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fogel</surname><given-names>A</given-names></name> <name><surname>Goh</surname><given-names>AT</given-names></name> <name><surname>Fries</surname><given-names>LR</given-names></name> <name><surname>Sadananthan</surname><given-names>SA</given-names></name> <name><surname>Velan</surname><given-names>SS</given-names></name> <name><surname>Michael</surname><given-names>N</given-names></name> <etal/></person-group>. <article-title>A description of an &#x2018;obesogenic&#x2019;eating style that promotes higher energy intake and is associated with greater adiposity in 4.5 year-old children: results from the GUSTO cohort</article-title>. <source>Physiol Behav</source>. (<year>2017</year>) <volume>176</volume>:<fpage>107</fpage>&#x2013;<lpage>16</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.physbeh.2017.02.013</pub-id>, PMID: <pub-id pub-id-type="pmid">28213204</pub-id></citation></ref>
<ref id="ref9"><label>9.</label><citation citation-type="other"><person-group person-group-type="author"><collab id="coll1">CDC Obesity</collab></person-group>. (<year>2024</year>) <source>Childhood obesity facts</source>. Available online at: <ext-link xlink:href="https://www.cdc.gov/obesity/childhood-obesity-facts/childhood-obesity-facts.html" ext-link-type="uri">https://www.cdc.gov/obesity/childhood-obesity-facts/childhood-obesity-facts.html</ext-link></citation></ref>
<ref id="ref10"><label>10.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Faith</surname><given-names>MS</given-names></name> <name><surname>Diewald</surname><given-names>LK</given-names></name> <name><surname>Crabbe</surname><given-names>S</given-names></name> <name><surname>Burgess</surname><given-names>B</given-names></name> <name><surname>Berkowitz</surname><given-names>RI</given-names></name></person-group>. <article-title>Reduced eating Pace (RePace) behavioral intervention for children prone to or with obesity: does the turtle win the race?</article-title> <source>Obesity</source>. (<year>2019</year>) <volume>27</volume>:<fpage>121</fpage>&#x2013;<lpage>9</lpage>. doi: <pub-id pub-id-type="doi">10.1002/oby.22329</pub-id>, PMID: <pub-id pub-id-type="pmid">30515992</pub-id></citation></ref>
<ref id="ref11"><label>11.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Scisco</surname><given-names>JL</given-names></name> <name><surname>Muth</surname><given-names>ER</given-names></name> <name><surname>Dong</surname><given-names>Y</given-names></name> <name><surname>Hoover</surname><given-names>AW</given-names></name></person-group>. <article-title>Slowing bite-rate reduces energy intake: an application of the bite counter device</article-title>. <source>J Am Diet Assoc</source>. (<year>2011</year>) <volume>111</volume>:<fpage>1231</fpage>&#x2013;<lpage>5</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jada.2011.05.005</pub-id>, PMID: <pub-id pub-id-type="pmid">21802572</pub-id></citation></ref>
<ref id="ref12"><label>12.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pearce</surname><given-names>AL</given-names></name> <name><surname>Cevallos</surname><given-names>MC</given-names></name> <name><surname>Romano</surname><given-names>O</given-names></name> <name><surname>Daoud</surname><given-names>E</given-names></name> <name><surname>Keller</surname><given-names>KL</given-names></name></person-group>. <article-title>Child meal microstructure and eating behaviors: a systematic review</article-title>. <source>Appetite</source>. (<year>2022</year>) <volume>168</volume>:<fpage>105752</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.appet.2021.105752</pub-id>, PMID: <pub-id pub-id-type="pmid">34662600</pub-id></citation></ref>
<ref id="ref13"><label>13.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pesch</surname><given-names>MH</given-names></name> <name><surname>Lumeng</surname><given-names>JC</given-names></name></person-group>. <article-title>Methodological considerations for observational coding of eating and feeding behaviors in children and their families</article-title>. <source>Int J Behav Nutr Phys Act</source>. (<year>2017</year>) <volume>14</volume>:<fpage>170</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12966-017-0619-3</pub-id>, PMID: <pub-id pub-id-type="pmid">29246234</pub-id></citation></ref>
<ref id="ref14"><label>14.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>McClung</surname><given-names>HL</given-names></name> <name><surname>Ptomey</surname><given-names>LT</given-names></name> <name><surname>Shook</surname><given-names>RP</given-names></name> <name><surname>Aggarwal</surname><given-names>A</given-names></name> <name><surname>Gorczyca</surname><given-names>AM</given-names></name> <name><surname>Sazonov</surname><given-names>ES</given-names></name> <etal/></person-group>. <article-title>Dietary intake and physical activity assessment: current tools, techniques, and technologies for use in adult populations</article-title>. <source>Am J Prev Med</source>. (<year>2018</year>) <volume>55</volume>:<fpage>e93</fpage>&#x2013;<lpage>e104</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.amepre.2018.06.011</pub-id>, PMID: <pub-id pub-id-type="pmid">30241622</pub-id></citation></ref>
<ref id="ref15"><label>15.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vu</surname><given-names>T</given-names></name> <name><surname>Lin</surname><given-names>F</given-names></name> <name><surname>Alshurafa</surname><given-names>N</given-names></name> <name><surname>Xu</surname><given-names>W</given-names></name></person-group>. <article-title>Wearable food intake monitoring technologies: a comprehensive review</article-title>. <source>Computers</source>. (<year>2017</year>) <volume>6</volume>:<fpage>4</fpage>. doi: <pub-id pub-id-type="doi">10.3390/computers6010004</pub-id></citation></ref>
<ref id="ref16"><label>16.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bell</surname><given-names>BM</given-names></name> <name><surname>Alam</surname><given-names>R</given-names></name> <name><surname>Alshurafa</surname><given-names>N</given-names></name> <name><surname>Thomaz</surname><given-names>E</given-names></name> <name><surname>Mondol</surname><given-names>AS</given-names></name> <name><surname>de la Haye</surname><given-names>K</given-names></name> <etal/></person-group>. <article-title>Automatic, wearable-based, in-field eating detection approaches for public health research: a scoping review</article-title>. <source>Npj Digit Med</source>. (<year>2020</year>) <volume>3</volume>:<fpage>1</fpage>&#x2013;<lpage>14</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41746-020-0246-2</pub-id>, PMID: <pub-id pub-id-type="pmid">32195373</pub-id></citation></ref>
<ref id="ref17"><label>17.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nour</surname><given-names>M</given-names></name> <name><surname>Gardoni</surname><given-names>M</given-names></name> <name><surname>Renaud</surname><given-names>J</given-names></name> <name><surname>Gauthier</surname><given-names>S</given-names></name></person-group>. <article-title>Real-time detection and motivation of eating activity in elderly people with dementia using pose estimation with TensorFlow and OpenCV</article-title>. <source>Adv Soc Sci Res J</source>. (<year>2021</year>) <volume>8</volume>:<fpage>28</fpage>&#x2013;<lpage>34</lpage>. doi: <pub-id pub-id-type="doi">10.14738/assrj.83.9763</pub-id></citation></ref>
<ref id="ref18"><label>18.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Konstantinidis</surname><given-names>D</given-names></name> <name><surname>Dimitropoulos</surname><given-names>K</given-names></name> <name><surname>Ioakimidis</surname><given-names>I</given-names></name> <name><surname>Langlet</surname><given-names>B</given-names></name> <name><surname>Daras</surname><given-names>P</given-names></name></person-group>. <article-title>A deep network for automatic video-based food bite detection</article-title>. In: <conf-name>Computer Vision Systems: 12th International Conference, ICVS</conf-name> (<year>2019</year>), <publisher-loc>Thessaloniki, Greece</publisher-loc>, <publisher-name>Springer</publisher-name>.</citation></ref>
<ref id="ref19"><label>19.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tufano</surname><given-names>M</given-names></name> <name><surname>Lasschuijt</surname><given-names>MP</given-names></name> <name><surname>Chauhan</surname><given-names>A</given-names></name> <name><surname>Feskens</surname><given-names>EJM</given-names></name> <name><surname>Camps</surname><given-names>G</given-names></name></person-group>. <article-title>Rule-based systems to automatically count bites from meal videos</article-title>. <source>Front Nutr</source>. (<year>2024</year>) <volume>11</volume>:<fpage>1343868</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fnut.2024.1343868</pub-id>, PMID: <pub-id pub-id-type="pmid">38826582</pub-id></citation></ref>
<ref id="ref20"><label>20.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hossain</surname><given-names>D</given-names></name> <name><surname>Ghosh</surname><given-names>T</given-names></name> <name><surname>Sazonov</surname><given-names>E</given-names></name></person-group>. <article-title>Automatic count of bites and chews from videos of eating episodes</article-title>. <source>IEEE Access</source>. (<year>2020</year>) <volume>8</volume>:<fpage>101934</fpage>&#x2013;<lpage>45</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ACCESS.2020.2998716</pub-id>, PMID: <pub-id pub-id-type="pmid">33747674</pub-id></citation></ref>
<ref id="ref21"><label>21.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rouast</surname><given-names>PV</given-names></name> <name><surname>Adam</surname><given-names>MTP</given-names></name></person-group>. <article-title>Learning deep representations for video-based intake gesture detection</article-title>. <source>IEEE J Biomed Health Inform</source>. (<year>2020</year>) <volume>24</volume>:<fpage>1727</fpage>&#x2013;<lpage>37</lpage>. doi: <pub-id pub-id-type="doi">10.1109/JBHI.2019.2942845</pub-id>, PMID: <pub-id pub-id-type="pmid">31567103</pub-id></citation></ref>
<ref id="ref22"><label>22.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hosseini</surname><given-names>A</given-names></name> <name><surname>Fazeli</surname><given-names>S</given-names></name> <name><surname>van Vliet</surname><given-names>E</given-names></name> <name><surname>Valencia</surname><given-names>L</given-names></name> <name><surname>Habre</surname><given-names>R</given-names></name> <name><surname>Sarrafzadeh</surname><given-names>M</given-names></name> <etal/></person-group>. <article-title>Children activity recognition: challenges and strategies</article-title>. <source>Conf Proc IEEE Eng Med Biol Soc</source>. (<year>2018</year>) <volume>2018</volume>:<fpage>4331</fpage>&#x2013;<lpage>4</lpage>. doi: <pub-id pub-id-type="doi">10.1109/EMBC.2018.8513320</pub-id></citation></ref>
<ref id="ref23"><label>23.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qian</surname><given-names>H</given-names></name> <name><surname>Mao</surname><given-names>Y</given-names></name> <name><surname>Xiang</surname><given-names>W</given-names></name> <name><surname>Wang</surname><given-names>Z</given-names></name></person-group>. <article-title>Recognition of human activities using SVM multi-class classifier</article-title>. <source>Patt Recogn Lett</source>. (<year>2010</year>) <volume>31</volume>:<fpage>100</fpage>&#x2013;<lpage>11</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.patrec.2009.09.019</pub-id></citation></ref>
<ref id="ref24"><label>24.</label><citation citation-type="book"><person-group person-group-type="author"><name><surname>Lei</surname><given-names>J</given-names></name> <name><surname>Qiu</surname><given-names>J</given-names></name> <name><surname>Lo</surname><given-names>FPW</given-names></name> <name><surname>Lo</surname><given-names>B</given-names></name></person-group>. <article-title>Assessing individual dietary intake in food sharing scenarios with food and human pose detection</article-title>. In: <person-group person-group-type="editor"><name><surname>Bimbo</surname><given-names>A</given-names><prefix>Del</prefix></name> <name><surname>Cucchiara</surname><given-names>R</given-names></name> <name><surname>Sclaroff</surname><given-names>S</given-names></name> <name><surname>Farinella</surname><given-names>GM</given-names></name> <name><surname>Mei</surname><given-names>T</given-names></name> <name><surname>Bertini</surname><given-names>M</given-names></name> <etal/></person-group>., editors. <source>Pattern recognition ICPR international workshops and challenges</source>. <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>; (<year>2021</year>).</citation></ref>
<ref id="ref25"><label>25.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Neuwald</surname><given-names>NV</given-names></name> <name><surname>Pearce</surname><given-names>AL</given-names></name> <name><surname>Adise</surname><given-names>S</given-names></name> <name><surname>Rolls</surname><given-names>BJ</given-names></name> <name><surname>Keller</surname><given-names>KL</given-names></name></person-group>. <article-title>Switching between foods: a potential behavioral phenotype of hedonic hunger and increased obesity risk in children</article-title>. <source>Physiol Behav</source>. (<year>2023</year>) <volume>270</volume>:<fpage>114312</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.physbeh.2023.114312</pub-id>, PMID: <pub-id pub-id-type="pmid">37543104</pub-id></citation></ref>
<ref id="ref26"><label>26.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Keller</surname><given-names>KL</given-names></name> <name><surname>Pearce</surname><given-names>AL</given-names></name> <name><surname>Fuchs</surname><given-names>B</given-names></name> <name><surname>Rolls</surname><given-names>BJ</given-names></name> <name><surname>Wilson</surname><given-names>SJ</given-names></name> <name><surname>Geier</surname><given-names>CF</given-names></name> <etal/></person-group>. <article-title>PACE: a novel eating behavior phenotype to assess risk for obesity in middle childhood</article-title>. <source>J Nutr</source>. (<year>2024</year>) <volume>154</volume>:<fpage>2176</fpage>&#x2013;<lpage>87</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.tjnut.2024.05.019</pub-id>, PMID: <pub-id pub-id-type="pmid">38795747</pub-id></citation></ref>
<ref id="ref27"><label>27.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bhat</surname><given-names>YR</given-names></name> <name><surname>Rolls</surname><given-names>BJ</given-names></name> <name><surname>Wilson</surname><given-names>SJ</given-names></name> <name><surname>Rose</surname><given-names>E</given-names></name> <name><surname>Geier</surname><given-names>CF</given-names></name> <name><surname>Fuchs</surname><given-names>B</given-names></name> <etal/></person-group>. <article-title>Eating in the absence of hunger is a stable predictor of adiposity gains in middle childhood</article-title>. <source>J Nutr</source>. (<year>2024</year>) <volume>154</volume>:<fpage>3726</fpage>&#x2013;<lpage>39</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.tjnut.2024.10.008</pub-id>, PMID: <pub-id pub-id-type="pmid">39393498</pub-id></citation></ref>
<ref id="ref28"><label>28.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fuchs</surname><given-names>BA</given-names></name> <name><surname>Pearce</surname><given-names>AL</given-names></name> <name><surname>Rolls</surname><given-names>BJ</given-names></name> <name><surname>Wilson</surname><given-names>SJ</given-names></name> <name><surname>Rose</surname><given-names>EJ</given-names></name> <name><surname>Geier</surname><given-names>CF</given-names></name> <etal/></person-group>. <article-title>Does &#x2018;portion size&#x2019; matter? Brain responses to food and non-food cues presented in varying amounts</article-title>. <source>Appetite</source>. (<year>2024</year>) <volume>196</volume>:<fpage>107289</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.appet.2024.107289</pub-id>, PMID: <pub-id pub-id-type="pmid">38423300</pub-id></citation></ref>
<ref id="ref29"><label>29.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fuchs</surname><given-names>BA</given-names></name> <name><surname>Pearce</surname><given-names>AL</given-names></name> <name><surname>Rolls</surname><given-names>BJ</given-names></name> <name><surname>Wilson</surname><given-names>SJ</given-names></name> <name><surname>Rose</surname><given-names>EJ</given-names></name> <name><surname>Geier</surname><given-names>CF</given-names></name> <etal/></person-group>. <article-title>The cerebellar response to visual portion size cues is associated with the portion size effect in children</article-title>. <source>Nutrients</source>. (<year>2024</year>) <volume>16</volume>:<fpage>738</fpage>. doi: <pub-id pub-id-type="doi">10.3390/nu16050738</pub-id>, PMID: <pub-id pub-id-type="pmid">38474866</pub-id></citation></ref>
<ref id="ref30"><label>30.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zuraikat</surname><given-names>FM</given-names></name> <name><surname>Bauman</surname><given-names>JM</given-names></name> <name><surname>Setzenfand</surname><given-names>MN</given-names></name> <name><surname>Arukwe</surname><given-names>DU</given-names></name> <name><surname>Rolls</surname><given-names>BJ</given-names></name> <name><surname>Keller</surname><given-names>KL</given-names></name></person-group>. <article-title>Dimensions of sleep quality are related to objectively measured eating behaviors among children at high familial risk for obesity</article-title>. <source>Obesity</source>. (<year>2023</year>) <volume>31</volume>:<fpage>1216</fpage>&#x2013;<lpage>26</lpage>. doi: <pub-id pub-id-type="doi">10.1002/oby.23754</pub-id>, PMID: <pub-id pub-id-type="pmid">37013867</pub-id></citation></ref>
<ref id="ref31"><label>31.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pearce</surname><given-names>AL</given-names></name> <name><surname>Hallisky</surname><given-names>K</given-names></name> <name><surname>Rolls</surname><given-names>BJ</given-names></name> <name><surname>Wilson</surname><given-names>SJ</given-names></name> <name><surname>Rose</surname><given-names>E</given-names></name> <name><surname>Geier</surname><given-names>CF</given-names></name> <etal/></person-group>. <article-title>Children at high familial risk for obesity show executive functioning deficits prior to development of excess weight status</article-title>. <source>Obesity</source>. (<year>2023</year>) <volume>31</volume>:<fpage>2998</fpage>&#x2013;<lpage>3007</lpage>. doi: <pub-id pub-id-type="doi">10.1002/oby.23892</pub-id>, PMID: <pub-id pub-id-type="pmid">37794530</pub-id></citation></ref>
<ref id="ref32"><label>32.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Neuwald</surname><given-names>NV</given-names></name> <name><surname>Pearce</surname><given-names>AL</given-names></name> <name><surname>Cunningham</surname><given-names>PM</given-names></name> <name><surname>Setzenfand</surname><given-names>MN</given-names></name> <name><surname>Koczwara</surname><given-names>L</given-names></name> <name><surname>Rolls</surname><given-names>BJ</given-names></name> <etal/></person-group>. <article-title>Food switching at a meal is positively associated with change in adiposity among children at high-familial risk for obesity</article-title>. <source>Appetite</source>. (<year>2025</year>) <volume>208</volume>:<fpage>107915</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.appet.2025.107915</pub-id></citation></ref>
<ref id="ref33"><label>33.</label><citation citation-type="book"><person-group person-group-type="author"><name><surname>Van Rossum</surname><given-names>G</given-names></name> <name><surname>Drake</surname><given-names>FL</given-names> <suffix>Jr</suffix></name></person-group>. <source>Python reference manual</source>. <publisher-loc>Amsterdam</publisher-loc>: <publisher-name>Centrum voor Wiskunde en Informatica</publisher-name> (<year>1995</year>).</citation></ref>
<ref id="ref34"><label>34.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vallat</surname><given-names>R</given-names></name></person-group>. <article-title>Pingouin: statistics in python</article-title>. <source>J Open Source Softw</source>. (<year>2018</year>) <volume>3</volume>:<fpage>1026</fpage>. doi: <pub-id pub-id-type="doi">10.21105/joss.01026</pub-id></citation></ref>
<ref id="ref35"><label>35.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Seabold</surname><given-names>S</given-names></name> <name><surname>Perktold</surname><given-names>J</given-names></name></person-group> <article-title>Statsmodels: econometric and statistical modeling with python</article-title>. In: <conf-name>9th Python in Science Conference</conf-name>. <publisher-loc>SciPy</publisher-loc>: <publisher-name>Austin, Texas</publisher-name> (<year>2010</year>).</citation></ref>
<ref id="ref36"><label>36.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pedregosa</surname><given-names>F</given-names></name> <name><surname>Varoquaux</surname><given-names>G</given-names></name> <name><surname>Gramfort</surname><given-names>A</given-names></name> <name><surname>Michel</surname><given-names>V</given-names></name> <name><surname>Thirion</surname><given-names>B</given-names></name> <name><surname>Grisel</surname><given-names>O</given-names></name> <etal/></person-group>. <article-title>Scikit-learn: machine learning in python</article-title>. <source>J Mach Learn Res</source>. (<year>2011</year>) <volume>12</volume>:<fpage>2825</fpage>&#x2013;<lpage>30</lpage>. doi: <pub-id pub-id-type="doi">10.5555/1953048.2078195</pub-id></citation></ref>
<ref id="ref37"><label>37.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Ansel</surname><given-names>J</given-names></name> <name><surname>Yang</surname><given-names>E</given-names></name> <name><surname>He</surname><given-names>H</given-names></name> <name><surname>Gimelshein</surname><given-names>N</given-names></name> <name><surname>Jain</surname><given-names>A</given-names></name> <name><surname>Voznesensky</surname><given-names>M</given-names></name> <etal/></person-group>. <article-title>PyTorch 2: faster machine learning through dynamic python bytecode transformation and graph compilation</article-title>. In:<conf-name>Proceedings of the 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems</conf-name> <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>; (<year>2024</year>).<fpage>929</fpage>&#x2013;<lpage>947</lpage>.</citation></ref>
<ref id="ref38"><label>38.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bradski</surname><given-names>G</given-names></name></person-group>. <article-title>The OpenCV library</article-title>. <source>Dr Dobb J Soft Tools</source>. (<year>2000</year>) <volume>120</volume>:<fpage>122</fpage>&#x2013;<lpage>5</lpage>.</citation></ref>
<ref id="ref39"><label>39.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hunter</surname><given-names>JD</given-names></name></person-group>. <article-title>Matplotlib: a 2D graphics environment</article-title>. <source>Comput Sci Eng</source>. (<year>2007</year>) <volume>9</volume>:<fpage>90</fpage>&#x2013;<lpage>5</lpage>. doi: <pub-id pub-id-type="doi">10.1109/MCSE.2007.55</pub-id></citation></ref>
<ref id="ref40"><label>40.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tomar</surname><given-names>S</given-names></name></person-group>. <article-title>Converting video formats with FFmpeg</article-title>. <source>Linux J</source>. (<year>2006</year>) <volume>2006</volume>:<fpage>10</fpage>. doi: <pub-id pub-id-type="doi">10.5555/1134782.1134792</pub-id></citation></ref>
<ref id="ref41"><label>41.</label><citation citation-type="book"><person-group person-group-type="author"><name><surname>Bengio</surname><given-names>Y</given-names></name></person-group>. <article-title>Practical recommendations for gradient-based training of deep architectures</article-title> In: <person-group person-group-type="editor"><name><surname>Montavon</surname><given-names>G</given-names></name> <name><surname>Orr</surname><given-names>GB</given-names></name> <name><surname>M&#x00FC;ller</surname><given-names>KR</given-names></name></person-group>, editors. <source>Neural Networks: Tricks of the Trade: 2nd Ed</source>. <publisher-loc>Berlin, Heidelberg</publisher-loc>: <publisher-name>Springer Berlin Heidelberg</publisher-name> (<year>2012</year>). <fpage>437</fpage>&#x2013;<lpage>78</lpage>.</citation></ref>
<ref id="ref42"><label>42.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Domingos</surname><given-names>P</given-names></name></person-group>. <article-title>A few useful things to know about machine learning</article-title>. <source>Commun ACM</source>. (<year>2012</year>) <volume>55</volume>:<fpage>78</fpage>&#x2013;<lpage>87</lpage>. doi: <pub-id pub-id-type="doi">10.1145/2347736.2347755</pub-id></citation></ref>
<ref id="ref43"><label>43.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shorten</surname><given-names>C</given-names></name> <name><surname>Khoshgoftaar</surname><given-names>TM</given-names></name></person-group>. <article-title>A survey on image data augmentation for deep learning</article-title>. <source>J Big Data</source>. (<year>2019</year>) <volume>6</volume>, <fpage>1</fpage>&#x2013;<lpage>17</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s40537-019-0197-0</pub-id>, PMID: <pub-id pub-id-type="pmid">40950642</pub-id></citation></ref>
<ref id="ref44"><label>44.</label><citation citation-type="other"><person-group person-group-type="author"><collab id="coll2">HumanSignal/labelImg</collab></person-group>. <source>HumanSignal</source>; (<year>2024</year>) [Available online at: <ext-link xlink:href="https://github.com/HumanSignal/labelImg" ext-link-type="uri">https://github.com/HumanSignal/labelImg</ext-link></citation></ref>
<ref id="ref45"><label>45.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Kaur</surname><given-names>P</given-names></name> <name><surname>Khehra</surname><given-names>BS</given-names></name> <name><surname>Mavi</surname><given-names>BS</given-names></name></person-group>. <article-title>Data augmentation for object detection: a review</article-title> In: <source>2021 IEEE international Midwest symposium on circuits and systems (MWSCAS)</source>. <publisher-loc>Piscataway, New Jersey, USA</publisher-loc>: <publisher-name>Institute of Electrical and Electronics Engineers (IEEE)</publisher-name> (<year>2021</year>).</citation></ref>
<ref id="ref46"><label>46.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Redmon</surname><given-names>J</given-names></name> <name><surname>Divvala</surname><given-names>S</given-names></name> <name><surname>Girshick</surname><given-names>R</given-names></name> <name><surname>Farhadi</surname><given-names>A</given-names></name></person-group>. <article-title>You only look once: unified, real-time object detection</article-title>. In <conf-name>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>. (<year>2016</year>).</citation></ref>
<ref id="ref47"><label>47.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>CY</given-names></name> <name><surname>Bochkovskiy</surname><given-names>A</given-names></name> <name><surname>Liao</surname><given-names>HYM</given-names></name></person-group>. <article-title>YOLOv7: trainable bag-of-freebies sets new state-of-the-art for real-time object detectors</article-title>. In: <conf-name>2023 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name> (<year>2023</year>) <fpage>7464</fpage>&#x2013;<lpage>7475</lpage>. Available online at: <ext-link xlink:href="https://ieeexplore.ieee.org/document/10204762" ext-link-type="uri">https://ieeexplore.ieee.org/document/10204762</ext-link></citation></ref>
<ref id="ref48"><label>48.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hilali</surname><given-names>I</given-names></name> <name><surname>Alfazi</surname><given-names>A</given-names></name> <name><surname>Arfaoui</surname><given-names>N</given-names></name> <name><surname>Ejbali</surname><given-names>R</given-names></name></person-group>. <article-title>Tourist mobility patterns: faster R-CNN versus YOLOv7 for places of interest detection</article-title>. <source>IEEE Access</source>. (<year>2023</year>) <volume>11</volume>:<fpage>130144</fpage>&#x2013;<lpage>54</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ACCESS.2023.3334633</pub-id></citation></ref>
<ref id="ref49"><label>49.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Robbins</surname><given-names>HE</given-names></name></person-group>. <article-title>A stochastic approximation method</article-title>. <source>Ann Math Stat</source>. (<year>1951</year>) <volume>22</volume>:<fpage>400</fpage>&#x2013;<lpage>7</lpage>. doi: <pub-id pub-id-type="doi">10.1214/aoms/1177729586</pub-id></citation></ref>
<ref id="ref50"><label>50.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Ren</surname><given-names>S</given-names></name> <name><surname>He</surname><given-names>K</given-names></name> <name><surname>Girshick</surname><given-names>R</given-names></name> <name><surname>Sun</surname><given-names>J</given-names></name></person-group>. <article-title>Faster R-CNN: towards real-time object detection with region proposal networks</article-title>. In: <person-group person-group-type="author"><name><surname>Cortes</surname><given-names>C</given-names></name> <name><surname>Lawrence</surname><given-names>N</given-names></name> <name><surname>Lee</surname><given-names>D</given-names></name> <name><surname>Sugiyama</surname><given-names>M</given-names></name> <name><surname>Garnett</surname><given-names>R</given-names></name></person-group>, editors. <source>Advances in neural information processing systems [internet]</source>. <publisher-name>Curran Associates, Inc.</publisher-name> (<year>2015</year>). Available online at: <ext-link xlink:href="https://proceedings.neurips.cc/paper_files/paper/2015/file/14bfa6bb14875e45bba028a21ed38046-Paper.pdf" ext-link-type="uri">https://proceedings.neurips.cc/paper_files/paper/2015/file/14bfa6bb14875e45bba028a21ed38046-Paper.pdf</ext-link></citation></ref>
<ref id="ref51"><label>51.</label><citation citation-type="book"><person-group person-group-type="author"><name><surname>Koonce</surname><given-names>B</given-names></name></person-group>. <article-title>ResNet 50</article-title> In: <person-group person-group-type="editor"><name><surname>Koonce</surname><given-names>B</given-names></name></person-group>, editor. <source>Convolutional neural networks with swift for Tensorflow: image recognition and dataset categorization</source>. <publisher-loc>Berkeley, CA</publisher-loc>: <publisher-name>Apress</publisher-name> (<year>2021</year>)</citation></ref>
<ref id="ref52"><label>52.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Lin</surname><given-names>TY</given-names></name> <name><surname>Doll&#x00E1;r</surname><given-names>P</given-names></name> <name><surname>Girshick</surname><given-names>R</given-names></name> <name><surname>He</surname><given-names>K</given-names></name> <name><surname>Hariharan</surname><given-names>B</given-names></name> <name><surname>Belongie</surname><given-names>S</given-names></name></person-group>. <article-title>Feature pyramid networks for object detection</article-title>. In: <conf-name>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name> (<year>2017</year>). p. <fpage>2117</fpage>&#x2013;<lpage>2125</lpage>.</citation></ref>
<ref id="ref53"><label>53.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Geisser</surname><given-names>S</given-names></name></person-group>. <article-title>The predictive sample reuse method with applications</article-title>. <source>J Am Stat Assoc</source>. (<year>1975</year>) <volume>70</volume>:<fpage>320</fpage>&#x2013;<lpage>8</lpage>. doi: <pub-id pub-id-type="doi">10.1080/01621459.1975.10479865</pub-id></citation></ref>
<ref id="ref54"><label>54.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>LaValle</surname><given-names>SM</given-names></name> <name><surname>Branicky</surname><given-names>MS</given-names></name> <name><surname>Lindemann</surname><given-names>SR</given-names></name></person-group>. <article-title>On the relationship between classical grid search and probabilistic roadmaps</article-title>. <source>Int J Robot Res</source>. (<year>2004</year>) <volume>23</volume>:<fpage>673</fpage>&#x2013;<lpage>92</lpage>. doi: <pub-id pub-id-type="doi">10.1177/0278364904045481</pub-id></citation></ref>
<ref id="ref55"><label>55.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Dadjouy</surname><given-names>S</given-names></name> <name><surname>Sajedi</surname><given-names>H</given-names></name></person-group>. <article-title>Gallbladder cancer detection in ultrasound images based on YOLO and faster R-CNN</article-title>. In: <conf-name>2024 10th International Conference on Artificial Intelligence and Robotics (QICAR)</conf-name> (<year>2024</year>); <fpage>227</fpage>&#x2013;<lpage>231</lpage>. Available online at: <ext-link xlink:href="https://ieeexplore.ieee.org/document/10496645" ext-link-type="uri">https://ieeexplore.ieee.org/document/10496645</ext-link></citation></ref>
<ref id="ref56"><label>56.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Henriques</surname><given-names>JF</given-names></name> <name><surname>Caseiro</surname><given-names>R</given-names></name> <name><surname>Martins</surname><given-names>P</given-names></name> <name><surname>Batista</surname><given-names>J</given-names></name></person-group>. <article-title>High-speed tracking with Kernelized correlation filters</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. (<year>2015</year>) <volume>37</volume>:<fpage>583</fpage>&#x2013;<lpage>96</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TPAMI.2014.2345390</pub-id>, PMID: <pub-id pub-id-type="pmid">26353263</pub-id></citation></ref>
<ref id="ref57"><label>57.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Pearce</surname><given-names>AL</given-names></name> <name><surname>Evens</surname><given-names>J</given-names></name> <name><surname>Romano</surname><given-names>O</given-names></name> <name><surname>Keller</surname><given-names>KL</given-names></name></person-group>. <source>Food and brain study - observational coding manual</source>. (<year>2023</year>). Available online at: <ext-link xlink:href="https://zenodo.org/records/8140896" ext-link-type="uri">https://zenodo.org/records/8140896</ext-link></citation></ref>
<ref id="ref58"><label>58.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pearce</surname><given-names>AL</given-names></name> <name><surname>Neuwald</surname><given-names>NV</given-names></name> <name><surname>Evans</surname><given-names>JS</given-names></name> <name><surname>Romano</surname><given-names>O</given-names></name> <name><surname>Rolls</surname><given-names>BJ</given-names></name> <name><surname>Keller</surname><given-names>KL</given-names></name></person-group>. <article-title>Child eating behaviors are consistently linked to intake across meals that vary in portion size</article-title>. <source>Appetite</source>. (<year>2024</year>) <volume>196</volume>:<fpage>107258</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.appet.2024.107258</pub-id>, PMID: <pub-id pub-id-type="pmid">38341036</pub-id></citation></ref>
<ref id="ref59"><label>59.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Koo</surname><given-names>TK</given-names></name> <name><surname>Li</surname><given-names>MY</given-names></name></person-group>. <article-title>A guideline of selecting and reporting Intraclass correlation coefficients for reliability research</article-title>. <source>J Chiropr Med</source>. (<year>2016</year>) <volume>15</volume>:<fpage>155</fpage>&#x2013;<lpage>63</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jcm.2016.02.012</pub-id>, PMID: <pub-id pub-id-type="pmid">27330520</pub-id></citation></ref>
<ref id="ref60"><label>60.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rodenburg</surname><given-names>FJ</given-names></name> <name><surname>Sawada</surname><given-names>Y</given-names></name> <name><surname>Hayashi</surname><given-names>N</given-names></name></person-group>. <article-title>Improving RNN performance by modelling informative missingness with combined indicators</article-title>. <source>Appl Sci</source>. (<year>2019</year>) <volume>9</volume>:<fpage>1623</fpage>. doi: <pub-id pub-id-type="doi">10.3390/app9081623</pub-id></citation></ref>
<ref id="ref61"><label>61.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>He</surname><given-names>J</given-names></name> <name><surname>Zhu</surname><given-names>F</given-names></name></person-group>. <article-title>Single-stage heavy-tailed food classification</article-title>. In: <conf-name>2023 IEEE International Conference on Image Processing (ICIP)</conf-name> (<year>2023</year>); <fpage>1115</fpage>&#x2013;<lpage>1119</lpage>. Available online at: <ext-link xlink:href="https://ieeexplore.ieee.org/document/10222925" ext-link-type="uri">https://ieeexplore.ieee.org/document/10222925</ext-link></citation></ref>
<ref id="ref62"><label>62.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bach</surname><given-names>M</given-names></name> <name><surname>Werner</surname><given-names>A</given-names></name> <name><surname>Palt</surname><given-names>M</given-names></name></person-group>. <article-title>The proposal of undersampling method for learning from imbalanced datasets</article-title>. <source>Proc Comput Sci</source>. (<year>2019</year>) <volume>159</volume>:<fpage>125</fpage>&#x2013;<lpage>34</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.procs.2019.09.167</pub-id></citation></ref>
<ref id="ref63"><label>63.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chawla</surname><given-names>NV</given-names></name> <name><surname>Bowyer</surname><given-names>KW</given-names></name> <name><surname>Hall</surname><given-names>LO</given-names></name> <name><surname>Kegelmeyer</surname><given-names>WP</given-names></name></person-group>. <article-title>SMOTE: synthetic minority over-sampling technique</article-title>. <source>J Artif Intell Res</source>. (<year>2002</year>) <volume>16</volume>:<fpage>321</fpage>&#x2013;<lpage>57</lpage>. doi: <pub-id pub-id-type="doi">10.1613/jair.953</pub-id></citation></ref>
<ref id="ref64"><label>64.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Tan</surname><given-names>M</given-names></name> <name><surname>Le</surname><given-names>Q</given-names></name></person-group>. <article-title>EfficientNet: rethinking model scaling for convolutional neural networks</article-title>. In: <source>Proceedings of the 36th International Conference on Machine Learning</source>. <publisher-name>PMLR</publisher-name>; (<year>2019</year>), p. <fpage>6105</fpage>&#x2013;<lpage>6114</lpage>. Available online at: <ext-link xlink:href="https://proceedings.mlr.press/v97/tan19a.html" ext-link-type="uri">https://proceedings.mlr.press/v97/tan19a.html</ext-link></citation></ref>
<ref id="ref65"><label>65.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hochreiter</surname><given-names>S</given-names></name> <name><surname>Schmidhuber</surname><given-names>J</given-names></name></person-group>. <article-title>Long short-term memory</article-title>. <source>Neural Comput</source>. (<year>1997</year>) <volume>9</volume>:<fpage>1735</fpage>&#x2013;<lpage>80</lpage>. doi: <pub-id pub-id-type="doi">10.1162/neco.1997.9.8.1735</pub-id>, PMID: <pub-id pub-id-type="pmid">9377276</pub-id></citation></ref>
<ref id="ref66"><label>66.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kingma</surname><given-names>DP</given-names></name> <name><surname>Ba</surname><given-names>J</given-names></name></person-group>. <article-title>Adam: a method for stochastic optimization</article-title>. <publisher-loc>San Diego, CA, USA</publisher-loc>: <publisher-name>International Conference on Learning Representations (ICLR)</publisher-name>. (<year>2014</year>).</citation></ref>
<ref id="ref67"><label>67.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pham</surname><given-names>TD</given-names></name></person-group>. <article-title>Time&#x2013;frequency time&#x2013;space LSTM for robust classification of physiological signals</article-title>. <source>Sci Rep</source>. (<year>2021</year>) <volume>11</volume>:<fpage>6936</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-021-86432-7</pub-id>, PMID: <pub-id pub-id-type="pmid">33767352</pub-id></citation></ref>
<ref id="ref68"><label>68.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Song</surname><given-names>Y</given-names></name> <name><surname>Kim</surname><given-names>I</given-names></name></person-group>. <article-title>Spatio-temporal action detection in untrimmed videos by using multimodal features and region proposals</article-title>. <source>Sensors</source>. (<year>2019</year>) <volume>19</volume>:<fpage>1085</fpage>. doi: <pub-id pub-id-type="doi">10.3390/s19051085</pub-id>, PMID: <pub-id pub-id-type="pmid">30832433</pub-id></citation></ref>
<ref id="ref69"><label>69.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Lucas</surname><given-names>BD</given-names></name> <name><surname>Kanade</surname><given-names>T</given-names></name></person-group>. <article-title>An iterative image registration technique with an application to stereo vision</article-title> In: <source>IJCAI&#x2019;81: 7th International Joint Conference on Artificial Intelligence</source> (<year>1981</year>). <fpage>674</fpage>&#x2013;<lpage>9</lpage>.</citation></ref>
<ref id="ref70"><label>70.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Haddad</surname><given-names>J</given-names></name> <name><surname>Ullah</surname><given-names>S</given-names></name> <name><surname>Bell</surname><given-names>L</given-names></name> <name><surname>Leslie</surname><given-names>E</given-names></name> <name><surname>Magarey</surname><given-names>A</given-names></name></person-group>. <article-title>The influence of home and school environments on children&#x2019;s diet and physical activity, and body mass index: a structural equation modelling approach</article-title>. <source>Matern Child Health J</source>. (<year>2018</year>) <volume>22</volume>:<fpage>364</fpage>&#x2013;<lpage>75</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10995-017-2386-9</pub-id>, PMID: <pub-id pub-id-type="pmid">29094228</pub-id></citation></ref>
<ref id="ref71"><label>71.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Strauss</surname><given-names>RS</given-names></name> <name><surname>Knight</surname><given-names>J</given-names></name></person-group>. <article-title>Influence of the home environment on the development of obesity in children</article-title>. <source>Pediatrics</source>. (<year>1999</year>) <volume>103</volume>:<fpage>e85</fpage>. doi: <pub-id pub-id-type="doi">10.1542/peds.103.6.e85</pub-id>, PMID: <pub-id pub-id-type="pmid">10353982</pub-id></citation></ref>
<ref id="ref72"><label>72.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gerards</surname><given-names>SMPL</given-names></name> <name><surname>Kremers</surname><given-names>SPJ</given-names></name></person-group>. <article-title>The role of food parenting skills and the home food environment in children&#x2019;s weight gain and obesity</article-title>. <source>Curr Obes Rep</source>. (<year>2015</year>) <volume>4</volume>:<fpage>30</fpage>&#x2013;<lpage>6</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s13679-015-0139-x</pub-id>, PMID: <pub-id pub-id-type="pmid">25741454</pub-id></citation></ref>
<ref id="ref73"><label>73.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Flake</surname><given-names>JK</given-names></name> <name><surname>Fried</surname><given-names>EI</given-names></name></person-group>. <article-title>Measurement schmeasurement: questionable measurement practices and how to avoid them</article-title>. <source>Adv Methods Pract Psychol Sci</source>. (<year>2020</year>) <volume>3</volume>:<fpage>456</fpage>&#x2013;<lpage>65</lpage>. doi: <pub-id pub-id-type="doi">10.1177/2515245920952393</pub-id></citation></ref>
<ref id="ref74"><label>74.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Brownell</surname><given-names>CA</given-names></name> <name><surname>Lemerise</surname><given-names>EA</given-names></name> <name><surname>Pelphrey</surname><given-names>KA</given-names></name> <name><surname>Roisman</surname><given-names>GI</given-names></name></person-group>. <article-title>Measuring socioemotional development</article-title> In: <source>Handbook of child psychology and developmental science: socioemotional processes</source> (<year>2015</year>)</citation></ref>
<ref id="ref75"><label>75.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dong</surname><given-names>Y</given-names></name> <name><surname>Hoover</surname><given-names>A</given-names></name> <name><surname>Scisco</surname><given-names>J</given-names></name> <name><surname>Muth</surname><given-names>E</given-names></name></person-group>. <article-title>A new method for measuring meal intake in humans via automated wrist motion tracking</article-title>. <source>Appl Psychophysiol Biofeedback</source>. (<year>2012</year>) <volume>37</volume>:<fpage>205</fpage>&#x2013;<lpage>15</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10484-012-9194-1</pub-id>, PMID: <pub-id pub-id-type="pmid">22488204</pub-id></citation></ref>
<ref id="ref76"><label>76.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>R</given-names></name> <name><surname>Amft</surname><given-names>O</given-names></name></person-group>. <article-title>Monitoring chewing and eating in free-living using smart eyeglasses</article-title>. <source>IEEE J Biomed Health Inform</source>. (<year>2018</year>) <volume>22</volume>:<fpage>23</fpage>&#x2013;<lpage>32</lpage>. doi: <pub-id pub-id-type="doi">10.1109/JBHI.2017.2698523</pub-id>, PMID: <pub-id pub-id-type="pmid">28463209</pub-id></citation></ref>
<ref id="ref77"><label>77.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Das</surname><given-names>SK</given-names></name> <name><surname>Miki</surname><given-names>AJ</given-names></name> <name><surname>Blanchard</surname><given-names>CM</given-names></name> <name><surname>Sazonov</surname><given-names>E</given-names></name> <name><surname>Gilhooly</surname><given-names>CH</given-names></name> <name><surname>Dey</surname><given-names>S</given-names></name> <etal/></person-group>. <article-title>Perspective: opportunities and challenges of technology tools in dietary and activity assessment: bridging stakeholder viewpoints</article-title>. <source>Adv Nutr</source>. (<year>2022</year>) <volume>13</volume>:<fpage>1</fpage>&#x2013;<lpage>15</lpage>. doi: <pub-id pub-id-type="doi">10.1093/advances/nmab103</pub-id>, PMID: <pub-id pub-id-type="pmid">34545392</pub-id></citation></ref>
<ref id="ref78"><label>78.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jegham</surname><given-names>I</given-names></name> <name><surname>Khalifa</surname><given-names>AB</given-names></name> <name><surname>Alouani</surname><given-names>I</given-names></name> <name><surname>Mahjoub</surname><given-names>MA</given-names></name></person-group>. <article-title>Vision-based human action recognition: an overview and real world challenges</article-title>. <source>Foren Sci Int</source>. (<year>2020</year>) <volume>32</volume>:<fpage>200901</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.fsidi.2019.200901</pub-id>, PMID: <pub-id pub-id-type="pmid">40957741</pub-id></citation></ref>
<ref id="ref79"><label>79.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Li</surname><given-names>W</given-names></name> <name><surname>Zhang</surname><given-names>Z</given-names></name> <name><surname>Liu</surname><given-names>Z</given-names></name></person-group>. <article-title>Action recognition based on a bag of 3D points</article-title> In: <source>IEEE Computer Society Conference on Computer Vision and Pattern Recognition - Workshops</source> (<year>2010</year>)</citation></ref>
<ref id="ref80"><label>80.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>YC</given-names></name> <name><surname>Onthoni</surname><given-names>DD</given-names></name> <name><surname>Mohapatra</surname><given-names>S</given-names></name> <name><surname>Irianti</surname><given-names>D</given-names></name> <name><surname>Sahoo</surname><given-names>PK</given-names></name></person-group>. <article-title>Deep-learning-assisted multi-dish food recognition application for dietary intake reporting</article-title>. <source>Electronics</source>. (<year>2022</year>) <volume>11</volume>:<fpage>1626</fpage>. doi: <pub-id pub-id-type="doi">10.3390/electronics11101626</pub-id></citation></ref>
<ref id="ref81"><label>81.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Sharma</surname><given-names>A</given-names></name> <name><surname>Czarnecki</surname><given-names>C</given-names></name> <name><surname>Chen</surname><given-names>Y</given-names></name> <name><surname>Xi</surname><given-names>P</given-names></name> <name><surname>Xu</surname><given-names>L</given-names></name> <name><surname>Wong</surname><given-names>A</given-names></name></person-group>. <article-title>How much you ate? Food portion estimation on spoons</article-title>. In: <conf-name>2024 IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)</conf-name> (<year>2024</year>), <fpage>3761</fpage>&#x2013;<lpage>3770</lpage>. Available online at: <ext-link xlink:href="https://ieeexplore.ieee.org/document/10678548" ext-link-type="uri">https://ieeexplore.ieee.org/document/10678548</ext-link></citation></ref>
</ref-list>
<glossary>
<def-list>
<title>Glossary</title>
<def-item>
<term>AI</term>
<def>
<p>Artificial Intelligence</p>
</def>
</def-item>
<def-item>
<term>API</term>
<def>
<p>Application Programming Interface</p>
</def>
</def-item>
<def-item>
<term>CNN</term>
<def>
<p>Convolutional Neural Network</p>
</def>
</def-item>
<def-item>
<term>CPU</term>
<def>
<p>Central Processing Unit</p>
</def>
</def-item>
<def-item>
<term>COVID-19</term>
<def>
<p>Coronavirus Disease 2019</p>
</def>
</def-item>
<def-item>
<term>F1</term>
<def>
<p>F1 Score (harmonic mean of precision and recall)</p>
</def>
</def-item>
<def-item>
<term>FN</term>
<def>
<p>False Negative</p>
</def>
</def-item>
<def-item>
<term>FP</term>
<def>
<p>False Positive</p>
</def>
</def-item>
<def-item>
<term>FPS</term>
<def>
<p>Frames Per Second</p>
</def>
</def-item>
<def-item>
<term>FPN</term>
<def>
<p>Feature Pyramid Network</p>
</def>
</def-item>
<def-item>
<term>GPU</term>
<def>
<p>Graphics Processing Unit</p>
</def>
</def-item>
<def-item>
<term>ICC</term>
<def>
<p>Intraclass Correlation Coefficient</p>
<p>IoU</p>
<p>Intersection of Union</p>
</def>
</def-item>
<def-item>
<term>KCF</term>
<def>
<p>Kernelized Correlation Filter</p>
</def>
</def-item>
<def-item>
<term>LSTM</term>
<def>
<p>Long Short-Term Memory</p>
</def>
</def-item>
<def-item>
<term>MP4</term>
<def>
<p>MPEG-4 Video Format</p>
</def>
</def-item>
<def-item>
<term>R-CNN</term>
<def>
<p>Regional-Convolutional Neural Network</p>
</def>
</def-item>
<def-item>
<term>RAM</term>
<def>
<p>Random Access Memory</p>
</def>
</def-item>
<def-item>
<term>ResNet</term>
<def>
<p>Residual Network</p>
</def>
</def-item>
<def-item>
<term>RMSE</term>
<def>
<p>Root Mean Square Error</p>
</def>
</def-item>
<def-item>
<term>RMSE%</term>
<def>
<p>Percentage Root Mean Square Error</p>
</def>
</def-item>
<def-item>
<term>RNN</term>
<def>
<p>Recurrent Neural Network</p>
</def>
</def-item>
<def-item>
<term>SGD</term>
<def>
<p>Stochastic Gradient Descent</p>
</def>
</def-item>
<def-item>
<term>SMOTE</term>
<def>
<p>Synthetic Minority Over-sampling Technique</p>
</def>
</def-item>
<def-item>
<term>TN</term>
<def>
<p>True Negative</p>
</def>
</def-item>
<def-item>
<term>TP</term>
<def>
<p>True Positive</p>
</def>
</def-item>
<def-item>
<term>YOLOv7</term>
<def>
<p>You Only Look Once, version 7</p>
</def>
</def-item>
</def-list>
</glossary>
</back>
</article>