<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article article-type="research-article" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Educ.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Education</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Educ.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2504-284X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/feduc.2025.1639273</article-id><article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading"><subject>Original Research</subject></subj-group>
</article-categories>
<title-group>
<article-title>Fixations, regressions, and results: eye-tracking metrics as real-time signals of cognitive engagement in flipped-class quizzes</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Yuanyuan</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3234773"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Xie</surname>
<given-names>Nina</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3065215"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Liu</surname>
<given-names>Yujun</given-names>
</name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3060752"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>Department of Management and Strategy, Hong Kong Metropolitan University</institution>, <city>Ho Man Tin</city>, <country>Hong Kong SAR, China</country></aff>
<aff id="aff2"><label>2</label><institution>Department of Management, Lingnan University</institution>, <city>Tuen Mun</city>, <country>Hong Kong SAR, China</country></aff>
<aff id="aff3"><label>3</label><institution>Faculty of Businesingnan University</institution>, <city>Tuen Mun</city>, <country>Hong Kong SAR, China</country></aff>
<author-notes><corresp id="c001"><label>&#x002A;</label>Correspondence: Nina Xie, <email xlink:href="mailto:ninaxie@ln.edu.hk">ninaxie@ln.edu.hk</email>; Yujun Liu, <email xlink:href="mailto:yujunliu@ln.hk">yujunliu@ln.hk</email></corresp></author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-11-06">
<day>06</day>
<month>11</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>10</volume>
<elocation-id>1639273</elocation-id>
<history>
<date date-type="received">
<day>01</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Wang, Xie and Liu.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wang, Xie and Liu</copyright-holder>
<license><ali:license_ref start_date="2025-11-06">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Educators need real time evidence of how students process pre class quiz items in flipped courses, not just whether answers are right or wrong. We examined whether two classroom feasible eye tracking metrics&#x2014;fixation intensity (total dwell time) and regression rate (proportion of backward saccades)&#x2014;provide interpretable, item level signals of cognitive engagement once surface text features are taken into account.</p>
</sec>
<sec>
<title>Methods</title>
<p>Thirty four undergraduates completed 320 analysable attempts on 55 multiple choice items coded by Bloom&#x2019;s taxonomy while a 60&#x202F;Hz tracker recorded gaze. Crossed mixed effects models included a covariate for each item&#x2019;s total word count. A logistic mixed model tested whether fixation intensity and regression rate predicted correctness beyond Bloom level, gender, and length. After each block, students reported perceived mental effort to compare subjective and gaze based indicators.</p>
</sec>
<sec>
<title>Results</title>
<p>After controlling for total word count, Bloom category did not uniquely predict fixation intensity or regression rate, suggesting that previously observed demand patterns largely reflected text length. In the accuracy model, fixation intensity showed a small, positive association with being correct, whereas regression rate showed a small, negative association.</p>
</sec>
<sec>
<title>Discussion</title>
<p>In authentic flipped class quizzes, fixation intensity and regression rate can serve as complementary, real time indicators of engagement, but only when item length and layout are standardised or statistically modelled. Claims about differences across Bloom levels should be made cautiously. We outline design guidance for future item banks&#x2014;length matched stems, fixed numbers of options, and pre registered word count covariates&#x2014;to enable firmer inferences and practical classroom diagnostics.</p>
</sec>
</abstract>
<kwd-group>
<kwd>eye tracking</kwd>
<kwd>fixation intensity</kwd>
<kwd>regression rate</kwd>
<kwd>flipped classroom</kwd>
<kwd>process data</kwd>
<kwd>adaptive assessment</kwd>
</kwd-group><funding-group><funding-statement>The author(s) declare that financial support was received for the research and/or publication of this article. We gratefully acknowledge the support of the Teaching Development Grant from Lingnan University for the project titled &#x201C;Navigating the Digital Learning Landscape with Eye-tracking: The Confluence of Flipped Classrooms and Experiential Education under OBATL.&#x201D;</funding-statement></funding-group>
<counts>
<fig-count count="0"/>
<table-count count="3"/>
<equation-count count="2"/>
<ref-count count="50"/>
<page-count count="12"/>
<word-count count="10349"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Assessment, Testing and Applied Measurement</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>Digital technologies have broadened when and where students learn, yet instructors still have limited access to real-time engagement evidence during pre-class work. In flipped courses, weekly formative multiple-choice (MC) quizzes help surface misconceptions before class, but right&#x2013;wrong scores alone miss how items are processed. We examine whether economical eye-tracking can supply response-process evidence in this formative context. The flipped classroom is an educational methodology in which traditional lecture content is delivered outside of class, typically via pre-recorded videos or readings, while in-class time is dedicated to active, collaborative, and higher-order learning tasks (<xref ref-type="bibr" rid="ref49">Zainuddin and Halili, 2016</xref>). It is not a prescriptive model but rather a flexible approach that can be adapted with various instructional activities, such as problem-solving, discussions, simulations, and quizzes. While our study uses quizzes as the primary pre-class activity, these are just one of many possible modalities in flipped pedagogy. Flipped designs seek to enhance learning and increase motivation by shifting exposition to pre-class assignments and allocating contact hours to higher-order tasks (<xref ref-type="bibr" rid="ref2">Ak&#x00E7;ay&#x0131;r and Ak&#x00E7;ay&#x0131;r, 2018</xref>; <xref ref-type="bibr" rid="ref22">Hew et al., 2021</xref>). Meta-analyses validate these advantages but caution that the benefits are most pronounced when educators can identify misconceptions promptly and modify classroom activities accordingly (<xref ref-type="bibr" rid="ref34">Lundin et al., 2018</xref>). Conventional right&#x2013;wrong quiz scores inform instructors about students&#x2019; correct responses but fail to capture their cognitive processing of items&#x2014;an oversight that eye-tracking data can overcome by providing detailed insights into visual attention and engagement patterns, essential for delivering personalized feedback and adaptive learning sequences (<xref ref-type="bibr" rid="ref45">Tehranchi et al., 2020</xref>).</p>
<p>While eye-tracking can index on-task processing in real time, routine classroom use hinges on pragmatic constraints&#x2014;calibration, cost, and privacy. We therefore frame eye tracking here as a classroom-compatible research instrument whose outputs can inform design rules and, in the longer run, lightweight diagnostics. Following the eye&#x2013;mind and immediacy assumptions (<xref ref-type="bibr" rid="ref27">Just and Carpenter, 1976</xref>), fixations can reflect ongoing processing at the fixated location, and regressions can mark re-inspection; however, these links are context-dependent and sensitive to text features (<xref ref-type="bibr" rid="ref24">Hy&#x00F6;n&#x00E4;, 2010</xref>). We therefore use the term visual effort to denote gaze-based indicators&#x2014;Fixation Intensity (FI) and Regression Rate (RR)&#x2014;and reserve subjective mental effort for self-reports. Across domains, fixation-based metrics index intrinsic and extraneous cognitive load (<xref ref-type="bibr" rid="ref30">Lai et al., 2013</xref>). These measures remain reliable at 60&#x202F;Hz on affordable trackers (<xref ref-type="bibr" rid="ref10">Beatty and Lucero-Wagoner, 2000</xref>) and can flag learners needing support before errors surface (<xref ref-type="bibr" rid="ref3">Alemdag and Cagiltay, 2018</xref>). Yet few studies align gaze behavior with Bloom-coded demand or test whether item-specific effort predicts immediate success, leaving the effort&#x2013;complexity link unsettled. We operationalize visual effort as fixation intensity and regression rate (i.e., effort inferred from eye movements).</p>
<p>In our setting, weekly pre-class multiple-choice quizzes were strictly formative&#x2014;informing instruction and self-regulation rather than grades&#x2014;within a flipped design that assigns lower-order processes to preparation and higher-order reasoning to class time (<xref ref-type="bibr" rid="ref29">Krathwohl, 2002</xref>; <xref ref-type="bibr" rid="ref49">Zainuddin and Halili, 2016</xref>). Prior findings on Bloom-aligned gaze demand are mixed. A key design risk is surface text: higher-order items are often more concise, so raw dwell time may confound conceptual demand with total word count across stem and options. We therefore model Total Word Count in all primary analyses and treat Bloom effects as interpretable only when surface features are standardized or statistically controlled (<xref ref-type="bibr" rid="ref38">&#x00D6;zdemir and Tosun, 2025</xref>), while others observe no significant difference when controlling for stem length. These inconsistencies underscore a design quandary: higher-order items tend to be more concise in terms of text length, potentially conflating conceptual complexity with the amount of reading required for each item. This study examines whether visual effort correlates with conceptual difficulty in real classroom settings by categorizing remember/understand objects as low demand and apply/analyze item as high demand.</p>
<p>Grounded in engagement theory, we target cognitive engagement&#x2014;the effort devoted to comprehension&#x2014;which relates most strongly to achievement (<xref ref-type="bibr" rid="ref16">Fredricks et al., 2019</xref>). We treat Fixation Intensity (longer dwell times) and Regression Rate (backward saccades/re-reading) as behavioral traces of that effort (<xref ref-type="bibr" rid="ref46">van Gog and Jarodzka, 2013</xref>). Because subjective mental-effort ratings often diverge from objective process measures (<xref ref-type="bibr" rid="ref40">Paas and Van Merri&#x00EB;nboer, 1994</xref>), we examine their correspondence: convergence supports construct validity, whereas systematic gaps clarify what each metric captures under cognitive load theory and how to interpret them for classroom analytics. We distinguish (a) cognitive load as a theoretical construct; (b) gaze-based effort as objective, procss-level indicators derived from Fixation Intensity (FI) and Regression Rate (RR); and (c) self-reported mental effort as a block-level subjective rating. FI and RR are interpreted as load-sensitive rather than direct measures of intrinsic or extraneous load; their validity depends on task control (e.g., text length) and statistical adjustment (here, Total word count included as a covariate). In addition, women generally exhibit slightly longer fixations and more regressions, whereas men tend to scan faster at comparable accuracy. While these effects are overshadowed by skill disparities, incorporating gender as a covariate facilitates an exploratory examination.</p>
<p>Accordingly, we examine whether higher-order items elicit more visual effort when controlling for TotalWC; test whether Fixation Intensity (FI) and Regression Rate (RR) predict item-level success over and above Bloom level, gender, and TotalWC; quantify the alignment between gaze-based effort and block-level subjective effort; and explore baseline gender differences in speed/strategy.</p>
<p>Classroom eye-tracking on multiple-choice tasks remains largely descriptive. In a systematic review of 17 studies, <xref ref-type="bibr" rid="ref41">Paskovske and Klizien&#x0117; (2024)</xref> note that most work still correlates mean dwell time with achievement; reviews in STEM education echo the need for multilevel modeling to separate student from item variance. We address this by using crossed mixed-effects models that nest attempts within students and items (<xref ref-type="bibr" rid="ref9">Barr et al., 2013</xref>), allowing us to test whether effort on a specific item predicts success on that item&#x2014;rather than only unit-level aggregates. To our knowledge, this is among the first Bloom-aligned, mixed-effects analyses of gaze in routine flipped-class quizzes. Recent STEM work shows gaze patterns can reveal strategies and misconceptions, not just accuracy (<xref ref-type="bibr" rid="ref11">Becker et al., 2023</xref>; <xref ref-type="bibr" rid="ref12">Becker et al., 2022</xref>; <xref ref-type="bibr" rid="ref15">Fehlinger et al., 2025</xref>).</p>
<p>We embedded economical eye-tracking in weekly flipped-quiz sessions: undergraduates answered Bloom-coded items while FI and RR were logged. We model (a) whether higher-order demand increases visual effort controlling TotalWC, (b) whether FI/RR add predictive value for item correctness beyond Bloom, gender, and TotalWC, (c) correspondence between gaze-based and subjective effort, and (d) baseline gender differences in speed/strategy. By pinpointing when FI and RR are valid and actionable signals, the study supplies instructors&#x2014;and adaptive algorithms&#x2014;with item-level evidence of visual effort vs. confusion, enabling targeted support without displacing in-class collaborative learning central to flipped pedagogy.</p>
<p>Advances in learning analytics make it feasible to pair real-time gaze data with AI to trigger just-in-time scaffolds (<xref ref-type="bibr" rid="ref14">D&#x2019;Mello et al., 2017</xref>). We treat AI-assisted use as a future pathway that depends on matched-length item banks, clear data-use policies, and replication across classes. In the present paper, eye tracking serves primarily to derive design guidance and to benchmark lighter proxies for eventual classroom diagnostics. At-scale use, however, hinges on affordable hardware, validated item banks, transparent data policies, and LMS integration.</p>
</sec>
<sec id="sec2">
<label>2</label>
<title>Literature review</title>
<sec id="sec3">
<label>2.1</label>
<title>Transitioning from flipped classroom to process analytics</title>
<p>The flipped classroom is an educational methodology rather than a fixed model; early work emphasized affective benefits (e.g., satisfaction, attendance) and used online quizzes primarily for pre-class compliance checks (<xref ref-type="bibr" rid="ref2">Ak&#x00E7;ay&#x0131;r and Ak&#x00E7;ay&#x0131;r, 2018</xref>). Meta-analyses now show medium achievement gains, conditional on tight alignment between pre-study work and in-class higher-order tasks (<xref ref-type="bibr" rid="ref22">Hew et al., 2021</xref>; <xref ref-type="bibr" rid="ref34">Lundin et al., 2018</xref>). Significantly, most outcome studies continue to depend on binary accuracy or final unit grades. Such product-centric metrics obscure how answers were produced. Neutrosophic cognitive diagnosis extends classical CDMs by representing knowledge, misconception, and uncertainty on the same scale, yielding richer profiles for adaptation (<xref ref-type="bibr" rid="ref35">Ma H. et al., 2023</xref>). Unlike conventional models that classify student responses into simply correct or incorrect, neutrosophic cognitive diagnosis captures the degree of uncertainty in students&#x2019; knowledge states, thereby offering a more nuanced and diagnostically rich profile for adaptive interventions. This approach aligns with the broader movement toward fine-grained, process-aware analytics in education. Likewise, models predicting cognitive presence in MOOCs achieve 92.5% accuracy by analyzing discussion traces instead of relying on sparse clickstreams alone (<xref ref-type="bibr" rid="ref31">Lee et al., 2022</xref>), while <xref ref-type="bibr" rid="ref19">Gijsen et al. (2024)</xref> demonstrate that combining clickstream data with think-aloud protocols in video-based learning uncovers deeper processing patterns that binary metrics miss. Intelligent Tutoring System (ITS) diagnostic engines refer to automated systems that analyze learner interactions (e.g., responses, clickstreams, or gaze data) to infer knowledge states, misconceptions, or areas of struggle and then adapt instruction accordingly. ITS aims to provide timely, personalized feedback but are limited by the granularity and specificity of the available process data (<xref ref-type="bibr" rid="ref20">Graesser et al., 2012</xref>). Its that depend solely on clickstreams or delayed self-reports falter in detecting misconceptions promptly and cannot direct limited instructional time to areas of greatest need.</p>
<p>Eye tracking is especially complementary to flipped education, as pre-class activities occur on-screen, making the integration of a low-cost tracker minimally burdensome. The emergence of AI-driven learning analytics has further raised the possibility of real-time, gaze-informed adaptations. Such systems can leverage eye-movement patterns&#x2014;such as prolonged fixations or frequent regressions&#x2014;to infer moments of struggle or disengagement, triggering tailored scaffolds before errors manifest (<xref ref-type="bibr" rid="ref3">Alemdag and Cagiltay, 2018</xref>). However, transforming these research prototypes into robust, classroom-ready tools remains a non-trivial engineering and validation challenge. Real-time gaze traces reveal the specific components of a question stem that capture immediate attention, the systematic comparison of options, and the moments when a learner experiences a &#x201C;stall&#x201D; on a challenging segment. Pilot implementations within learning management systems have demonstrated that identifying the pattern &#x201C;low fixation + high error&#x201D; enables instructors to provide follow-up explanations more effectively (<xref ref-type="bibr" rid="ref3">Alemdag and Cagiltay, 2018</xref>). Nonetheless, these proof-of-concept studies seldom correlate gaze behavior with Bloom-coded cognitive demand, nor do they associate process data with immediate in-class performance&#x2014;two deficiencies that constrain both theoretical understanding and practical application.</p>
<p>Rectifying these inadequacies provides two advantages. Initially, trial-level gaze evidence enhances the response-process dimension of validity highlighted&#x2014;but infrequently substantiated&#x2014;in the Standards for Educational and Psychological Testing (<xref ref-type="bibr" rid="ref4">American Educational Research Association, American Psychological Association, &#x0026; National Council on Measurement in Education, 2014</xref>) and in contemporary digital assessment frameworks. Secondly, affluent process signals provide actionable inputs for AI-driven personalisation frameworks: recommendation systems can activate timely scaffolds, and predictive dashboards can identify students in need of human intervention. This study incorporates eye tracking into standard flipped-class quizzes and correlates gaze patterns with Bloom&#x2019;s taxonomy, accuracy, and self-reported effort, advancing the development of a process-aware, AI-enhanced future.</p>
<p>The rise of process data analytics in education&#x2014;fueled by advances in educational technology and artificial intelligence&#x2014;now allows researchers and instructors to move beyond snapshots of achievement (scores, grades) to continuous, longitudinal analysis of learning behaviors (<xref ref-type="bibr" rid="ref14">D&#x2019;Mello et al., 2017</xref>). For example, AI-driven analytics can detect subtle patterns in eye movements, keystrokes, or physiological signals that precede errors or signal conceptual breakthroughs, enabling just-in-time scaffolding or adaptive task sequencing. However, the reliable implementation of such systems requires robust evidence for the validity and generalizability of process-based indicators, a focus of the present study.</p>
</sec>
<sec id="sec4">
<label>2.2</label>
<title>Cognitive-engagement framework</title>
<p>Student engagement is widely conceptualized as a triad of behavioral, emotional, and cognitive dimensions (<xref ref-type="bibr" rid="ref16">Fredricks et al., 2019</xref>). Cognitive engagement&#x2014;the strategic and meta visual effort learners dedicate to comprehending and mastering content&#x2014;exhibits the most consistent correlation with long-term success, surpassing both time-on-task and emotional enthusiasm (<xref ref-type="bibr" rid="ref32">Lei et al., 2018</xref>). Cognitive engagement also refers to the depth of student involvement in learning tasks; mental effort denotes the subjective experience of cognitive exertion. Time-on-task has long been recognized as a robust indicator of engagement and learning success. In digital environments, efficient gaze allocation&#x2014;such as longer fixations and fewer regressions&#x2014;reflects focused visual effort, whereas fragmented or hesitant reading patterns may signal confusion or disengagement (<xref ref-type="bibr" rid="ref44">Spichtig et al., 2017</xref>). Tracking these metrics enables a more nuanced understanding of how students allocate effort during formative assessments, beyond simple accuracy scoring. Flipped pedagogy is designed to enhance cognitive engagement: learners self-regulate during content preview and thereafter utilize contact hours to study, apply, or evaluate topics (<xref ref-type="bibr" rid="ref49">Zainuddin and Halili, 2016</xref>). In this paper, we use &#x201C;gaze-based effort&#x201D; to denote FI and RR (objective, process-level), and &#x201C;self-reported mental effort&#x201D; to denote the block-level subjective ratings; &#x201C;cognitive load&#x201D; is treated as the broader theoretical construct.</p>
<p>Student engagement in flipped classrooms is often measured using retrospective self-report instruments, such as the Motivated Strategies for Learning Questionnaire (MSLQ; <xref ref-type="bibr" rid="ref42">Pintrich, 2004</xref>), the Flipped Learning Student Engagement Scale (FLSES; <xref ref-type="bibr" rid="ref48">Yan and Lv, 2023</xref>), or single-item post-activity questionnaires administered after the learning experience. However, these retrospective measures are vulnerable to recall bias and social desirability effects, which may lead students to overestimate or misremember their actual engagement (<xref ref-type="bibr" rid="ref17">Fuller et al., 2018</xref>). In contrast, real-time process data&#x2014;such as gaze patterns or interaction logs&#x2014;offer a more immediate and objective window into cognitive engagement.</p>
<p>Self-report scales like the Student Engagement Instrument (SEI: <xref ref-type="bibr" rid="ref5">Appleton et al., 2006</xref>) and Psychological State of Cognitive Presence Cognitive Engagement Scale (PSCP: <xref ref-type="bibr" rid="ref39">&#x00D6;zek and Fer, 2025</xref>) have demonstrated strong validity and reliability in capturing sub-factors such as cognitive attention and effort. However, such retrospective measures correlate only modestly with process data (<xref ref-type="bibr" rid="ref21">Han, 2023</xref>).</p>
<p>To overcome these limitations, researchers advocate integrating established frameworks and automated methods. The ICAP Model (Interactive, Constructive, Active, Passive; <xref ref-type="bibr" rid="ref13">Chi and Wylie, 2014</xref>) increases the reliability of engagement measurement by providing a theoretically grounded framework that distinguishes qualitatively different levels of cognitive involvement. Higher ICAP modes (Interactive, Constructive) are consistently associated with deeper learning outcomes, supporting the use of process data to infer engagement quality rather than mere time-on-task (<xref ref-type="bibr" rid="ref47">Xu et al., 2023</xref>), distinguishing quality beyond time-on-task. Advanced, Analytic, Automated (AAA) approaches further enrich measurement by inferring cognitive engagement from real-time behavioral and physiological signals&#x2014;such as facial expressions, eye tracking, and clickstream data&#x2014;offering fine-grained insights that self-reports miss (<xref ref-type="bibr" rid="ref14">D&#x2019;Mello et al., 2017</xref>). While these automated techniques require robust infrastructure and raise privacy considerations, their combination with self-report instruments and observational checklists yields the most comprehensive assessment of cognitive engagement in flipped classrooms (<xref ref-type="bibr" rid="ref7">Barlow and Brown, 2020</xref>; <xref ref-type="bibr" rid="ref33">Liu et al., 2022</xref>).</p>
</sec>
<sec id="sec5">
<label>2.3</label>
<title>Eye-movement metrics as cognitive load proxies</title>
<p>Eye-tracking may serve as indirect, load-sensitive indicators of processing effort under specified task conditions (e.g., text length and layout controlled), rather than direct measures of intrinsic or extraneous load (<xref ref-type="bibr" rid="ref44">Spichtig et al., 2017</xref>; <xref ref-type="bibr" rid="ref25">Inhoff et al., 2019</xref>; <xref ref-type="bibr" rid="ref30">Lai et al., 2013</xref>). However, these metrics should not be interpreted as direct or unambiguous measures of specific cognitive load components (e.g., intrinsic, extraneous), as fixation duration and regressions are influenced by multiple factors, including reading skill, task familiarity, and item complexity (<xref ref-type="bibr" rid="ref12">Becker et al., 2022</xref>).</p>
<p>Fixation intensity and regression rate may serve as indirect, behaviorally observable indicators of visual effort under specific conditions, particularly when text complexity and task demands are carefully controlled. Accordingly, in our study FI/RR are interpreted as load-sensitive only after statistically controlling for Total word count (stem+options) at the attempt level and reporting item-level checks.</p>
<p>Fixation-based metrics provide a sensitive window on processing effort. Longer fixations and more regressions typically signal greater cognitive demand or lower reading efficiency; regressions, in particular, index comprehension difficulty and, in modeling studies, help predict individual differences in reading comprehension (<xref ref-type="bibr" rid="ref25">Inhoff et al., 2019</xref>; <xref ref-type="bibr" rid="ref28">Kim et al., 2022</xref>; <xref ref-type="bibr" rid="ref37">Man and Harring, 2019</xref>). Proficiency contrasts are robust: efficient readers show shorter/ fewer fixations and fewer regressions, whereas struggling readers maintain elevated levels into high school (<xref ref-type="bibr" rid="ref44">Spichtig et al., 2017</xref>). Beyond description, fixation counts and regression patterns have been used to estimate item-specific attention and difficulty, highlighting how process data discriminate effortless from effortful reading in ways outcome scores cannot (<xref ref-type="bibr" rid="ref37">Man and Harring, 2019</xref>).</p>
<p>For classroom use, practicality matters. Pupillometry can index effort but typically requires &#x2265;120&#x202F;Hz to separate effort-related changes from light reflexes (<xref ref-type="bibr" rid="ref10">Beatty and Lucero-Wagoner, 2000</xref>). By contrast, fixation intensity (FI) and regression rate (RR) are stable at 60&#x202F;Hz, the sampling rate of economical trackers (<xref ref-type="bibr" rid="ref46">van Gog and Jarodzka, 2013</xref>), so we focus on these signals here. FI reflects prolonged, high-resolution processing of stems and options&#x2014;sometimes accompanying conceptual reorganization in expository text (<xref ref-type="bibr" rid="ref36">Ma X. et al., 2023</xref>). RR captures strategic re-inspection when learners confront contradictions across representations (<xref ref-type="bibr" rid="ref1">Abt et al., 2024</xref>). Although pupil diameter was recorded, it was not analyzed due to expected noise at 60&#x202F;Hz. Embedding FI and RR in flipped-course quizzes yields time-stamped evidence of engagement that self-reports and clickstreams miss, enabling instructors&#x2014;and adaptive algorithms&#x2014;to identify confusion and deliver targeted, just-in-time support.</p>
</sec>
<sec id="sec6">
<label>2.4</label>
<title>Bloom demand and item characteristics</title>
<p>Bloom&#x2019;s new taxonomy categorizes cognitive activities in a continuum ranging from remembering to comprehending, applying, analysing, and ultimately producing (<xref ref-type="bibr" rid="ref29">Krathwohl, 2002</xref>). Meta-analytic research suggests that flipped courses achieve the greatest professional competency improvements when classroom time is allocated to application and analysis rather than to rote memorisation (<xref ref-type="bibr" rid="ref34">Lundin et al., 2018</xref>). The extent to which higher-order things provoke more visual effort remains ambiguous. A persistent risk is confounding conceptual demand with surface reading: higher-order items in MC banks are often shorter because they presume context, making raw dwell time uninterpretable unless length is controlled. <xref ref-type="bibr" rid="ref38">&#x00D6;zdemir and Tosun (2025)</xref> observed prolonged fixation durations on analysis-level questions, but <xref ref-type="bibr" rid="ref1">Abt et al. (2024)</xref> found no demand impact after adjusting for stem length, underscoring the risk of confounding conceptual complexity with textual superficiality. Research utilizing multiple-choice formats indicates that higher-Bloom stems are frequently intentionally concise, since they assume prior context, rendering raw dwell time an unreliable indicator until length is taken into account. This dichotomy mirrors flipped sequencing (pre-class fundamentals vs. in-class application/analysis).</p>
<p>To achieve a discernible contrast while maintaining statistical power, we categorize Bloom levels 1&#x2013;2 (Remember, Understand) as low demand and levels 3&#x2013;4 (Apply, Analyse) as high demand. This division reflects the instructional cadence of flipped classrooms&#x2014;fundamentals before class versus in-depth exploration during class&#x2014;and aligns with systematic evaluations categorizing levels 3&#x2013;4 as &#x201C;higher-order cognition&#x201D; (<xref ref-type="bibr" rid="ref49">Zainuddin and Halili, 2016</xref>). By examining whether gaze-based effort increases or unexpectedly decreases on these higher-order items, we directly investigate the prevalent notion that heightened load-sensitive indicators invariably results in prolonged fixation and increased regressions.</p>
</sec>
<sec id="sec7">
<label>2.5</label>
<title>Gender as a exploratory moderator</title>
<p>Minor yet consistent sex differences in eye movement behavior can skew demand or accuracy estimates if not properly managed. <xref ref-type="bibr" rid="ref18">Gabel et al. (2025)</xref> demonstrate that eye-tracking uncovers teachers&#x2019; implicit gender biases&#x2014;pre-service teachers fixate more on female students in ways that mirror their IAT-measured attitudes&#x2014;while <xref ref-type="bibr" rid="ref6">Argunsah et al. (2025)</xref> reveal that female medical students exhibit stronger visual learning preferences and higher GPAs, suggesting gendered differences in attention and performance. Meta-analyses report small, task-dependent sex differences (women: slightly longer fixations/more regressions; men: faster scanning at comparable accuracy). Given our unbalanced cohort (&#x2248; 76% female), Gender is treated as a covariate; all moderation is exploratory.</p>
</sec>
<sec id="sec8">
<label>2.6</label>
<title>Development of research questions</title>
<p>Current learning analytics roadmaps emphasize the integration of multimodal, fine-grained process data&#x2014;such as eye tracking, keystroke logging, and physiological sensors&#x2014;to complement traditional outcome measures (<xref ref-type="bibr" rid="ref14">D&#x2019;Mello et al., 2017</xref>). These frameworks highlight a paradigm shift toward real-time, data-informed personalization in digital learning environments, where actionable insights are derived not only from what learners answer but also from how they engage, hesitate, or struggle during task performance. Guided by the preceding review, the present study addresses four interrelated questions concerning cognitive demand, visual engagement, performance, and learner characteristics.</p>
<sec id="sec9">
<label>2.6.1</label>
<title>Research questions (model-explicit)</title>
<p>We study trial-level relations among Bloom demand (High vs. Low), two gaze metrics&#x2014;Fixation Intensity (zFI) and Regression Rate (zRR)&#x2014;and Accuracy, while statistically controlling item text length with TotalWC_z (z-scored word count of stem+options). Gender is included as a covariate and all gender findings are exploratory. Grounded in the flipped-learning context and prior evidence that fixation duration and regressions can index processing effort under appropriate controls, we asked four questions:</p>
<sec id="sec10">
<label>2.6.1.1</label>
<title>Research question 1 &#x2013; demand and gaze-based effort</title>
<p><italic>Do higher-order items (Apply/Analyse) elicit greater visual effort than lower-order items (Remember/Understand) once item length is taken into account?</italic> Visual effort is operationalized by Fixation Intensity (FI)&#x2014;total dwell time on stem + options&#x2014;and Regression Rate (RR)&#x2014;the proportion of backward saccades. This analysis will determine whether higher cognitive demand is reflected not only in eventual correctness but also in the moment-by-moment allocation of visual attention during task performance.</p>
</sec>
<sec id="sec11">
<label>2.6.1.2</label>
<title>Research question 2 &#x2013; gaze-based effort and performance</title>
<p><italic>RQ2 (Gaze-based effort &#x2192; performance). Do FI and RR, above and beyond Bloom demand and Total word count (stem + options), predict the probability of answering an item correctly?</italic> This approach helps disentangle the effects of genuine conceptual challenge from other item features (e.g., text length or surface layout).</p>
</sec>
<sec id="sec12">
<label>2.6.1.3</label>
<title>Research question 3 &#x2013; objective versus subjective load</title>
<p><italic>RQ3 (Objective vs. subjective effort). To what extent do block-level self-reports of mental effort (SR_LOAD) align with objective, trial-level gaze indicators (FI, RR) and block accuracy?</italic> This approach allows us to compare fine-grained, moment-by-moment gaze data with learners&#x2019; retrospective, aggregate perceptions of effort for each block, highlighting the strengths and limitations of each measurement strategy.</p>
</sec>
<sec id="sec13">
<label>2.6.1.4</label>
<title>Research question 4 &#x2013; exploratory gender check</title>
<p><italic>Do females and males differ in average FI or RR, and does gender moderate the relation between FI and success?</italic> Given the small and imbalanced subsample, all gender analyses are treated as exploratory.</p>
<p>These questions were addressed with crossed mixed-effects models at the attempt level (trials nested within both students and items). Length was modeled with a z-standardized TotalWC covariate; item-level word-count diagnostics are provided in the Supplement.</p>
</sec>
</sec>
</sec>
</sec>
<sec sec-type="methods" id="sec14">
<label>3</label>
<title>Method</title>
<sec id="sec15">
<label>3.1</label>
<title>Participants and ethical procedures</title>
<p>Forty-five undergraduate volunteers (29 women, 16 men; <italic>M</italic> age&#x202F;=&#x202F;20.4&#x202F;years, <italic>SD</italic>&#x202F;=&#x202F;1.2) enrolled in an English-medium business-skills course (Organizational Behavior) at a research-intensive university took part in the eye-tracking study. After the data cleaning, we left 34 with analysable record. All participants were familiar with flipped classroom instruction through previous module experiences, but none had prior exposure to eye-tracking technology. Gender was included as a covariate primarily to control for known differences in eye-movement patterns, as prior research has shown that gender can influence fixation duration and regression rates. Due to our small sample size and unbalanced gender distribution, all findings related to gender should be interpreted as exploratory and hypothesis-generating rather than confirmatory. Participation was elective and rewarded with course credit plus shopping coupons (&#x2248; US$15) if students completed at least three of the scheduled laboratory sessions. The institutional review board approved all procedures (Ref. 2023-EC134-2324). Students signed written consent that described data uses, anonymity safeguards, and their right to withdraw at any time without penalty. All data were de-identified at source and analysed only in aggregate, in accordance with the <italic>Standards for Educational and Psychological Testing</italic> (<xref ref-type="bibr" rid="ref4">American Educational Research Association, American Psychological Association, &#x0026; National Council on Measurement in Education, 2014</xref>).</p>
</sec>
<sec id="sec16">
<label>3.2</label>
<title>Course context, session structure, and multiple-choice bank</title>
<p>This course is a second-year core module on organisational behavior delivered in a flipped format. Before each contact session, students studied a chapter in the McGraw-Hill <italic>Connect</italic> e-book, short screencasts, and self-check quizlets. During the 11-week laboratory phase that forms the present dataset, students attended weekly 30-min eye-tracking blocks scheduled immediately before the regular lesson. Each block comprised five Bloom-coded multiple-choice (MC) questions (one <italic>remembers</italic>, two <italic>understand</italic>, one <italic>apply</italic>, one <italic>analyse</italic>) drawn without replacement from an expert-reviewed bank of 55 items. The stems tested the chapter of the week; distractors targeted common misconceptions identified in earlier cohorts.</p>
<p>Two instructional designers first classified every item according to the revised Bloom taxonomy: inter-rater <italic>&#x03BA;</italic>&#x202F;=&#x202F;0.82 (96% agreement). To maximize statistical power in the mixed-models analysis, we later collapsed the four categories into Low-Demand (<italic>remember + understand</italic>) and High-Demand (<italic>apply + analyse</italic>). Power analysis was conducted prior to data collection using the variance components observed in a pilot sample. With a projected intraclass correlation coefficient (ICC) of approximately 0.20, the planned 320 item attempts were estimated to provide 80% power to detect medium fixed effects, given the observed intraclass correlation. This design deliberately maximized within-subject contrasts while recognizing the trade-off in generalisability due to a modest N for demographic subgroup comparisons.</p>
</sec>
<sec id="sec17">
<label>3.3</label>
<title>Apparatus and area-of-interest (AOI) definition</title>
<p>Eye movements were recorded with a Tobii Pro Nano eye tracker (60&#x202F;Hz; manufacturer-reported accuracy &#x003C; 0.4&#x00B0;) mounted below a 14-inch laptop display (1,920&#x202F;&#x00D7;&#x202F;1,080 px). Each session began with a five-point calibration; data collection proceeded only when the average gaze-position error was &#x2264; 0.8&#x00B0;, otherwise calibration was repeated.</p>
<p>Items were presented in a fixed HTML layout. Using Tobii Pro Lab v1.204, we drew non-overlapping rectangular AOIs that were coextensive with each on-screen component: the stem and the five options (A&#x2013;E). AOI coordinates were held constant across items. Fixations were attributed to the AOI entered at the first in-bounds sample; fixations that straddled boundaries were assigned to the recipient AOI at entry. Transitions between successive fixations located in different AOIs were logged to characterize navigation among question components (e.g., stem &#x2194; option back-tracking).</p>
</sec>
<sec id="sec18">
<label>3.4</label>
<title>Event parsing</title>
<p>Fixations and saccades were parsed with Tobii Pro Lab&#x2019;s dispersion-based algorithm (dispersion threshold&#x202F;=&#x202F;30 px; minimum fixation duration&#x202F;=&#x202F;60&#x202F;ms). These settings are reported once here to avoid duplication elsewhere.</p>
<p>Item characteristics and text-length control.</p>
<p>Item word-counting and covariate. To disentangle conceptual demand from surface reading, we operationalized item length at three levels: StemWC (stem words), OptionsWC (sum across retained options), and TotalWC&#x202F;=&#x202F;StemWC + OptionsWC. Word counts were computed on the rendered HTML (whitespace-delimited tokens), then merged back to trial records. For modeling, TotalWC was z-standardized across attempts (TotalWC_z) and entered as a covariate in all primary models.</p>
<p>Item-level check. Because length is an item property, we compared per-item means (each QuestionID counted once) between Low- vs. High-Bloom items using Welch tests. Effects were small and not statistically significant at the item level; for transparency we report Low/High means, High&#x2013;Low differences, and t (df), p in <xref ref-type="supplementary-material" rid="SM1">Supplementary Table S1</xref>. Given attempt-level differences and to be conservative, TotalWC_z is retained as a control in the primary mixed-effects analyses.</p>
<p>Pre-registration / power note. The planned &#x2265;300 attempts with ICC&#x202F;&#x2248;&#x202F;0.20 were expected (Monte-Carlo) to provide ~80% power to detect medium fixed effects (<italic>&#x03B2;</italic>&#x202F;&#x2248;&#x202F;0.35 SD). Length control does not change this power calculation but reduces bias in the Bloom coefficient.</p>
</sec>
<sec id="sec19">
<label>3.5</label>
<title>Self-report instruments</title>
<p>Immediately after each eye-tracking block, students completed a four-item, 5-point Likert checklist adapted from the NASA-TLX mental-effort dimension and the <xref ref-type="bibr" rid="ref40">Paas and Van Merri&#x00EB;nboer (1994)</xref> single-item scale:</p>
<p>&#x201C;How much mental effort did you exert to understand the questions?&#x201D;</p>
<p>&#x201C;How difficult were the underlying concepts?&#x201D;</p>
<p>&#x201C;How complex were the questions?&#x201D;</p>
<p>We adapted three block-level prompts (5-point Likert) from the NASA-TLX mental-effort dimension and Paas &#x0026; Van Merri&#x00EB;nboer&#x2019;s single-item index. The self-reported cognitive-load index (SR_LOAD) is the mean of the three items (<italic>&#x03B1;</italic>&#x202F;=&#x202F;0.86). We acknowledge that NASA-TLX does not separate intrinsic from extraneous load; our choice prioritized brevity and ecological validity during weekly labs. In line with reviewer guidance, we treat SR_LOAD as a coarse, block-level comparator to objective gaze signals rather than as a multidimensional load diagnostic; future work should add instruments such as the Cognitive Load Scale for load-type decomposition. Self-report ratings were collected at the block level to reduce participant burden and better reflect the overall visual effort required for each 5-item set, recognizing that this approach limits the per-item, fine-grained correspondence with gaze-based indicators but maintains ecological validity for classroom settings. While our self-reported cognitive load index (SR_LOAD) was adapted from established scales, it does not differentiate between intrinsic and extrinsic cognitive load, as do more recently developed instruments such as the Cognitive Load Scale. Future studies should incorporate these validated tools for finer-grained analysis of cognitive load types in educational settings.</p>
</sec>
<sec id="sec20">
<label>3.6</label>
<title>Eye-movement metrics</title>
<p>We computed two load-sensitive gaze measures per attempt by summing across stem and options AOIs: Fixation Intensity (FI)&#x2014;total dwell time (ms); and Regression Rate (RR)&#x2014;the proportion of backward saccades relative to total saccades. To reduce leverage of extreme scan-paths, both metrics were winsorised at the 98th percentile, then z-standardized within participant (grand-mean&#x202F;=&#x202F;0, SD&#x202F;=&#x202F;1) to remove baseline speed differences. Saccade-velocity and pupil signals were exploratory and are not analysed due to known noise at 60&#x202F;Hz.</p>
<p>Following prior work, we treat Fixation Intensity and Regression Rate as &#x201C;load-sensitive&#x201D; metrics: longer fixations often reflect deeper semantic processing or greater integrative demand, and more regressions tend to accompany ambiguity or inconsistency (<xref ref-type="bibr" rid="ref44">Spichtig et al., 2017</xref>; <xref ref-type="bibr" rid="ref25">Inhoff et al., 2019</xref>; <xref ref-type="bibr" rid="ref30">Lai et al., 2013</xref>; <xref ref-type="bibr" rid="ref46">van Gog and Jarodzka, 2013</xref>). These associations are context-dependent, influenced by text complexity, prior knowledge, reading skill, and task design (<xref ref-type="bibr" rid="ref12">Becker et al., 2022</xref>; <xref ref-type="bibr" rid="ref11">Becker et al., 2023</xref>). Notably, when items differ in length or layout&#x2014;as in this study&#x2014;Fixation Intensity may not cleanly index load-sensitive indicators. We therefore interpret these measures as indicators of processing effort when task characteristics are held constant, while cautioning that they are not direct, unambiguous measures of load-sensitive indicators. Their validity as proxies hinges on controlling extraneous factors and aligning use with the empirical contexts in which they were originally validated (e.g., <xref ref-type="bibr" rid="ref30">Lai et al., 2013</xref>; <xref ref-type="bibr" rid="ref44">Spichtig et al., 2017</xref>).</p>
</sec>
<sec id="sec21">
<label>3.7</label>
<title>Data structure and analytic power</title>
<p>After excluding 12 trials with &#x003E;30% data loss, the analytic file comprises 320 item attempts completed by 34 students across 55 items (median&#x202F;=&#x202F;9 attempts per learner). The crossed structure yields most precision from the large number of level-1 observations. This structure is well suited for multilevel models, which gain precision primarily from the number of level-1 (item) observations rather than the number of level-2 (person) units (<xref ref-type="bibr" rid="ref9">Barr et al., 2013</xref>).</p>
<p>Post-hoc power analysis (reported in Methods, Section 3.6) indicates that, with the observed intraclass correlation coefficient (ICC&#x202F;&#x2248;&#x202F;0.20), this design provides &#x003E;80% power to detect medium-sized fixed effects (<italic>&#x03B2;</italic>&#x202F;&#x2248;&#x202F;0.35 SD, OR &#x2248; 1.4) for our primary gaze metrics. However, subgroup analyses (e.g., gender interactions) and detection of small effects remain underpowered, as expected with modest N. We therefore interpret all subgroup and interaction findings as exploratory and hypothesis-generating, not confirmatory.</p>
</sec>
<sec id="sec22">
<label>3.8</label>
<title>Analysis plan</title>
<p>All predictors were grand-mean centered. Fixation Intensity (FI) and Regression Rate (RR) were winsorized at the 98th percentile and standardized within participant (zFI, zRR). TotalWC_z denotes the z-scored total word count of each item (stem + options) and was included as a covariate in all primary models.</p>
<p>RQ1: Demand &#x2192; gaze-based effort. We estimated two linear mixed-effects models in which zFI and zRR were the dependent variables. Fixed effects were Bloom demand (High vs. Low; effects-coded &#x00B1;0.5), Gender (male&#x202F;=&#x202F;1), and TotalWC_z. Each model included crossed random intercepts for Student and Item to account for clustering of attempts within persons and questions.</p>
<disp-formula id="E1">
<mml:math id="M1">
<mml:mtable columnalign="left" displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi mathvariant="normal">Y</mml:mi>
<mml:mi>ij</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x03B2;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x03B2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mtext>Demand</mml:mtext>
<mml:mi>ij</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x03B2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msub>
<mml:mtext>Gender</mml:mtext>
<mml:mi mathvariant="normal">j</mml:mi>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x03B2;</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mtext>TotalWC</mml:mtext>
<mml:mo>_</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">z</mml:mi>
<mml:mi>ij</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mi>0j</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mi>0i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x03B5;</mml:mi>
<mml:mi>ij</mml:mi>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>RQ2: Gaze-based effort &#x2192; success. Item-level correctness (0/1) was modeled with a binomial logit generalized linear mixed-effects model. Predictors were zFI, zRR, Bloom demand, Gender, and TotalWC_z, with crossed random intercepts for Student and Item. An exploratory zFI&#x202F;&#x00D7;&#x202F;Gender term tested moderation; given limited power, this interaction is interpreted cautiously while the main Gender effects are retained in the fixed-effects set.</p>
<disp-formula id="E2">
<mml:math id="M2">
<mml:mtable columnalign="left" displaystyle="true">
<mml:mtr>
<mml:mtd>
<mml:mtext>logit</mml:mtext>
<mml:mspace width="0.33em"/>
<mml:mo stretchy="true">{</mml:mo>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mspace width="0.33em"/>
<mml:mo stretchy="true">(</mml:mo>
<mml:msub>
<mml:mtext>Accuracy</mml:mtext>
<mml:mi>ij</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="true">)</mml:mo>
<mml:mo stretchy="true">}</mml:mo>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x03B2;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x03B2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>zFI</mml:mi>
<mml:mi>ij</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x03B2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>zRR</mml:mi>
<mml:mrow>
<mml:mi>ij</mml:mi>
<mml:mo>+</mml:mo>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mspace width="0.33em"/>
<mml:msub>
<mml:mi>&#x03B2;</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:msub>
<mml:mtext>Demand</mml:mtext>
<mml:mrow>
<mml:mi>ij</mml:mi>
<mml:mo>+</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.33em"/>
<mml:msub>
<mml:mi>&#x03B2;</mml:mi>
<mml:mn>4</mml:mn>
</mml:msub>
<mml:msub>
<mml:mtext>Gender</mml:mtext>
<mml:mi>ij</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x03B2;</mml:mi>
<mml:mn>5</mml:mn>
</mml:msub>
<mml:mtext>TotalWC</mml:mtext>
<mml:mo>_</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">z</mml:mi>
<mml:mrow>
<mml:mi>ij</mml:mi>
<mml:mo>+</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.33em"/>
<mml:msub>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mi>0j</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">u</mml:mi>
<mml:mi>0i</mml:mi>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>RQ3: Objective vs. subjective load. Because self-reported mental effort (SR_LOAD) was collected at the block level, trial data were aggregated to Student &#x00D7; Block (mean zFI, mean zRR, and block accuracy). Associations were summarized with Pearson correlations and Fisher-transformed 95% confidence intervals for the pairs SR_LOAD &#x00D7; mean zFI, SR_LOAD &#x00D7; mean zRR, and SR_LOAD &#x00D7; accuracy.</p>
<p>Estimation and inference. Linear mixed models were fitted with lme4/lmerTest using REML&#x202F;=&#x202F;FALSE for comparability; denominator degrees of freedom followed the Satterthwaite approximation. The GLMM was estimated by Laplace approximation (optimizer bobyqa, maxfun&#x202F;=&#x202F;2&#x202F;&#x00D7;&#x202F;10^5). For LMMs we report unstandardized coefficients (&#x03B2;), standard errors, 95% CIs, and random-effect variances; for the GLMM we report odds ratios with 95% Wald CIs in addition to variance components. Potential singular fits (near-zero random-effect variance) are flagged and interpreted with caution. Robustness was further evaluated via 2,000 non-parametric bootstraps on fixed-effect estimates.</p>
<p>Software and reproducibility. All analyses were conducted in R (version [fill in]; R Core Team), using the following packages: lme4 for mixed models (<xref ref-type="bibr" rid="ref9001">Bates et al., 2015</xref>), lmerTest for Satterthwaite degrees of freedom (<xref ref-type="bibr" rid="ref9003">Kuznetsova et al., 2017</xref>), dplyr for data manipulation (<xref ref-type="bibr" rid="ref9004">Wickham et al., 2023</xref>), and effsize for standardized mean-difference estimates. Exact package versions and the R session details are reported in Supplementary S5 (Computing environment). Replicable model formulas are given in Section 3.6; code to re-run the models is supplied in the Supplement.</p>
<p>Descriptive statistics for attempt-level zFI, zRR, and accuracy by Bloom level appear in <xref ref-type="table" rid="tab1">Table 1</xref>. Item-level length characteristics and Welch tests are provided in <xref ref-type="supplementary-material" rid="SM1">Supplementary Table S1</xref>.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Descriptive statistics by Bloom demand (attempt-level).</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Demand level</th>
<th align="center" valign="top">zFI Mean &#x00B1; SD</th>
<th align="center" valign="top">zRR Mean &#x00B1; SD</th>
<th align="center" valign="top">Accuracy mean &#x00B1; SD</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Low (Bloom 1&#x2013;2)</td>
<td align="center" valign="top">0.04&#x202F;&#x00B1;&#x202F;0.99</td>
<td align="center" valign="top">0.02&#x202F;&#x00B1;&#x202F;0.99</td>
<td align="center" valign="top">0.35&#x202F;&#x00B1;&#x202F;0.48</td>
</tr>
<tr>
<td align="left" valign="top">High (Bloom 3&#x2013;4)</td>
<td align="center" valign="top">&#x2212;0.15&#x202F;&#x00B1;&#x202F;1.02</td>
<td align="center" valign="top">&#x2212;0.06&#x202F;&#x00B1;&#x202F;1.06</td>
<td align="center" valign="top">0.31&#x202F;&#x00B1;&#x202F;0.47</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Attempt-level means (FI and RR are within-participant z-scores). Sample: 320 attempts, 34 students, 55 items.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="sec23">
<label>3.9</label>
<title>Rationale for the gender term</title>
<p>Small but systematic gender differences in reading and STEM eye-movements have been reported (e.g., slightly longer fixations and more regressions for females), potentially confounding demand effects (Zhan et al., 2020; <xref ref-type="bibr" rid="ref23">Huang and Chen, 2016</xref>). Women typically show slightly longer fixations and more regressions, whereas men scan more quickly while achieving comparable accuracy. We therefore include Gender (male&#x202F;=&#x202F;1) as a covariate in all primary models to absorb speed/strategy differences. Given the unbalanced sample (26F, 8&#x202F;M) and Monte-Carlo power &#x2264;5% for small interactions under our variance structure, all gender findings&#x2014;including zFI&#x202F;&#x00D7;&#x202F;Gender&#x2014;are labeled exploratory. Gender here is included primarily as a covariate to control for known speed-accuracy trade-offs.</p>
</sec>
</sec>
<sec sec-type="results" id="sec24">
<label>4</label>
<title>Results</title>
<sec id="sec25">
<label>4.1</label>
<title>Portrait of the dataset</title>
<p>Students attempted 320 items (34 learners; 55 items). At the descriptive level (<xref ref-type="table" rid="tab1">Table 1</xref>), higher-order items were answered slightly less often and received shorter fixation times (&#x2248;0.2 SD lower FI). RR and block-level SR_LOAD were very similar across Bloom levels. These patterns already suggest that text length may be driving dwell-time differences more than conceptual demand, motivating the inclusion of TotalWC in the primary models.</p>
</sec>
<sec id="sec26">
<label>4.2</label>
<title>Research questions 1: Does demand alter gaze-based effort once length is controlled?</title>
<p>Two LMMs regressed zFI and zRR on Bloom demand (High vs. Low), Gender, and TotalWC_z, with crossed random intercepts for Student and Item. When FI and RR were modeled from Bloom demand with TotalWC and Gender as covariates (random intercepts for students and items), item length&#x2014;not Bloom level&#x2014;was the reliable predictor of FI. Longer/shorter items were associated with correspondingly lower/greater FI (TotalWC term, <italic>p</italic>&#x202F;=&#x202F;0.004), and the nominal Bloom contrast no longer reached significance after this control. RR showed no detectable change by Bloom. Thus, in this authentic quiz bank, how much text students had to process mattered more for dwell time than whether the item targeted lower- or higher-order cognition. Full coefficients appear in <xref ref-type="table" rid="tab2">Table 2</xref>.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Linear mixed-effects models for gaze metrics (zFI, zRR) controlling total word count.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Outcome</th>
<th align="left" valign="top">Predictor</th>
<th align="center" valign="top">&#x03B2;</th>
<th align="center" valign="top">SE</th>
<th align="center" valign="top">
<italic>t</italic>
</th>
<th align="center" valign="top">df (Satt.)</th>
<th align="center" valign="top">
<italic>p</italic>
</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">zFI</td>
<td align="left" valign="top">ItemType (High vs. Low)</td>
<td align="center" valign="top">&#x2212;0.119</td>
<td align="center" valign="top">0.193</td>
<td align="center" valign="top">&#x2212;0.614</td>
<td align="center" valign="top">45.775</td>
<td align="center" valign="top">0.542</td>
</tr>
<tr>
<td/>
<td align="left" valign="top">Gender (1&#x202F;=&#x202F;male)</td>
<td align="center" valign="top">&#x2212;0.779</td>
<td align="center" valign="top">0.244</td>
<td align="center" valign="top">&#x2212;3.191</td>
<td align="center" valign="top">33.613</td>
<td align="center" valign="top">0.003</td>
</tr>
<tr>
<td/>
<td align="left" valign="top">TotalWC_z (per 1 SD)</td>
<td align="center" valign="top">&#x2212;0.232</td>
<td align="center" valign="top">0.077</td>
<td align="center" valign="top">&#x2212;3.002</td>
<td align="center" valign="top">46.791</td>
<td align="center" valign="top">0.004</td>
</tr>
<tr>
<td align="left" valign="top">zRR</td>
<td align="left" valign="top">ItemType (High vs. Low)</td>
<td align="center" valign="top">&#x2212;0.101</td>
<td align="center" valign="top">0.133</td>
<td align="center" valign="top">&#x2212;0.764</td>
<td align="center" valign="top">291.590</td>
<td align="center" valign="top">0.446</td>
</tr>
<tr>
<td/>
<td align="left" valign="top">Gender (1&#x202F;=&#x202F;male)</td>
<td align="center" valign="top">&#x2212;0.314</td>
<td align="center" valign="top">0.180</td>
<td align="center" valign="top">&#x2212;1.748</td>
<td align="center" valign="top">31.346</td>
<td align="center" valign="top">0.090</td>
</tr>
<tr>
<td/>
<td align="left" valign="top">TotalWC_z (per 1 SD)</td>
<td align="center" valign="top">&#x2212;0.025</td>
<td align="center" valign="top">0.053</td>
<td align="center" valign="top">&#x2212;0.476</td>
<td align="center" valign="top">301.166</td>
<td align="center" valign="top">0.635</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Crossed random intercepts for students and items. Fit&#x2014;zFI: AIC 794.6, BIC 820.9, logLik &#x2212;390.3; random SDs: item 0.487, student 0.543, residual 0.663. Fit&#x2014;zRR: AIC 903.5, BIC 929.9, logLik &#x2212;444.8; item intercept variance &#x2248; 0 (singular), student SD 0.294, residual SD 0.940. Predictors grand-mean centred; FI/RR winsorised at 98th percentile and z-scored within participant. TotalWC_z&#x202F;=&#x202F;z-scored total word count across stem + options.</p>
</table-wrap-foot>
</table-wrap>
<p>A note on sensitivity: A stem-only specification (using StemWC in place of TotalWC) produced the expected positive association between stem length and FI and a small negative Bloom contrast, underlining that text-surface features can easily masquerade as &#x201C;demand effects.&#x201D; Details of this check are reported beneath <xref ref-type="table" rid="tab2">Table 2</xref>.</p>
</sec>
<sec id="sec27">
<label>4.3</label>
<title>Research questions 2: Do gaze metrics predict correctness?</title>
<p>A logistic GLMM with crossed random intercepts (Student, Item) predicted accuracy from FI, RR, Bloom demand, and gender (<xref ref-type="table" rid="tab3">Table 3</xref>).</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Logistic GLMM predicting item accuracy.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Predictor</th>
<th align="center" valign="top">logit &#x03B2;</th>
<th align="center" valign="top">SE</th>
<th align="center" valign="top">
<italic>z</italic>
</th>
<th align="center" valign="top">
<italic>p</italic>
</th>
<th align="center" valign="top">OR</th>
<th align="center" valign="top">95% CI (OR)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">z_FI (per 1 SD)</td>
<td align="center" valign="top">0.262</td>
<td align="center" valign="top">0.143</td>
<td align="center" valign="top">1.826</td>
<td align="center" valign="top">0.068</td>
<td align="center" valign="top">1.299</td>
<td align="center" valign="top">[0.981, 1.72]</td>
</tr>
<tr>
<td align="left" valign="top">z_RR (per 1 SD)</td>
<td align="center" valign="top">&#x2212;0.215</td>
<td align="center" valign="top">0.149</td>
<td align="center" valign="top">&#x2212;1.444</td>
<td align="center" valign="top">0.149</td>
<td align="center" valign="top">0.806</td>
<td align="center" valign="top">[0.602, 1.08]</td>
</tr>
<tr>
<td align="left" valign="top">ItemType (High vs. Low)</td>
<td align="center" valign="top">&#x2212;0.451</td>
<td align="center" valign="top">0.446</td>
<td align="center" valign="top">&#x2212;1.011</td>
<td align="center" valign="top">0.312</td>
<td align="center" valign="top">0.637</td>
<td align="center" valign="top">[0.266, 1.527]</td>
</tr>
<tr>
<td align="left" valign="top">Gender (1&#x202F;=&#x202F;male)</td>
<td align="center" valign="top">0.052</td>
<td align="center" valign="top">0.334</td>
<td align="center" valign="top">0.156</td>
<td align="center" valign="top">0.876</td>
<td align="center" valign="top">1.053</td>
<td align="center" valign="top">[0.547, 2.027]</td>
</tr>
<tr>
<td align="left" valign="top">TotalWC_z (per 1 SD)</td>
<td align="center" valign="top">0.253</td>
<td align="center" valign="top">0.178</td>
<td align="center" valign="top">1.419</td>
<td align="center" valign="top">0.156</td>
<td align="center" valign="top">1.287</td>
<td align="center" valign="top">[0.908, 1.825]</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Crossed random intercepts for students and items; student variance at boundary (&#x2248; 0). Fit: AIC 405.4, BIC 435.5, logLik &#x2212;194.7. Odds ratios are exponentiated coefficients with Wald 95% CIs.</p>
</table-wrap-foot>
</table-wrap>
<p>A logistic GLMM (logit link) predicted Accuracy from zFI, zRR, Bloom demand, Gender, and TotalWC_z, with the same random-effects structure (Question random intercept retained; Student random intercept at boundary). The accuracy model (GLMM) indicated a clear tendency: items on which students fixated longer were more likely to be answered correctly (&#x2248;1.30&#x202F;&#x00D7;&#x202F;odds per +1 SD FI), whereas more frequent regressions tended to accompany lower odds (&#x2248;0.81&#x202F;&#x00D7;&#x202F;per +1 SD RR).</p>
<p>With this small sample these effects approached but did not meet conventional significance levels; nevertheless, effect sizes were educationally meaningful and consistent with theory. Bloom demand, gender, and length did not add unique predictive value once FI and RR were in the model. Notably, the student random intercept sat on the boundary while the item intercept was substantial, indicating that items varied more in difficulty than students varied in overall performance. See <xref ref-type="table" rid="tab3">Table 3</xref> for model summaries.</p>
</sec>
<sec id="sec28">
<label>4.4</label>
<title>Research questions 3: How do objective and subjective load relate?</title>
<p>Block-level SR_LOAD showed near-zero correlations with mean FI, mean RR, and block accuracy; Fisher 95% CIs exclude even modest associations. In other words, the retrospective &#x201C;how hard did that block feel?&#x201D; rating did not track the micro-fluctuations captured by gaze. This reinforces the value of unobtrusive process signals for formative diagnostics. Supplementary correlation estimates appear in Supplementary Table S4.</p>
</sec>
<sec id="sec29">
<label>4.5</label>
<title>Research questions 4: exploratory gender effects</title>
<p>Males exhibited shorter fixation times on average (faster processing) with no reliable difference in RR. Crucially, gender neither predicted accuracy nor changed the beneficial slope of FI. Given the small and imbalanced male subgroup, these observations are treated as controls rather than confirmatory findings. Relevant terms are reported alongside the fixed-effect tables.</p>
</sec>
<sec id="sec30">
<label>4.6</label>
<title>Result summary</title>
<p>Contrary to the simple &#x201C;harder &#x2192; longer&#x201D; expectation, higher-order items did not demand more dwell time once length was controlled. Instead, item length was the proximate driver of FI. Yet visual effort still mattered: longer fixation tended to help and frequent regressions tended to hinder success, pointing to two complementary process cues that conventional correctness scores miss. Paired with the divergence between self-reports and gaze, these results support the use of classroom-friendly eye-tracking as a response-process lens for flipped-class diagnostics, while also highlighting the necessity of length-matched item banks for clean causal interpretation. <xref ref-type="table" rid="tab1">Tables 1</xref>&#x2013;<xref ref-type="table" rid="tab3">3</xref> and <xref ref-type="supplementary-material" rid="SM1">Supplementary Tables</xref> (word-count checks; SR_LOAD correlations) document the underlying estimates.</p>
</sec>
</sec>
<sec sec-type="discussion" id="sec31">
<label>5</label>
<title>Discussion</title>
<sec id="sec32">
<label>5.1</label>
<title>General discussion</title>
<p>This study adds response-process evidence to flipped-class assessment by showing that two simple gaze metrics&#x2014;fixation intensity (FI) and regression rate (RR)&#x2014;carry complementary instructional signals during authentic, pre-class MCQs. In our crossed mixed-effects models, longer dwell time tended to help (OR&#x2248;1.30 per SD), whereas frequent back-tracking tended to hurt (OR&#x2248;0.81), while block-level self-reports showed near-zero correspondence with either gaze metric. Equally important, the apparent &#x201C;higher-Bloom &#x21D2; more time&#x201D; intuition did not hold once surface text was considered: with total word count (stem+options) entered as a covariate, the Bloom&#x2013;fixation association attenuated to non-significance, revealing a &#x201C;harder-but-shorter&#x201D; design pattern rather than a pure demand effect. Together, these findings reframe classroom eye tracking as measurement-aware diagnostics: FI and RR are informative when surface features are standardized or modeled, and they illuminate moment-to-moment engagement that correctness and retrospective ratings miss. The small gender speed difference we observed (men fixated less without an accuracy penalty) did not alter the fixation&#x2013;performance link, suggesting that process-aware feedback rules can be applied equitably in similar cohorts.</p>
<p>Practically, the results point to a concrete design protocol for future item banks and for scalable analytics: (i) standardize total word count in narrow bands; (ii) equalize option lengths and hold the number of options fixed; (iii) pre-register TotalWC as a covariate in primary models; and (iv) replace block-level self-reports with brief, item-level, multidimensional load measures. Under these conditions, classroom-friendly 60&#x202F;Hz trackers can provide response-process validity evidence and serve as a &#x201C;gold reference&#x201D; to benchmark lighter-weight proxies (e.g., response-time distributions, option-comparison sequences, click/keystroke traces, or privacy-preserving webcam gaze approximations). A conservative pathway&#x2014;standardized bank &#x2192; small-class pilots &#x2192; multi-class validation&#x2014;can move flipped-class diagnostics toward a practical balance of cost, usability, and validity, complementing (not displacing) human instruction.</p>
</sec>
<sec id="sec33">
<label>5.2</label>
<title>Limitations</title>
<p>We note four limitations:</p>
<list list-type="simple">
<list-item>
<p>(1)&#x00A0;&#x00A0;Surface-text confound (interpretation risk).</p>
</list-item>
</list>
<list list-type="simple">
<list-item>
<p>The Bloom&#x2013;fixation link vanished once TotalWC (stem + options) was controlled, and TotalWC negatively predicted Fixation Intensity (FI). High-Bloom stems were, on average, shorter, so the earlier &#x201C;harder-but-shorter&#x201D; pattern is best explained by text length rather than conceptual demand. Without balancing or adjusting for length/layout, gaze metrics may misrepresent difficulty. Future work should: (a) construct length-matched item pairs within Bloom levels or (b) statistically adjust for characters/words and layout complexity; (c) include a manipulation check to verify parity before analysis.</p>
</list-item>
</list>
<list list-type="simple">
<list-item>
<p>(2)&#x00A0;&#x00A0;Sampling and power (generalizability).</p>
</list-item>
</list>
<list list-type="simple">
<list-item>
<p>With 34 students and ~320 attempts, trial-level fixed effects were estimated with acceptable precision; however, subgroup contrasts (e.g., Gender &#x00D7; FI) were under-powered and should be treated as exploratory. Replications across courses and institutions&#x2014;with larger, more balanced cohorts&#x2014;are needed to confirm demographic patterns and strengthen external validity. In the present study, gender served primarily as a covariate to account for known speed&#x2013;accuracy differences; all gender-related inferences remain hypothesis-generating.</p>
</list-item>
</list>
<list list-type="simple">
<list-item>
<p>(3)&#x00A0;&#x00A0;Self-report granularity (construct alignment).</p>
</list-item>
</list>
<list list-type="simple">
<list-item>
<p>Block-level self-reported effort (SR_LOAD) showed near-zero correlations with FI/RR, consistent with a level-of-analysis mismatch (block vs. item). To test convergent validity with process data, subsequent studies should collect item-level, multidimensional cognitive-load ratings (e.g., intrinsic vs. extraneous) and align their timing with each response. Where feasible, triangulate with brief, low-friction prompts embedded in the quiz flow.</p>
</list-item>
</list>
<list list-type="simple">
<list-item>
<p>(4)&#x00A0;&#x00A0;Signals and sampling (measurement scope).</p>
</list-item>
</list>
<list list-type="simple">
<list-item>
<p>The 60&#x202F;Hz tracker was sufficient for aggregate FI and regression counts (RR) but too coarse for micro-saccades or fine-grained pupillometry. We therefore restrict inference to fixation- and regression-based indicators. Replicating with &#x2265;120&#x202F;Hz devices would test robustness when higher-frequency information is available. For richer load diagnostics, future work should add multimodal signals (e.g., luminance-corrected pupil dilation, electrodermal activity) to improve sensitivity while monitoring privacy and classroom burden.</p>
</list-item>
</list>
</sec>
<sec id="sec34">
<label>5.3</label>
<title>Implications and design suggestions</title>
<p>The analytical strategy&#x2014;including random intercepts for both students and items, and robust estimation of confidence intervals via bootstrapping&#x2014;was specifically selected to address the data&#x2019;s hierarchical structure and mitigate the limitations imposed by a modest participant sample. These choices align with current best practices for analyzing nested educational data with small to moderate samples (<xref ref-type="bibr" rid="ref43">Snijders and Bosker, 2012</xref>).</p>
<p>This study shows that classroom-friendly eye tracking can yield actionable process signals during routine formative work. Scaling such use requires plug-and-play integration with LMSs, clear privacy/consent policies, and&#x2014;most importantly&#x2014;validated item banks so that adaptive algorithms respond to genuine cognitive demand rather than surface features. In near-term classroom practice, gaze can flag low-dwell/high-regression episodes for targeted scaffolds during pre-class study, while recognizing that reliable triggers require length-matched items or TotalWC-aware rules.</p>
<p>Because the Bloom&#x2013;fixation association disappeared once TotalWC (stem + options) was controlled&#x2014;and TotalWC negatively predicted fixation intensity&#x2014;future banks should: (i) standardize TotalWC within narrow bands by Bloom level; (ii) equalize option lengths and fix the number of options; (iii) pre-register TotalWC (and layout features) as covariates; and (iv) replace block-level self-reports with item-level, multidimensional load prompts to separate intrinsic and extraneous load. Practically, gaze metrics remain useful when surface features are either balanced by design or explicitly modeled.</p>
<p>An economical 60&#x202F;Hz tracker, or a high-resolution webcam with model-based gaze estimation is sufficient for fixation- and regression-based indicators. Embed the device in the pre-class quiz interface and stream two z-scored signals to an analytics microservice: dwell time and back-tracking frequency. Flag a potential struggle episode when dwell time falls &#x003E;1 SD below a student&#x2019;s baseline and regressions rise &#x003E;1 SD above baseline. Trigger just-in-time scaffolds (e.g., &#x201C;re-read stem,&#x201D; concise glossary, or a worked example) before submission. In borderline cases (short dwell without excessive regressions), surface low-cost supports (definitions/examples) rather than full hints. For finer-grained pupillometry or micro-saccades, consider &#x2265;120&#x202F;Hz devices in future iterations.</p>
<p>Use these pipelines to strengthen response-process validity as outlined in the Standards for Educational and Psychological Testing: confirm that students attend to the intended elements of higher-order items. Pair the gaze assessments with per-item micro self-reports (single-tap confidence or perceived difficulty). Joint modeling of objective (gaze) and subjective (self-report) evidence will reveal which nudges (extra time, hints, recap videos) best close gaps between perceived and actual effort and will iteratively refine personalisation over semesters.</p>
<p>However, certain compliance and ethical expectations must be taken into considerations. Adopt data-minimisation, local processing where feasible, opt-in consent, and transparent learner dashboards. Provide instructor controls to disable interventions, export diagnostics, and review item-level balance checks. Before real-time, gaze-informed interventions are deployed at scale, invest first in high-quality, standardized item banks and a light-touch analytics layer that privileges measurement integrity over automation speed. Careful design and staged validation will prevent text-length artefacts from being misread as cognitive struggle and will make adaptive support both responsible and reliable.</p>
</sec>
<sec id="sec35">
<label>5.4</label>
<title>Future research direction</title>
<p>Future work should implement parallel, length-matched forms at each Bloom level&#x2014;equating word/character count, layout, and option length&#x2014;and counterbalance presentation order across students. A preregistered analysis plan should include equivalence tests to determine whether Bloom effects remain negligible once TotalWC is controlled, alongside re-estimation of Fixation Intensity (R<sup>1</sup>) and Regression Rate (R<sup>2</sup>) using crossed mixed-effects models. Prospective power analyses should be calibrated for small effects and incorporate item- and student-level ICCs to ensure adequate precision for both fixed and random components.</p>
<p>To evaluate construct convergence at the appropriate grain size, block-level self-reports ought to be replaced with item-level, multidimensional prompts (e.g., intrinsic vs. extraneous load, single-tap confidence/difficulty). Analyses should prioritize within-person associations between R<sup>1</sup>/R<sup>2</sup> and self-reports and use ROC and precision&#x2013;recall curves to identify data-driven thresholds for flagging &#x201C;struggle&#x201D; episodes. Reporting convergent and discriminant validity will clarify what each metric uniquely captures and the conditions under which it is most informative.</p>
<p>Finally, the field needs evidence for causal impact. We recommend randomized A/B experiments or within-student micro-randomized trials in which hints, definitions, or worked examples are triggered by prespecified R<sup>1</sup>/R<sup>2</sup> thresholds. Primary outcomes should include next-item accuracy, time-to-mastery, and delayed retention; secondary outcomes should track false-positive/negative rates and any latency costs to ensure that supports are beneficial and efficient. Decision rules (including stopping boundaries) should be preregistered to prevent analytical flexibility.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="sec36">
<label>6</label>
<title>Conclusion</title>
<p>This exploratory study set out to recover what conventional flipped-class metrics miss: the moment-to-moment visual effort students expend while answering routine multiple-choice questions. Using a classroom-friendly 60&#x202F;Hz eye tracker and crossed mixed-effects models, we found that fixation time and regression frequency behave as complementary process signals&#x2014;longer dwell time tends to support success (OR &#x2248; 1.30 per +1 SD), whereas frequent back-tracking tends to undermine it (OR &#x2248; 0.81)&#x2014;while block-level self-reports add little diagnostic value. Critically, once total word count across stem + options (TotalWC) is entered as a covariate, the high- vs. low-Bloom difference in fixation time attenuates to non-significance, indicating that the earlier &#x201C;harder-but-shorter&#x201D; pattern is largely a surface-text effect rather than a pure demand effect. Gender introduced a small speed difference but neither predicted accuracy nor moderated the fixation&#x2013;performance link, supporting equitable interpretation of the gaze-performance association in this sample.</p>
<p>Taken together, these results reframe classroom eye tracking as a measurement-aware diagnostic: gaze metrics are informative when surface features are standardized or explicitly modeled. Practically, we recommend that future MCQ banks (i) standardize total word count in narrow bands, (ii) equalize option lengths and hold the number of options fixed, and (iii) pre-register TotalWC as a covariate in primary models. With these controls in place, fixation intensity and regression rate provide distinct, actionable cues (productive deep processing vs. struggle/inefficient re-inspection) for formative diagnostics. This positioning also clarifies the contribution of the present work: not a universal Bloom effect on gaze, but conditions under which gaze signals can be valid and useful for flipped-class assessment and for benchmarking affordable proxies (e.g., response-time distributions, option-comparison sequences).</p>
<p>A conservative next step is a &#x201C;standardized bank &#x2192; small-class pilots &#x2192; multi-class validation&#x201D; programme: replicate with length-matched, layout-matched items, expand to larger and more balanced cohorts, and test lightweight multimodal signals alongside item-level self-reports to confirm that the observed patterns are not artifacts of surface features or sampling noise. Even as a pilot, however, the workflow charts a feasible pathway from 60&#x202F;Hz gaze capture to actionable diagnostics in flipped learning, advancing response-process validity without consuming class time and pointing toward AI-assisted personalisation that complements&#x2014;rather than replaces&#x2014;human teaching.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec37">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec sec-type="ethics-statement" id="sec38">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Office of Research and Knowledge Transfer, Lingnan University. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p>
</sec>
<sec sec-type="author-contributions" id="sec39">
<title>Author contributions</title>
<p>YW: Conceptualization, Data curation, Formal analysis, Funding acquisition, Methodology, Project administration, Supervision, Validation, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. NX: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Project administration, Resources, Supervision, Validation, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. YL: Data curation, Formal analysis, Resources, Software, Visualization, Writing &#x2013; review &#x0026; editing.</p>
</sec>

<sec sec-type="COI-statement" id="sec41">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec42">
<title>Generative AI statement</title>
<p>The authors declare that Gen AI was used in the creation of this manuscript. We used ChatGPT-4o (OpenAI, May 2025 release) for limited language polishing and brainstorming of wording; all content was subsequently fact-checked, edited, and approved by the human authors.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="sec43">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec44">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/feduc.2025.1639273/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/feduc.2025.1639273/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Supplementary_file_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Abt</surname><given-names>M.</given-names></name> <name><surname>Leuders</surname><given-names>T.</given-names></name> <name><surname>Loibl</surname><given-names>K.</given-names></name> <name><surname>Strohmaier</surname><given-names>A. R.</given-names></name> <name><surname>Van Dooren</surname><given-names>W.</given-names></name> <name><surname>Reinhold</surname><given-names>F.</given-names></name></person-group> (<year>2024</year>). <article-title>How can eye-tracking data be used to understand cognitive processes when comparing data sets with box plots?</article-title> <source>Front. Educ.</source> <volume>9</volume>:<fpage>1425663</fpage>. doi: <pub-id pub-id-type="doi">10.3389/feduc.2024.1425663</pub-id></mixed-citation></ref>
<ref id="ref2"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ak&#x00E7;ay&#x0131;r</surname><given-names>G.</given-names></name> <name><surname>Ak&#x00E7;ay&#x0131;r</surname><given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>The flipped classroom: a review of its advantages and challenges</article-title>. <source>Comput. Educ.</source> <volume>126</volume>, <fpage>334</fpage>&#x2013;<lpage>345</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compedu.2018.07.021</pub-id></mixed-citation></ref>
<ref id="ref3"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Alemdag</surname><given-names>E.</given-names></name> <name><surname>Cagiltay</surname><given-names>K.</given-names></name></person-group> (<year>2018</year>). <article-title>A systematic review of eye tracking research on multimedia learning</article-title>. <source>Comput. Educ.</source> <volume>125</volume>, <fpage>413</fpage>&#x2013;<lpage>428</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compedu.2018.06.023</pub-id></mixed-citation></ref>
<ref id="ref4"><mixed-citation publication-type="book"><person-group person-group-type="author"><collab id="coll1">American Educational Research Association, American Psychological Association, &#x0026; National Council on Measurement in Education</collab></person-group> (<year>2014</year>). <source>Standards for educational and psychological testing</source>. <publisher-loc>Washington, US</publisher-loc>: <publisher-name>American Educational Research Association</publisher-name>.</mixed-citation></ref>
<ref id="ref5"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Appleton</surname><given-names>J.</given-names></name> <name><surname>Christenson</surname><given-names>S.</given-names></name> <name><surname>Kim</surname><given-names>D.</given-names></name> <name><surname>Reschly</surname><given-names>A.</given-names></name></person-group> (<year>2006</year>). <article-title>Measuring cognitive and psychological engagement: validation of the student engagement instrument</article-title>. <source>J. Sch. Psychol.</source> <volume>44</volume>, <fpage>427</fpage>&#x2013;<lpage>445</lpage>. doi: <pub-id pub-id-type="doi">10.1016/J.JSP.2006.04.002</pub-id></mixed-citation></ref>
<ref id="ref6"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Argunsah</surname><given-names>H.</given-names></name> <name><surname>Alt&#x0131;nta&#x015F;</surname><given-names>L.</given-names></name> <name><surname>&#x015E;ahiner</surname><given-names>M.</given-names></name></person-group> (<year>2025</year>). <article-title>Eye-tracking insights into cognitive strategies, learning styles, and academic outcomes of Turkish medicine students</article-title>. <source>BMC Med. Educ.</source> <volume>25</volume>:<fpage>276</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12909-025-06855-y</pub-id>, PMID: <pub-id pub-id-type="pmid">39979922</pub-id></mixed-citation></ref>
<ref id="ref7"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Barlow</surname><given-names>A.</given-names></name> <name><surname>Brown</surname><given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>Correlations between modes of student cognitive engagement and instructional practices in undergraduate STEM courses</article-title>. <source>Int. J. STEM Educ.</source> <volume>7</volume>, <fpage>1</fpage>&#x2013;<lpage>15</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s40594-020-00214-7</pub-id></mixed-citation></ref>
<ref id="ref9"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Barr</surname><given-names>D. J.</given-names></name> <name><surname>Levy</surname><given-names>R.</given-names></name> <name><surname>Scheepers</surname><given-names>C.</given-names></name> <name><surname>Tily</surname><given-names>H. J.</given-names></name></person-group> (<year>2013</year>). <article-title>Random effects structure for confirmatory hypothesis testing: keep it maximal</article-title>. <source>J. Mem. Lang.</source> <volume>68</volume>, <fpage>255</fpage>&#x2013;<lpage>278</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jml.2012.11.001</pub-id>, PMID: <pub-id pub-id-type="pmid">24403724</pub-id></mixed-citation></ref>
<ref id="ref9001"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bates</surname><given-names>D.</given-names></name> <name><surname>M&#x00E4;chler</surname><given-names>M.</given-names></name> <name><surname>Bolker</surname><given-names>B.</given-names></name> <name><surname>Walker</surname><given-names>S.</given-names></name></person-group> (<year>2015</year>). <article-title>Fitting Linear Mixed-Effects Models Using lme4</article-title>. <source>J. Stat. Softw.</source> <volume>67</volume>, <fpage>1</fpage>&#x2013;<lpage>48</lpage>. doi: <pub-id pub-id-type="doi">10.18637/jss.v067.i01</pub-id></mixed-citation></ref>
<ref id="ref10"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Beatty</surname><given-names>J.</given-names></name> <name><surname>Lucero-Wagoner</surname><given-names>B.</given-names></name></person-group> (<year>2000</year>). <article-title>The pupillary system</article-title>. In <source>Handbook of psychophysiology</source>. (eds.) <person-group person-group-type="editor"><name><surname>Cacioppo</surname><given-names>J. T.</given-names></name> <name><surname>Tassinary</surname><given-names>L. G.</given-names></name> <name><surname>Berntson</surname><given-names>G. G.</given-names></name></person-group>, (<publisher-name>Cambridge University Press</publisher-name>), 2nd ed., pp. <fpage>142</fpage>&#x2013;<lpage>162</lpage>.</mixed-citation></ref>
<ref id="ref11"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Becker</surname><given-names>S.</given-names></name> <name><surname>Knippertz</surname><given-names>L.</given-names></name> <name><surname>Ruzika</surname><given-names>S.</given-names></name> <name><surname>Kuhn</surname><given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>Persistence, context, and visual strategy of graph understanding: gaze patterns reveal student difficulties in interpreting graphs</article-title>. <source>Phys. Rev. Phys. Educ. Res.</source> <volume>19</volume>:<fpage>020142</fpage>. doi: <pub-id pub-id-type="doi">10.1103/PhysRevPhysEducRes.19.020142</pub-id></mixed-citation></ref>
<ref id="ref12"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Becker</surname><given-names>S.</given-names></name> <name><surname>K&#x00FC;chemann</surname><given-names>S.</given-names></name> <name><surname>Klein</surname><given-names>P.</given-names></name> <name><surname>Lichtenberger</surname><given-names>A.</given-names></name> <name><surname>Kuhn</surname><given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>Gaze patterns enhance response prediction: more than correct or incorrect</article-title>. <source>Phys. Rev. Phys. Educ. Res.</source> <volume>18</volume>:<fpage>020107</fpage>. doi: <pub-id pub-id-type="doi">10.1103/PhysRevPhysEducRes.18.020107</pub-id></mixed-citation></ref>
<ref id="ref13"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chi</surname><given-names>M.</given-names></name> <name><surname>Wylie</surname><given-names>R.</given-names></name></person-group> (<year>2014</year>). <article-title>The ICAP framework: linking cognitive engagement to active learning outcomes</article-title>. <source>Educ. Psychol.</source> <volume>49</volume>, <fpage>219</fpage>&#x2013;<lpage>243</lpage>. doi: <pub-id pub-id-type="doi">10.1080/00461520.2014.965823</pub-id></mixed-citation></ref>
<ref id="ref14"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>D&#x2019;Mello</surname><given-names>S.</given-names></name> <name><surname>Dieterle</surname><given-names>E.</given-names></name> <name><surname>Duckworth</surname><given-names>A.</given-names></name></person-group> (<year>2017</year>). <article-title>Advanced, analytic, automated (AAA) measurement of engagement during learning</article-title>. <source>Educ. Psychol.</source> <volume>52</volume>, <fpage>104</fpage>&#x2013;<lpage>123</lpage>. doi: <pub-id pub-id-type="doi">10.1080/00461520.2017.1281747</pub-id></mixed-citation></ref>
<ref id="ref15"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fehlinger</surname><given-names>P.</given-names></name> <name><surname>Becker-Genschow</surname><given-names>S.</given-names></name> <name><surname>Watzka</surname><given-names>B.</given-names></name></person-group> (<year>2025</year>). <article-title>Gaze behavior as a key to revealing strategies for identifying indirectly proportional graphs in thermodynamic and mathematical context</article-title>. <source>Phys. Rev. Phys. Educ. Res.</source> <volume>21</volume>:<fpage>020129</fpage>. doi: <pub-id pub-id-type="doi">10.1103/4pn3-fs4y</pub-id></mixed-citation></ref>
<ref id="ref16"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Fredricks</surname><given-names>J. A.</given-names></name> <name><surname>Hofkens</surname><given-names>T. L.</given-names></name> <name><surname>Wang</surname><given-names>M.-T.</given-names></name> <name><surname>Renninger</surname><given-names>K. A.</given-names></name> <name><surname>Hidi</surname><given-names>S. E.</given-names></name></person-group> (<year>2019</year>). <article-title>Addressing the challenge of measuring student engagement</article-title>. In <source>The Cambridge handbook of motivation and learning</source> (pp. <fpage>689</fpage>&#x2013;<lpage>712</lpage>). Chapter, <publisher-loc>Cambridge</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>.</mixed-citation></ref>
<ref id="ref17"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fuller</surname><given-names>K.</given-names></name> <name><surname>Karunaratne</surname><given-names>N.</given-names></name> <name><surname>Naidu</surname><given-names>S.</given-names></name> <name><surname>Exintaris</surname><given-names>B.</given-names></name> <name><surname>Short</surname><given-names>J.</given-names></name> <name><surname>Wolcott</surname><given-names>M.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Development of a self-report instrument for measuring in-class student engagement reveals that pretending to engage is a significant, unrecognized problem</article-title>. <source>PLoS One</source> <volume>13</volume>:<fpage>e0205828</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0205828</pub-id>, PMID: <pub-id pub-id-type="pmid">30332460</pub-id></mixed-citation></ref>
<ref id="ref18"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gabel</surname><given-names>S.</given-names></name> <name><surname>Alijagic</surname><given-names>A.</given-names></name> <name><surname>Keskin</surname><given-names>&#x00D6;.</given-names></name> <name><surname>Gegenfurtner</surname><given-names>A.</given-names></name></person-group> (<year>2025</year>). <article-title>Teacher gaze and attitudes toward student gender: evidence from eye tracking and implicit association tests</article-title>. <source>Soc. Psychol. Educ.</source> <volume>28</volume>:<fpage>72</fpage>. doi: <pub-id pub-id-type="doi">10.1007/s11218-025-10036-6</pub-id></mixed-citation></ref>
<ref id="ref19"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gijsen</surname><given-names>M.</given-names></name> <name><surname>Catrysse</surname><given-names>L.</given-names></name> <name><surname>De Maeyer</surname><given-names>S.</given-names></name> <name><surname>Gijbels</surname><given-names>D.</given-names></name></person-group> (<year>2024</year>). <article-title>Mapping cognitive processes in video-based learning by combining trace and think-aloud data</article-title>. <source>Learn. Instr.</source> <volume>90</volume>:<fpage>101851</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.learninstruc.2023.101851</pub-id></mixed-citation></ref>
<ref id="ref20"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Graesser</surname><given-names>A. C.</given-names></name> <name><surname>Conley</surname><given-names>M. W.</given-names></name> <name><surname>Olney</surname><given-names>A.</given-names></name></person-group> (<year>2012</year>). &#x201C;<article-title>Intelligent tutoring systems</article-title>&#x201D; in <source>APA educational psychology handbook, Vol. 3. Application to learning and teaching</source>. eds. <person-group person-group-type="editor"><name><surname>Harris</surname><given-names>K. R.</given-names></name> <name><surname>Graham</surname><given-names>S.</given-names></name> <name><surname>Urdan</surname><given-names>T.</given-names></name> <name><surname>Bus</surname><given-names>A. G.</given-names></name> <name><surname>Major</surname><given-names>S.</given-names></name> <name><surname>Swanson</surname><given-names>H. L.</given-names></name></person-group> (<publisher-loc>Washington, DC</publisher-loc>: <publisher-name>American Psychological Association</publisher-name>), <fpage>451</fpage>&#x2013;<lpage>473</lpage>.</mixed-citation></ref>
<ref id="ref21"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Han</surname><given-names>F.</given-names></name></person-group> (<year>2023</year>). <article-title>Relations between students&#x2019; study approaches, perceptions of the learning environment, and academic achievement in flipped classroom learning: evidence from self-reported and process data</article-title>. <source>J. Educ. Comput. Res.</source> <volume>61</volume>, <fpage>1252</fpage>&#x2013;<lpage>1274</lpage>. doi: <pub-id pub-id-type="doi">10.1177/07356331231162823</pub-id></mixed-citation></ref>
<ref id="ref22"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hew</surname><given-names>K. F.</given-names></name> <name><surname>Bai</surname><given-names>S.</given-names></name> <name><surname>Dawson</surname><given-names>P.</given-names></name> <name><surname>Lo</surname><given-names>C. K.</given-names></name></person-group> (<year>2021</year>). <article-title>Meta-analyses of flipped classroom studies: a review of methodology</article-title>. <source>Educ. Res. Rev.</source> <volume>33</volume>:<fpage>100393</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.edurev.2021.100393</pub-id></mixed-citation></ref>
<ref id="ref23"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname><given-names>P. S.</given-names></name> <name><surname>Chen</surname><given-names>H. C.</given-names></name></person-group> (<year>2016</year>). <article-title>Gender differences in eye movements in solving text-and-diagram science problems</article-title>. <source>Int. J. Sci. Math. Educ.</source> <volume>14</volume>, <fpage>327</fpage>&#x2013;<lpage>346</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10763-015-9644-3</pub-id></mixed-citation></ref>
<ref id="ref24"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hy&#x00F6;n&#x00E4;</surname><given-names>J.</given-names></name></person-group> (<year>2010</year>). <article-title>The use of eye movements in the study of multimedia learning</article-title>. <source>Learn. Instr.</source> <volume>20</volume>, <fpage>172</fpage>&#x2013;<lpage>176</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.learninstruc.2009.02.013</pub-id></mixed-citation></ref>
<ref id="ref25"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Inhoff</surname><given-names>A.</given-names></name> <name><surname>Kim</surname><given-names>A.</given-names></name> <name><surname>Radach</surname><given-names>R.</given-names></name></person-group> (<year>2019</year>). <article-title>Regressions during reading</article-title>. <source>Vision</source> <volume>3</volume>:<fpage>35</fpage>. doi: <pub-id pub-id-type="doi">10.3390/vision3030035</pub-id>, PMID: <pub-id pub-id-type="pmid">31735836</pub-id></mixed-citation></ref>
<ref id="ref27"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Just</surname><given-names>M. A.</given-names></name> <name><surname>Carpenter</surname><given-names>P. A.</given-names></name></person-group> (<year>1976</year>). <article-title>Eye fixations and cognitive processes</article-title>. <source>Cogn. Psychol.</source> <volume>8</volume>, <fpage>441</fpage>&#x2013;<lpage>480</lpage>. doi: <pub-id pub-id-type="doi">10.1016/0010-0285(76)90015-3</pub-id></mixed-citation></ref>
<ref id="ref28"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Kim</surname><given-names>Y.</given-names></name> <name><surname>Ademola</surname><given-names>A.</given-names></name> <name><surname>Ko</surname><given-names>J.</given-names></name> <name><surname>Kim</surname><given-names>H.</given-names></name></person-group> (<year>2022</year>). <source>Knuir at the ntcir-16 rcir: Predicting comprehension level using regression models based on eye-tracking metadata. Proceedings of the 16th NTCIR Conference on Evaluation of Information Access Technologies (NTCIR-16). Tokyo, Japan</source>.</mixed-citation></ref>
<ref id="ref29"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Krathwohl</surname><given-names>D. R.</given-names></name></person-group> (<year>2002</year>). <article-title>A revision of bloom&#x2019;s taxonomy: an overview</article-title>. <source>Theory Into Pract.</source> <volume>41</volume>, <fpage>212</fpage>&#x2013;<lpage>218</lpage>. doi: <pub-id pub-id-type="doi">10.1207/s15430421tip4104_2</pub-id></mixed-citation></ref>
<ref id="ref9003"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kuznetsova</surname><given-names>A.</given-names></name> <name><surname>Brockhoff</surname><given-names>P. B.</given-names></name> <name><surname>Christensen</surname><given-names>R. H. B.</given-names></name></person-group> (<year>2017</year>). <article-title>lmerTest Package: Tests in Linear Mixed Effects Models</article-title>. <source>J. Stat. Softw.</source> <volume>82</volume>, <fpage>1</fpage>&#x2013;<lpage>26</lpage>. doi: <pub-id pub-id-type="doi">10.18637/jss.v082.i13</pub-id></mixed-citation></ref>
<ref id="ref30"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lai</surname><given-names>M. L.</given-names></name> <name><surname>Tsai</surname><given-names>M. J.</given-names></name> <name><surname>Yang</surname><given-names>F. Y.</given-names></name> <name><surname>Hsu</surname><given-names>C. Y.</given-names></name> <name><surname>Liu</surname><given-names>T. C.</given-names></name> <name><surname>Lee</surname><given-names>S. W. Y.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>A review of using eye-tracking technology in exploring learning from 2000 to 2012</article-title>. <source>Educ. Res. Rev.</source> <volume>10</volume>, <fpage>90</fpage>&#x2013;<lpage>115</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.edurev.2013.10.001</pub-id></mixed-citation></ref>
<ref id="ref31"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname><given-names>J.</given-names></name> <name><surname>Soleimani</surname><given-names>F.</given-names></name> <name><surname>Hosmer</surname><given-names>I. V. J.</given-names></name> <name><surname>Soylu</surname><given-names>M. Y.</given-names></name> <name><surname>Finkelberg</surname><given-names>R.</given-names></name> <name><surname>Chatterjee</surname><given-names>S.</given-names></name></person-group> (<year>2022</year>). <article-title>Predicting cognitive presence in at-scale online learning: MOOC and for-credit online course environments</article-title>. <source>Online Learn.</source> <volume>26</volume>, <fpage>58</fpage>&#x2013;<lpage>79</lpage>. doi: <pub-id pub-id-type="doi">10.24059/olj.v26i1.3060</pub-id></mixed-citation></ref>
<ref id="ref32"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lei</surname><given-names>H.</given-names></name> <name><surname>Cui</surname><given-names>Y.</given-names></name> <name><surname>Zhou</surname><given-names>W.</given-names></name></person-group> (<year>2018</year>). <article-title>Relationships between student engagement and academic achievement: a meta-analysis</article-title>. <source>Soc. Behav. Personal.</source> <volume>46</volume>, <fpage>517</fpage>&#x2013;<lpage>528</lpage>. doi: <pub-id pub-id-type="doi">10.2224/sbp.7054</pub-id></mixed-citation></ref>
<ref id="ref33"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>S.</given-names></name> <name><surname>Liu</surname><given-names>S.</given-names></name> <name><surname>Liu</surname><given-names>Z.</given-names></name> <name><surname>Peng</surname><given-names>X.</given-names></name> <name><surname>Yang</surname><given-names>Z.</given-names></name></person-group> (<year>2022</year>). <article-title>Automated detection of emotional and cognitive engagement in MOOC discussions to predict learning achievement</article-title>. <source>Comput. Educ.</source> <volume>181</volume>:<fpage>104461</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compedu.2022.104461</pub-id></mixed-citation></ref>
<ref id="ref34"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lundin</surname><given-names>M.</given-names></name> <name><surname>Bergviken Rensfeldt</surname><given-names>A.</given-names></name> <name><surname>Hillman</surname><given-names>T.</given-names></name> <name><surname>Lantz-Andersson</surname><given-names>A.</given-names></name> <name><surname>Peterson</surname><given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>Higher education dominance and siloed knowledge: a systematic review of flipped classroom research</article-title>. <source>Int J Educ. Technol. Higher Educ.</source> <volume>15</volume>:<fpage>20</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s41239-018-0101-6</pub-id></mixed-citation></ref>
<ref id="ref35"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname><given-names>H.</given-names></name> <name><surname>Huang</surname><given-names>Z.</given-names></name> <name><surname>Tang</surname><given-names>W.</given-names></name> <name><surname>Zhu</surname><given-names>H.</given-names></name> <name><surname>Zhang</surname><given-names>H.</given-names></name> <name><surname>Li</surname><given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>Predicting student performance in future exams via neutrosophic cognitive diagnosis in personalized E-learning environment</article-title>. <source>IEEE Trans. Learn. Technol.</source> <volume>16</volume>, <fpage>680</fpage>&#x2013;<lpage>693</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TLT.2023.3240931</pub-id></mixed-citation></ref>
<ref id="ref36"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname><given-names>X.</given-names></name> <name><surname>Liu</surname><given-names>Y.</given-names></name> <name><surname>Clariana</surname><given-names>R.</given-names></name> <name><surname>Gu</surname><given-names>C.</given-names></name> <name><surname>Li</surname><given-names>P.</given-names></name></person-group> (<year>2023</year>). <article-title>From eye movements to scan-path networks: a method for studying individual differences in expository text reading</article-title>. <source>Behav. Res. Methods</source> <volume>55</volume>, <fpage>730</fpage>&#x2013;<lpage>750</lpage>. doi: <pub-id pub-id-type="doi">10.3758/s13428-022-01842-3</pub-id>, PMID: <pub-id pub-id-type="pmid">35445941</pub-id></mixed-citation></ref>
<ref id="ref37"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Man</surname><given-names>K.</given-names></name> <name><surname>Harring</surname><given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>Negative binomial models for visual fixation counts on test items</article-title>. <source>Educ. Psychol. Meas.</source> <volume>79</volume>, <fpage>617</fpage>&#x2013;<lpage>635</lpage>. doi: <pub-id pub-id-type="doi">10.1177/0013164418824148</pub-id>, PMID: <pub-id pub-id-type="pmid">32655176</pub-id></mixed-citation></ref>
<ref id="ref38"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>&#x00D6;zdemir</surname><given-names>&#x015E;.</given-names></name> <name><surname>Tosun</surname><given-names>C.</given-names></name></person-group> (<year>2025</year>). <article-title>Investigation of eighth-grade students&#x2019; processes of solving skill-based science questions by eye tracking technique</article-title>. <source>Educ. Inf. Technol.</source> <volume>30</volume>, <fpage>2237</fpage>&#x2013;<lpage>2275</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10639-024-12841-6</pub-id></mixed-citation></ref>
<ref id="ref39"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>&#x00D6;zek</surname><given-names>F.</given-names></name> <name><surname>Fer</surname><given-names>S.</given-names></name></person-group> (<year>2025</year>). <article-title>Pscp cognitive engagement scale: a scale development study</article-title>. <source>Educ. Academic Res.</source> <fpage>80</fpage>&#x2013;<lpage>90</lpage>. doi: <pub-id pub-id-type="doi">10.33418/education.1527281</pub-id></mixed-citation></ref>
<ref id="ref40"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Paas</surname><given-names>F. G.</given-names></name> <name><surname>Van Merri&#x00EB;nboer</surname><given-names>J. J.</given-names></name></person-group> (<year>1994</year>). <article-title>Instructional control of cognitive load in the training of complex cognitive tasks</article-title>. <source>Educ. Psychol. Rev.</source> <volume>6</volume>, <fpage>351</fpage>&#x2013;<lpage>371</lpage>. doi: <pub-id pub-id-type="doi">10.1037/0022-0663.86.1.122</pub-id></mixed-citation></ref>
<ref id="ref41"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Paskovske</surname><given-names>A.</given-names></name> <name><surname>Klizien&#x0117;</surname><given-names>I.</given-names></name></person-group> (<year>2024</year>). <article-title>Eye tracking technology on children&#x2019;s mathematical education: systematic review</article-title>. <source>Front. Educ.</source> <volume>9</volume>:<fpage>Article 1386487</fpage>. doi: <pub-id pub-id-type="doi">10.3389/feduc.2024.1386487</pub-id></mixed-citation></ref>
<ref id="ref42"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pintrich</surname><given-names>P. R.</given-names></name></person-group> (<year>2004</year>). <article-title>A conceptual framework for assessing motivation and self-regulated learning</article-title>. <source>Educ. Psychol. Rev.</source> <volume>16</volume>, <fpage>385</fpage>&#x2013;<lpage>407</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10648-004-0006-x</pub-id></mixed-citation></ref>
<ref id="ref43"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Snijders</surname><given-names>T. A. B.</given-names></name> <name><surname>Bosker</surname><given-names>R. J.</given-names></name></person-group> (<year>2012</year>). <source>Multilevel analysis: An introduction to basic and advanced multilevel modeling</source>. <edition>2nd</edition> Edn. <publisher-loc>London</publisher-loc>: <publisher-name>Sage Publications</publisher-name>.</mixed-citation></ref>
<ref id="ref44"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Spichtig</surname><given-names>A. N.</given-names></name> <name><surname>Pascoe</surname><given-names>J. P.</given-names></name> <name><surname>Ferrara</surname><given-names>J. D.</given-names></name> <name><surname>Vorstius</surname><given-names>C.</given-names></name></person-group> (<year>2017</year>). <article-title>A comparison of eye movement measures across Reading efficiency quartile groups in elementary, middle, and high school students in the U.S</article-title>. <source>J. Eye Mov. Res.</source> <volume>10</volume>. doi: <pub-id pub-id-type="doi">10.16910/jemr.10.4.5</pub-id>, PMID: <pub-id pub-id-type="pmid">33828663</pub-id></mixed-citation></ref>
<ref id="ref45"><mixed-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Tehranchi</surname><given-names>F.</given-names></name> <name><surname>Ritter</surname><given-names>F. E.</given-names></name> <name><surname>Chae</surname><given-names>C.</given-names></name></person-group> (<year>2020</year>). <conf-name>Visual attention during e-learning: eye-tracking shows that making salient areas more prominent helps learning in online tutors. Proceedings of the 42nd Annual Meeting of the Cognitive Science Society</conf-name>, <fpage>3165</fpage>&#x2013;<lpage>3171</lpage></mixed-citation></ref>
<ref id="ref46"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>van Gog</surname><given-names>T.</given-names></name> <name><surname>Jarodzka</surname><given-names>H.</given-names></name></person-group> (<year>2013</year>). &#x201C;<article-title>Eye tracking as a tool to study and enhance cognitive and metacognitive processes in computer-based learning environments</article-title>&#x201D; in <source>International handbook of metacognition and learning technologies (pp. 143&#x2013;156)</source>. eds. <person-group person-group-type="editor"><name><surname>Azevedo</surname><given-names>R.</given-names></name> <name><surname>Aleven</surname><given-names>V.</given-names></name></person-group> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Springer New York</publisher-name>).</mixed-citation></ref>
<ref id="ref9004"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Wickham</surname><given-names>H.</given-names></name> <name><surname>Fran&#x00E7;ois</surname><given-names>R.</given-names></name> <name><surname>Henry</surname><given-names>L.</given-names></name> <name><surname>M&#x00FC;ller</surname><given-names>K.</given-names></name> <name><surname>Vaughan</surname><given-names>D.</given-names></name></person-group> (<year>2023</year>). <article-title>dplyr: A grammar of data manipulation (Version 1.1.2)</article-title>. <ext-link xlink:href="https://dplyr.tidyverse.org" ext-link-type="uri">https://dplyr.tidyverse.org</ext-link></mixed-citation></ref>
<ref id="ref47"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname><given-names>Q.</given-names></name> <name><surname>Wei</surname><given-names>Y.</given-names></name> <name><surname>Gao</surname><given-names>J.</given-names></name> <name><surname>Yao</surname><given-names>H.</given-names></name> <name><surname>Liu</surname><given-names>Q.</given-names></name></person-group> (<year>2023</year>). <article-title>ICAPD framework and simAM-YOLOv8n for student cognitive engagement detection in the classroom</article-title>. <source>IEEE Access</source> <volume>11</volume>, <fpage>136063</fpage>&#x2013;<lpage>136076</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ACCESS.2023.3337435</pub-id></mixed-citation></ref>
<ref id="ref48"><mixed-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Yan</surname><given-names>J.</given-names></name> <name><surname>Lv</surname><given-names>H.</given-names></name></person-group> (<year>2023</year>). <conf-name>The development of the flipped learning student engagement scale. In 2023 4th international conference on big data and Informatization education (ICBDIE 2023)</conf-name> (pp. <fpage>631</fpage>&#x2013;<lpage>646</lpage>). <publisher-name>Atlantis Press</publisher-name>.</mixed-citation></ref>
<ref id="ref49"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zainuddin</surname><given-names>Z.</given-names></name> <name><surname>Halili</surname><given-names>S. H.</given-names></name></person-group> (<year>2016</year>). <article-title>Flipped classroom research and trends from different fields of study</article-title>. <source>Int. Rev. Res. Open Distributed Learn.</source> <volume>17</volume>, <fpage>313</fpage>&#x2013;<lpage>340</lpage>. doi: <pub-id pub-id-type="doi">10.19173/irrodl.v17i3.2274</pub-id></mixed-citation></ref>
</ref-list><fn-group><fn id="fn0001" fn-type="custom" custom-type="edited-by"><p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/414396/overview">Yan Liu</ext-link>, Carleton University, Canada</p></fn>
<fn id="fn0002" fn-type="custom" custom-type="reviewed-by"><p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/589245/overview">Stefan Daniel Keller</ext-link>, University of Teacher Education Zuerich, Switzerland; <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1976156/overview">Sebastian Becker-Genschow</ext-link>, University of Cologne, Germany; <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3164784/overview">Neila Chettaoui</ext-link>, National Engineering School of Sfax, Tunisia</p></fn></fn-group></back>
</article>