<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2025.1540539</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Exploring the EEG representation of English listening comprehension under hypoxic conditions</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Song</surname> <given-names>Yanhui</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Yu</surname> <given-names>Ye</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2915266/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Pingdingshan University</institution>, <addr-line>Pingdingshan</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Dazhou Vocational and Technical College, Dazhou</institution>, <addr-line>Sichuan</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: David Crist&#x000F3;bal Andrade, University of Antofagasta, Chile</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Jianfei Fu, Tongji Hospital Affiliated to Tongji University, China</p>
<p>Dhriti Majumder, Alliance University, India</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Yanhui Song <email>isbo1s&#x00040;163.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>18</day>
<month>06</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>19</volume>
<elocation-id>1540539</elocation-id>
<history>
<date date-type="received">
<day>06</day>
<month>12</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>22</day>
<month>05</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Song and Yu.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Song and Yu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Understanding the impact of hypoxic conditions on cognitive functions, including English listening comprehension, has garnered increasing attention due to its implications for high-altitude education and cognitive resilience. Traditional research in this domain has often relied on behavioral assessments or simple physiological metrics, which lack the granularity to capture the neural underpinnings of cognitive performance.</p></sec>
<sec>
<title>Methods</title>
<p>This study proposes a novel framework combining electroencephalography (EEG)-based neural decoding with the Dynamic Linguistic Enhancement Model (DLEM) to investigate English listening comprehension in hypoxic environments. DLEM integrates adaptive vocabulary acquisition, grammar contextualization, and cultural embedding, leveraging EEG to provide real-time, personalized insights into linguistic processing.</p></sec>
<sec>
<title>Results</title>
<p>The experimental results demonstrate significant improvements in comprehension accuracy and cognitive load management, particularly under adaptive curriculum strategies outlined by the Contextual Augmented Learning Strategy (CALS).</p></sec>
<sec>
<title>Discussion</title>
<p>By bridging physiological responses with advanced educational methodologies, this work contributes a scalable and flexible approach to enhancing cognitive performance under hypoxia, aligning with the goals of understanding both physiological and pathological responses to high-altitude conditions.</p></sec></abstract>
<kwd-group>
<kwd>hypoxia</kwd>
<kwd>EEG</kwd>
<kwd>English comprehension</kwd>
<kwd>cognitive modeling</kwd>
<kwd>high-altitude learning</kwd>
</kwd-group>
<counts>
<fig-count count="8"/>
<table-count count="5"/>
<equation-count count="43"/>
<ref-count count="46"/>
<page-count count="16"/>
<word-count count="10347"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Autonomic Neuroscience</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Understanding how hypoxic conditions affect cognitive processes, such as English listening comprehension, is crucial due to its implications in environments like aviation, deep-sea diving, and medical conditions (Agung and Surtikanti, <xref ref-type="bibr" rid="B1">2020</xref>). This research area not only advances theoretical insights into neural mechanisms but also contributes to developing adaptive systems for individuals working under such conditions (Liang, <xref ref-type="bibr" rid="B21">2021</xref>). Electroencephalography (EEG) has emerged as a vital tool for studying real-time brain activity during cognitive tasks (Yao and Ma, <xref ref-type="bibr" rid="B40">2021</xref>). By examining EEG patterns, researchers can identify the specific neural correlates and disruptions caused by hypoxia (Hu and Yao, <xref ref-type="bibr" rid="B14">2021</xref>). Such investigations provide opportunities for designing mitigation strategies, enhancing cognitive resilience in hypoxic scenarios, and improving human performance in extreme environments. This review outlines the evolution of methods in this field, highlighting limitations and opportunities across three major methodological phases (Kashinathan and Aziz, <xref ref-type="bibr" rid="B19">2021</xref>).</p>
<p>Early studies of EEG in cognitive tasks under hypoxic conditions relied heavily on traditional symbolic AI and knowledge-based approaches to model human cognition (Zheng et al., <xref ref-type="bibr" rid="B45">2021</xref>). These methods focused on understanding predefined patterns and relationships, often using rule-based systems to interpret EEG data (Chen, <xref ref-type="bibr" rid="B8">2021</xref>) Knowledge representation frameworks were employed to classify brainwave patterns associated with various cognitive states, including attentional focus and memory retention (Lee and Hwang, <xref ref-type="bibr" rid="B20">2022</xref>). While these approaches laid foundational insights into neural mechanisms, they were constrained by the rigidity of predefined rules and limited capacity to account for the dynamic nature of cognitive processes under stressors like hypoxia (Sun et al., <xref ref-type="bibr" rid="B34">2020</xref>). Moreover, manual curation of EEG features was labor-intensive and failed to capture subtle temporal variations, reducing their practical applicability to real-world scenarios (Richards and Pun, <xref ref-type="bibr" rid="B25">2021</xref>). The emergence of data-driven and machine learning techniques addressed many limitations of symbolic methods by enabling automated feature extraction and adaptive modeling (Sihn and Kim, <xref ref-type="bibr" rid="B31">2022</xref>). Researchers began employing classifiers such as support vector machines (SVM) and random forests to distinguish EEG patterns corresponding to varying degrees of hypoxia (Hendriks-Balk et al., <xref ref-type="bibr" rid="B13">2020</xref>). Machine learning models facilitated the identification of nuanced EEG biomarkers of cognitive degradation, offering greater flexibility and scalability (Karlen-Amarante et al., <xref ref-type="bibr" rid="B18">2024</xref>). However, these methods were often dependent on extensive labeled datasets and were sensitive to noise inherent in EEG recordings (Iturriaga et al., <xref ref-type="bibr" rid="B17">2023</xref>). Furthermore, the lack of interpretability in these models posed challenges in understanding the underlying neurophysiological processes and tailoring interventions (Iturriaga and Castillo-Gal&#x000E1;n, <xref ref-type="bibr" rid="B16">2022</xref>).</p>
<p>In recent years, deep learning and pre-trained models have revolutionized EEG analysis in hypoxic cognitive research. Convolutional neural networks (CNNs) and recurrent neural networks (RNNs) have been utilized to capture spatiotemporal dynamics of EEG signals, while transformer-based models leverage self-attention mechanisms for contextual encoding of brain activity. These methods outperform traditional machine learning models in accuracy and robustness, especially in handling complex, high-dimensional EEG data. Pre-trained models fine-tuned on task-specific datasets further enhance transfer learning, enabling cross-population studies. Despite their promise, these approaches often require significant computational resources and large-scale datasets, which may not always be feasible in hypoxic research. The opacity of deep models raises concerns about the interpretability and generalizability of findings. To overcome these limitations, we propose a novel approach that integrates the interpretability of symbolic methods, the adaptability of machine learning, and the sophistication of deep learning models. By leveraging a hybrid architecture, our method aims to provide high accuracy in EEG analysis while maintaining computational efficiency and neurophysiological relevance. This approach aligns with the need for robust, scalable, and interpretable solutions in understanding English listening comprehension under hypoxic conditions.</p>
<p>We summarize our contributions as follows:</p>
<list list-type="bullet">
<list-item><p>Introduces a hybrid architecture combining symbolic reasoning with deep learning for enhanced interpretability and accuracy.</p></list-item>
<list-item><p>Designed for multi-scenario adaptability, including varying hypoxic levels, ensuring high efficiency and generalizability.</p></list-item>
<list-item><p>Demonstrates superior performance in decoding EEG representations, with statistically significant improvements in accuracy and robustness.</p></list-item>
</list></sec>
<sec id="s2">
<title>2 Related work</title>
<sec>
<title>2.1 EEG analysis in language comprehension</title>
<p>The study of electroencephalography (EEG) has become a cornerstone in understanding cognitive processes, including language comprehension (Ariastuti and Wahyudin, <xref ref-type="bibr" rid="B5">2022</xref>). EEG allows researchers to measure brain activity with high temporal resolution, enabling the examination of neural responses to linguistic stimuli (Wu et al., <xref ref-type="bibr" rid="B39">2022</xref>). A significant body of research focuses on the temporal dynamics of event-related potentials (ERPs) during language processing tasks (Zou et al., <xref ref-type="bibr" rid="B46">2021</xref>). For example, the N400 component is widely studied for its role in semantic processing, revealing insights into how the brain resolves meaning inconsistencies (Coleman, <xref ref-type="bibr" rid="B9">2021</xref>). Similarly, the P600 component has been linked to syntactic processing and reanalysis during sentence comprehension (Aoyama, <xref ref-type="bibr" rid="B4">2021</xref>). These findings underscore the importance of EEG in mapping the temporal stages of language comprehension (Elliott and Hodgson, <xref ref-type="bibr" rid="B11">2021</xref>). Furthermore, frequency-based EEG analyses, such as alpha and theta power modulations, have been explored to understand attentional and memory mechanisms during language tasks (Zhao et al., <xref ref-type="bibr" rid="B44">2020</xref>). While these studies provide a robust foundation, the impact of external factors, such as environmental stressors or altered physiological conditions like hypoxia, remains less understood (Simamora and Oktaviani, <xref ref-type="bibr" rid="B32">2020</xref>). Examining how hypoxia modulates these EEG markers could reveal how adverse conditions affect language processing (Iturriaga, <xref ref-type="bibr" rid="B15">2023</xref>).</p>
</sec>
<sec>
<title>2.2 Cognitive impairment under hypoxia</title>
<p>Hypoxia, characterized by reduced oxygen availability, has profound effects on brain function, including cognitive and linguistic abilities (Bae and Park, <xref ref-type="bibr" rid="B6">2020</xref>). Existing research highlights how hypoxia impacts attention, memory, and executive function (Yunita and Maisarah, <xref ref-type="bibr" rid="B41">2020</xref>). Studies employing neuroimaging and behavioral assessments have demonstrated significant cognitive deficits under acute and chronic hypoxic conditions (Septiyanti et al., <xref ref-type="bibr" rid="B29">2020</xref>). These include slower reaction times, decreased working memory capacity, and impaired decision-making. Despite these findings, there is a notable gap in the literature concerning hypoxia&#x00027;s effects on specific cognitive domains such as language comprehension (Seo, <xref ref-type="bibr" rid="B28">2020</xref>). Investigating this relationship is essential, as language comprehension relies on the integration of multiple cognitive resources, including attention and working memory. Hypoxia-induced changes in brain physiology, such as reduced cerebral oxygenation and altered neurotransmitter dynamics, may disrupt these processes (Rusmiyanto et al., <xref ref-type="bibr" rid="B26">2023</xref>). EEG studies could provide valuable insights by identifying how hypoxia modulates neural correlates of language comprehension, such as ERP components and oscillatory activity patterns (Alfallaj et al., <xref ref-type="bibr" rid="B3">2021</xref>).</p>
</sec>
<sec>
<title>2.3 Multimodal interaction of stressors</title>
<p>The interaction of hypoxia with other stressors, such as cognitive load or emotional stress, presents a complex challenge to understanding brain function (Sallam, <xref ref-type="bibr" rid="B27">2023</xref>). Multimodal studies investigating combined stress effects are relatively sparse, yet crucial for understanding real-world scenarios (Zein et al., <xref ref-type="bibr" rid="B42">2020</xref>). Cognitive tasks, such as language comprehension, often occur under conditions involving multiple concurrent demands. The interplay between hypoxia and additional stressors may exacerbate neural inefficiencies, leading to amplified cognitive deficits (Shaikh et al., <xref ref-type="bibr" rid="B30">2023</xref>). Research utilizing EEG has shown that stressors such as mental fatigue or anxiety can modulate brainwave patterns, particularly in the alpha and beta frequency bands (Renganathan, <xref ref-type="bibr" rid="B24">2021</xref>). The integration of EEG with other physiological measures, such as heart rate variability or blood oxygen saturation, could provide a holistic view of how hypoxia interacts with stress (Syakur et al., <xref ref-type="bibr" rid="B35">2020</xref>). Moreover, advanced analytical techniques, such as machine learning models, could be employed to decode complex neural patterns arising from multimodal stress conditions (Sofyan, <xref ref-type="bibr" rid="B33">2021</xref>). This direction not only addresses theoretical questions but also has practical implications for environments where individuals face simultaneous cognitive and physiological challenges.</p></sec>
</sec>
<sec id="s3">
<title>3 Method</title>
<sec>
<title>3.1 Overview</title>
<p>In recent years, English education has become a critical area of focus due to its global significance in academic, professional, and social contexts. This subsection provides an overview of the methodology employed to enhance English learning outcomes, particularly in environments where English is taught as a second language (ESL). We aim to tackle challenges in comprehension, expression, and fluency through the integration of novel pedagogical strategies, leveraging technological advancements, and understanding linguistic nuances.</p>
<p>The upcoming subsections will address various facets of our approach. In Section 3.2, we formalize the problem of English education by analyzing common linguistic barriers and presenting a structured framework to model them. This foundational section establishes the key challenges in vocabulary acquisition, grammar comprehension, and cultural fluency, emphasizing their interconnected nature. In Section 3.3, we introduce a novel framework, hereafter referred to as the Dynamic Linguistic Enhancement Model (DLEM). This model builds upon insights from cognitive science and language processing to deliver adaptive and personalized learning pathways. Key components of DLEM include contextualized learning environments and multi-modal interactions, which are meticulously designed to simulate real-world communication. In Section 3.4, we propose an innovative strategy, termed the Contextual Augmented Learning Strategy (CALS), to integrate our model effectively into diverse educational settings. This strategy focuses on adaptive curriculum design, dynamic feedback systems, and the utilization of gamification to foster learner engagement and motivation. The emphasis is on scalability and flexibility, ensuring applicability across varied cultural and institutional contexts.</p>
</sec>
<sec>
<title>3.2 Preliminaries</title>
<p>English education, particularly in environments where it is taught as a second language, presents unique challenges that require careful analysis and systematic formalization. To address these challenges, we introduce a mathematical and conceptual framework that captures the complexities of language acquisition, comprehension, and usage. The English learning process can be represented as a multi-stage system:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="script"><mml:mi>S</mml:mi></mml:mstyle><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>V</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="script"><mml:mi>G</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In this framework, <inline-formula><mml:math id="M50"><mml:mstyle mathvariant="script"><mml:mi>V</mml:mi></mml:mstyle></mml:math></inline-formula> represents the domain of vocabulary acquisition, encompassing the process of learning and retaining new words. <inline-formula><mml:math id="M51"><mml:mstyle mathvariant="script"><mml:mi>G</mml:mi></mml:mstyle></mml:math></inline-formula> denotes the grammatical structures of the language, including syntax and morphology, which govern sentence construction. <inline-formula><mml:math id="M52"><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:math></inline-formula> embodies the cultural and contextual understanding that is essential for meaningful and effective language use. These components interact dynamically within the cognitive capabilities of the learner and the environmental influences they encounter, creating a complex and interdependent system. This framework provides a structured approach to understanding and addressing the multifaceted nature of English language learning. The vocabulary learning process can be modeled as a stochastic process, where the probability of acquiring a word <italic>w</italic><sub><italic>i</italic></sub> at time <italic>t</italic> is dependent on exposure <italic>E</italic>(<italic>w</italic><sub><italic>i</italic></sub>, <italic>t</italic>) and reinforcement <italic>R</italic>(<italic>w</italic><sub><italic>i</italic></sub>, <italic>t</italic>). Formally:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>E</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mi>R</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>f</italic> is a monotonic function that combines exposure and reinforcement effects. Reinforcement often depends on the frequency and utility of <italic>w</italic><sub><italic>i</italic></sub> in specific contexts:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>R</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mtext>freq</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B2;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mtext>utility</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>with &#x003B1; and &#x003B2; as tunable parameters representing learner-specific sensitivity. Grammar is structured around a set of syntactic rules <inline-formula><mml:math id="M53"><mml:mstyle mathvariant="script"><mml:mi>R</mml:mi></mml:mstyle></mml:math></inline-formula> &#x0003D; {<italic>r</italic><sub>1</sub>, <italic>r</italic><sub>2</sub>, &#x02026;, <italic>r</italic><sub><italic>n</italic></sub>}, where <italic>r</italic><sub><italic>i</italic></sub> defines transformations or associations between linguistic constructs. The learner&#x00027;s ability to internalize <inline-formula><mml:math id="M54"><mml:mstyle mathvariant="script"><mml:mi>R</mml:mi></mml:mstyle></mml:math></inline-formula> is influenced by cognitive factors such as memory and reasoning capabilities. Define <inline-formula><mml:math id="M55"><mml:mstyle mathvariant="script"><mml:mi>P</mml:mi></mml:mstyle></mml:math></inline-formula>(<italic>r</italic><sub><italic>i</italic></sub>, <italic>t</italic>) as the probability of mastering rule <italic>r</italic><sub><italic>i</italic></sub> over time:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="script"><mml:mi>P</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:mstyle><mml:mtext>engage</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:mtext>feedback</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>&#x003C4;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mi>&#x003C4;</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where engage(<italic>r</italic><sub><italic>i</italic></sub>, &#x003C4;) represents active interaction with <italic>r</italic><sub><italic>i</italic></sub>, and feedback(<italic>r</italic><sub><italic>i</italic></sub>, &#x003C4;) captures corrective signals received during learning. The contextual use of English involves the integration of vocabulary and grammar with socio-cultural norms. We model this integration using a latent semantic space <inline-formula><mml:math id="M57"><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:math></inline-formula>, where each word or phrase <italic>w</italic><sub><italic>i</italic></sub> maps to a point <inline-formula><mml:math id="M5"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>. Contextual similarity between two phrases <italic>w</italic><sub><italic>i</italic></sub> and <italic>w</italic><sub><italic>j</italic></sub> is measured by:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo class="qopname">cos</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Cultural fluency is then represented as the learner&#x00027;s capacity to form coherent trajectories in <inline-formula><mml:math id="M58"><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:math></inline-formula>, connecting semantic and pragmatic elements effectively. Given a target proficiency level <inline-formula><mml:math id="M59"><mml:mstyle mathvariant="script"><mml:mi>T</mml:mi></mml:mstyle></mml:math></inline-formula> defined across the axes of vocabulary, grammar, and context, the objective is to design an optimal learning pathway <inline-formula><mml:math id="M60"><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:math></inline-formula><sup>&#x0002A;</sup> such that:</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mo class="qopname">arg</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo class="qopname">max</mml:mo></mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:munder></mml:mstyle><mml:mstyle displaystyle="true"><mml:msubsup><mml:mrow><mml:mo class="qopname">&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup></mml:mstyle><mml:mstyle mathvariant="script"><mml:mi>U</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mi>t</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M61"><mml:mstyle mathvariant="script"><mml:mi>U</mml:mi></mml:mstyle></mml:math></inline-formula>(<inline-formula><mml:math id="M62"><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:math></inline-formula>(<italic>t</italic>)) denotes the utility function capturing linguistic growth at time <italic>t</italic>.</p>
</sec>
<sec>
<title>3.3 Dynamic Linguistic Enhancement Model</title>
<p>To advance the field of English education and address multifaceted challenges in second-language learning, we propose the Dynamic Linguistic Enhancement Model (DLEM). This innovative framework synergizes cognitive science, adaptive learning methodologies, and computational advancements to deliver a personalized, structured, and engaging approach to language acquisition (as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>). Below, we detail its three core innovations as following.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Dynamic Linguistic Enhancement Model (DLEM): an integrated framework for adaptive multimodal learning, leveraging Global Feature Alignment (GFA) and Local Feature Imagination (LFI) to enhance vocabulary, grammar, and cultural fluency. The model employs multimodal processing units, feature alignment mechanisms, and transformer-based embedding systems to deliver personalized and contextual second-language acquisition.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1540539-g0001.tif"/>
</fig>
<sec>
<title>3.3.1 Dynamic vocabulary graphs for adaptive learning</title>
<p>The vocabulary acquisition module is built upon the concept of a dynamically evolving knowledge graph <inline-formula><mml:math id="M63"><mml:mstyle mathvariant="script"><mml:mi>G</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>vocab</italic></sub> &#x0003D; (<inline-formula><mml:math id="M64"><mml:mstyle mathvariant="script"><mml:mi>V</mml:mi></mml:mstyle></mml:math></inline-formula>, <inline-formula><mml:math id="M65"><mml:mstyle mathvariant="script"><mml:mi>E</mml:mi></mml:mstyle></mml:math></inline-formula>), which provides a structured representation of words and their interrelations to facilitate contextual and personalized vocabulary learning. In this graph, <inline-formula><mml:math id="M66"><mml:mstyle mathvariant="script"><mml:mi>V</mml:mi></mml:mstyle></mml:math></inline-formula> represents the nodes, where each node corresponds to a vocabulary term, and <inline-formula><mml:math id="M67"><mml:mstyle mathvariant="script"><mml:mi>E</mml:mi></mml:mstyle></mml:math></inline-formula> denotes the edges, capturing semantic, syntactic, or phonetic relationships. The model dynamically adapts the graph structure and learning strategies to the individual learner&#x00027;s progress through a personalized transition probability matrix <bold>P</bold>(<italic>t</italic>), which updates over time based on performance and engagement. The learning probability of a specific vocabulary node <italic>v</italic><sub><italic>i</italic></sub>&#x02208;<inline-formula><mml:math id="M68"><mml:mstyle mathvariant="script"><mml:mi>V</mml:mi></mml:mstyle></mml:math></inline-formula> at time <italic>t</italic> is governed by the relationship:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mstyle mathvariant="script"><mml:mi>N</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:mstyle><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:mstyle mathvariant="script"><mml:mi>V</mml:mi></mml:mstyle></mml:mrow></mml:msub></mml:mstyle><mml:mi>&#x003C8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M69"><mml:mstyle mathvariant="script"><mml:mi>N</mml:mi></mml:mstyle></mml:math></inline-formula>(<italic>v</italic><sub><italic>i</italic></sub>) is the set of neighboring nodes (contextually related words), <bold>P</bold><sub><italic>ij</italic></sub>(<italic>t</italic>) is the transition probability from <italic>v</italic><sub><italic>j</italic></sub> to <italic>v</italic><sub><italic>i</italic></sub>, and &#x003C8;(<italic>v</italic><sub><italic>j</italic></sub>) represents the contextual relevance score of <italic>v</italic><sub><italic>j</italic></sub> to the learner&#x00027;s current state. To improve long-term retention and adapt to user interactions, an adaptive reinforcement mechanism is introduced. The transition probabilities <bold>P</bold>(<italic>t</italic>) are updated iteratively through:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B6;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BA;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="script"><mml:mi>A</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B6; is the learning rate, &#x003BA;<sub><italic>ij</italic></sub> is the Kronecker delta indicating direct interaction between <italic>v</italic><sub><italic>i</italic></sub> and <italic>v</italic><sub><italic>j</italic></sub>, and <inline-formula><mml:math id="M70"><mml:mstyle mathvariant="script"><mml:mi>A</mml:mi></mml:mstyle></mml:math></inline-formula>(<italic>v</italic><sub><italic>i</italic></sub>, <italic>t</italic>) quantifies the learner&#x00027;s performance on <italic>v</italic><sub><italic>i</italic></sub>, such as accuracy or frequency of correct usage. Furthermore, a temporal decay function &#x003BB;(<italic>t</italic>) is incorporated to account for the natural forgetting curve, modifying &#x003C8;(<italic>v</italic><sub><italic>j</italic></sub>) dynamically as:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>&#x003C8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>t</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mo>&#x003BB;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B2;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="script"><mml:mi>F</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B2; is a reinforcement factor, and <inline-formula><mml:math id="M71"><mml:mstyle mathvariant="script"><mml:mi>F</mml:mi></mml:mstyle></mml:math></inline-formula>(<italic>v</italic><sub><italic>j</italic></sub>, <italic>t</italic>) measures recent interactions with <italic>v</italic><sub><italic>j</italic></sub>. The vocabulary graph also integrates a semantic clustering mechanism, grouping words into thematic clusters <inline-formula><mml:math id="M72"><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>k</italic></sub>&#x02286;<inline-formula><mml:math id="M73"><mml:mstyle mathvariant="script"><mml:mi>V</mml:mi></mml:mstyle></mml:math></inline-formula>, each defined by a centroid <bold>c</bold><sub><italic>k</italic></sub>, and dynamically recalculates these centroids based on usage statistics:</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>c</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mstyle><mml:mi>&#x003C8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mstyle><mml:mi>&#x003C8;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <bold>v</bold><sub><italic>i</italic></sub> is the embedding vector of <italic>v</italic><sub><italic>i</italic></sub>. By aligning the learner&#x00027;s progression with these semantic clusters, the system enhances thematic learning and contextual reinforcement, fostering both breadth and depth in vocabulary acquisition.</p></sec>
<sec>
<title>3.3.2 Hybrid Grammar Contextualization Engine</title>
<p>The grammar module is designed to integrate neural network-based learning and symbolic grammar rules, forming a hybrid framework that leverages both statistical learning and explicit rule-based syntax constraints (as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>). This engine comprises two principal components: a neural parser, represented as LSTM<sub>parse</sub>, and a symbolic validator, denoted as CRF<sub>validate</sub>. The parser identifies hierarchical sentence structures by learning latent representations of syntactic patterns, while the validator ensures contextual consistency by applying explicit rules and relationships from the syntactic set <inline-formula><mml:math id="M74"><mml:mstyle mathvariant="script"><mml:mi>R</mml:mi></mml:mstyle></mml:math></inline-formula> and contextual embeddings <inline-formula><mml:math id="M75"><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:math></inline-formula>. The model&#x00027;s objective is to maximize the conditional likelihood:</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">grammar</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mo class="qopname">log</mml:mo><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>S</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02223;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>R</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M77"><mml:mstyle mathvariant="script"><mml:mi>S</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> represents the sentence structure at time <italic>t</italic>, <inline-formula><mml:math id="M78"><mml:mstyle mathvariant="script"><mml:mi>R</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> is the active rule set, and <inline-formula><mml:math id="M79"><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> is the corresponding context vector derived from embeddings. The neural parsing component LSTM<sub>parse</sub> outputs a probability distribution over parse trees <inline-formula><mml:math id="M80"><mml:mstyle mathvariant="script"><mml:mi>T</mml:mi></mml:mstyle></mml:math></inline-formula>:</p>
<disp-formula id="E12"><label>(12)</label><mml:math id="M13"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>81</mml:mi></mml:mstyle><mml:mo>&#x02223;</mml:mo><mml:mstyle mathvariant="script"><mml:mi>S</mml:mi></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x0220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mi>&#x003C3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>b</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>h</italic><sub><italic>i</italic></sub> is the hidden state of the LSTM at position <italic>i</italic>, <italic>W</italic> is a learnable weight matrix, and &#x003C3; is the activation function. To align predictions with predefined syntactic rules, the CRF layer imposes constraints by computing a score for valid sentence parses:</p>
<disp-formula id="E13"><label>(13)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">score</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>82</mml:mi></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="script"><mml:mi>83</mml:mi></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:mstyle mathvariant="script"><mml:mi>E</mml:mi></mml:mstyle></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="script"><mml:mi>84</mml:mi></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M85"><mml:mstyle mathvariant="script"><mml:mi>E</mml:mi></mml:mstyle></mml:math></inline-formula> denotes the edges in the parse tree, &#x003B1;<sub><italic>ij</italic></sub> represents transition probabilities between nodes <italic>i</italic> and <italic>j</italic>, and <inline-formula><mml:math id="M86"><mml:mstyle mathvariant="script"><mml:mi>R</mml:mi></mml:mstyle></mml:math></inline-formula>(<italic>i, j</italic>) evaluates rule validity.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Hybrid Grammar Contextualization Engine (HGCE): a multimodal framework integrating visual, audio, and linguistic features through low-rank multimodal fusion. The system generates unified multimodal representations, leveraging low-rank factorization across modalities for efficient feature extraction, which informs predictions. This engine bridges neural network-based parsing and symbolic validation for robust grammar contextualization.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1540539-g0002.tif"/>
</fig>
<p>The integration of context <inline-formula><mml:math id="M87"><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> further enhances the adaptability of the grammar engine by embedding semantic nuances into rule application. Contextual embeddings are computed as:</p>
<disp-formula id="E14"><label>(14)</label><mml:math id="M15"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>c</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munder><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>e</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>w</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>W</italic><sub><italic>t</italic></sub> is the set of words in the sentence at time <italic>t</italic>, and <bold>e</bold><sub><italic>w</italic></sub> is the embedding of word <italic>w</italic>. These embeddings are dynamically updated using attention weights &#x003B2;<sub><italic>i</italic></sub> to prioritize contextually relevant terms:</p>
<disp-formula id="E15"><label>(15)</label><mml:math id="M16"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>c</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">attn</mml:mtext></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>e</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:msub><mml:mrow><mml:mi>&#x003B2;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>e</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>q</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="false"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>e</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>q</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <bold>q</bold> is a query vector representing the task focus. By combining these mechanisms, the grammar module enables robust syntactic learning while maintaining contextual adaptability, ensuring grammatical accuracy and relevance in diverse linguistic environments. Furthermore, a feedback loop reinforces correct parses by updating <inline-formula><mml:math id="M88"><mml:mstyle mathvariant="script"><mml:mi>R</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> based on validated structures, facilitating adaptive learning and refinement of syntactic understanding over time.</p></sec>
<sec>
<title>3.3.3 Cultural embedding for cross-cultural fluency</title>
<p>To effectively integrate linguistic nuances with cultural context, DLEM employs a sophisticated cultural embedding space <inline-formula><mml:math id="M89"><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>cultural</italic></sub>, constructed using transformer-based architectures. This space encodes words, phrases, and expressions as multi-dimensional vectors <inline-formula><mml:math id="M17"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, enriched with cultural attributes <bold>c</bold><sub><italic>j</italic></sub> that reflect specific sociolinguistic and cultural features. Each embedding is dynamically adapted to capture cross-cultural intricacies, modeled as:</p>
<disp-formula id="E16"><label>(16)</label><mml:math id="M18"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">cultural</mml:mtext></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>c</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B3;<sub><italic>j</italic></sub> are attention weights derived through a self-attention mechanism, ensuring that culturally relevant attributes <bold>c</bold><sub><italic>j</italic></sub> are emphasized according to the context of use. These attributes are generated from transformer encoder layers trained on diverse multilingual and multimodal datasets, enabling the model to infer cultural subtleties embedded in language.</p>
<p>To facilitate learning, the model aligns the embeddings of learner expressions <inline-formula><mml:math id="M19"><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">learner</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:math></inline-formula> with target cultural embeddings <inline-formula><mml:math id="M20"><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">native</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:math></inline-formula>. The similarity metric, defined as:</p>
<disp-formula id="E17"><label>(17)</label><mml:math id="M21"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">learner</mml:mtext></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">native</mml:mtext></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">learner</mml:mtext></mml:mrow></mml:msubsup><mml:mo>&#x000B7;</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">native</mml:mtext></mml:mrow></mml:msubsup></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">learner</mml:mtext></mml:mrow></mml:msubsup><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">native</mml:mtext></mml:mrow></mml:msubsup><mml:mo>|</mml:mo><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>is optimized to maximize alignment, ensuring that learners internalize culturally appropriate usage patterns. This alignment is guided by a loss function <inline-formula><mml:math id="M90"><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:math></inline-formula><sub>alignment</sub>, which penalizes discrepancies between learner and target embeddings:</p>
<disp-formula id="E18"><label>(18)</label><mml:math id="M22"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">alignment</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mo class="qopname">log</mml:mo><mml:mtext>sim</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">learner</mml:mtext></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">native</mml:mtext></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Cultural embeddings are further enhanced by integrating contextual elements derived from the learning environment. Context vectors <bold>q</bold><sub><italic>t</italic></sub> are constructed dynamically as:</p>
<disp-formula id="E19"><label>(19)</label><mml:math id="M23"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>q</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <bold>x</bold><sub><italic>k</italic></sub> are feature vectors representing situational cues (e.g., location, time, interlocutor profile), and &#x003B4;<sub><italic>k</italic></sub> are their respective importance weights. This enables DLEM to adapt its cultural encoding in real time, ensuring relevance to the learner&#x00027;s immediate context.</p>
<p>The optimization objective of DLEM integrates vocabulary, grammar, and cultural components through a utility function:</p>
<disp-formula id="E20"><label>(20)</label><mml:math id="M24"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="script"><mml:mi>U</mml:mi></mml:mstyle><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C9;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>U</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">vocab</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C9;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>U</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">grammar</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C9;</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>U</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">cultural</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M91"><mml:mstyle mathvariant="script"><mml:mi>U</mml:mi></mml:mstyle></mml:math></inline-formula><sub>cultural</sub> evaluates the learner&#x00027;s alignment with cultural embeddings as:</p>
<disp-formula id="E21"><label>(21)</label><mml:math id="M25"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>U</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">cultural</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle><mml:mo>|</mml:mo></mml:mrow></mml:munderover></mml:mstyle><mml:mtext>sim</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">learner</mml:mtext></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">native</mml:mtext></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mi>&#x003BA;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">learner</mml:mtext></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>v</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">native</mml:mtext></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>with <italic>d</italic>(&#x000B7;, &#x000B7;) representing a distance metric and &#x003BA; a sensitivity parameter. This formulation emphasizes both similarity and proximity in embedding space, fostering cultural fluency and adaptability. Through its nuanced approach to embedding cultural attributes, DLEM empowers learners to achieve linguistic mastery within the sociocultural contexts of their target languages.</p>
</sec>
</sec>
<sec>
<title>3.4 Contextual Augmented Learning Strategy</title>
<p>The Contextual Augmented Learning Strategy (CALS) introduces a comprehensive framework designed to facilitate the seamless integration of advanced linguistic models into diverse educational and digital platforms (as shown in <xref ref-type="fig" rid="F3">Figure 3</xref>). This section highlights the three key innovations in CALS: Adaptive Curriculum Design, Dynamic Feedback Systems, and Gamified Engagement Frameworks.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Contextual Augmented Learning Strategy (CALS) architecture: The diagram illustrates the core components of CALS, showcasing the sequential data processing pipeline. Starting with input signals, depthwise convolution for feature extraction, followed by average pooling for dimensionality reduction. Positional encoding integrates positional encoding to capture contextual relationships within the transformer network, enabling advanced linguistic modeling and adaptive learning.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1540539-g0003.tif"/>
</fig>
<sec>
<title>3.4.1 Adaptive curriculum design</title>
<p>CALS ensures a highly personalized learning experience by dynamically tailoring the curriculum to align with each learner&#x00027;s evolving proficiency. At any given time <italic>t</italic>, the learner&#x00027;s linguistic state is captured by the vector <bold>L</bold><sub><italic>t</italic></sub> &#x0003D; (<inline-formula><mml:math id="M92"><mml:mstyle mathvariant="script"><mml:mi>X</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub>, <inline-formula><mml:math id="M93"><mml:mstyle mathvariant="script"><mml:mi>Y</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub>, <inline-formula><mml:math id="M94"><mml:mstyle mathvariant="script"><mml:mi>Z</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub>), representing the learner&#x00027;s levels across three critical dimensions: vocabulary proficiency (<inline-formula><mml:math id="M95"><mml:mstyle mathvariant="script"><mml:mi>X</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub>), grammatical comprehension (<inline-formula><mml:math id="M96"><mml:mstyle mathvariant="script"><mml:mi>Y</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub>), and cultural-contextual understanding (<inline-formula><mml:math id="M97"><mml:mstyle mathvariant="script"><mml:mi>Z</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub>). The system continually evaluates the learner&#x00027;s state against a target proficiency profile <bold>L</bold><sub>desired</sub>, defined as the optimal levels of linguistic competence. The proficiency gap is quantified as:</p>
<disp-formula id="E22"><label>(22)</label><mml:math id="M26"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mo>&#x00394;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>L</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">desired</mml:mtext></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>L</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where ||&#x000B7;||<sub>2</sub> represents the Euclidean distance, ensuring a holistic measurement of the gap across dimensions. To bridge this gap, CALS leverages a dynamic optimization approach by minimizing a weighted loss function:</p>
<disp-formula id="E23"><label>(23)</label><mml:math id="M27"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">total</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">vocab</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B2;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">grammar</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B3;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">culture</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M98"><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:math></inline-formula><sub>vocab</sub>, <inline-formula><mml:math id="M99"><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:math></inline-formula><sub>grammar</sub>, and <inline-formula><mml:math id="M100"><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:math></inline-formula><sub>culture</sub> are losses associated with vocabulary acquisition, grammatical proficiency, and cultural understanding, respectively. The weights &#x003B1;(<italic>t</italic>), &#x003B2;(<italic>t</italic>), and &#x003B3;(<italic>t</italic>) adapt dynamically based on diagnostic assessments and learner progress, ensuring that emphasis is placed on areas requiring the most improvement. Furthermore, CALS incorporates a predictive feedback loop to anticipate future learning trajectories. The predicted proficiency vector <bold>L</bold><sub><italic>t</italic>&#x0002B;1</sub> is modeled as:</p>
<disp-formula id="E24"><label>(24)</label><mml:math id="M28"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>L</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>L</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B7;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>G</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <bold>G</bold><sub><italic>t</italic></sub> &#x0003D; (<inline-formula><mml:math id="M101"><mml:mstyle mathvariant="script"><mml:mi>G</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>X</italic></sub>, <inline-formula><mml:math id="M102"><mml:mstyle mathvariant="script"><mml:mi>G</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>Y</italic></sub>, <inline-formula><mml:math id="M103"><mml:mstyle mathvariant="script"><mml:mi>G</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>Z</italic></sub>) represents the gradient of learning improvements across the dimensions, and &#x003B7; is a learning rate determined by the learner&#x00027;s responsiveness. To refine this process further, CALS employs an iterative gradient update mechanism:</p>
<disp-formula id="E25"><label>(25)</label><mml:math id="M29"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>L</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>L</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mo>&#x003BB;</mml:mo><mml:mo>&#x02207;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">total</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>L</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003BB; is an adaptive step size adjusted based on the learner&#x00027;s learning velocity and performance variability. This ensures convergence toward the desired proficiency with maximal efficiency. CALS also uses probabilistic sampling to select the next instructional focus area, balancing reinforcement of strong skills and addressing weaker areas. By integrating real-time analytics, predictive modeling, and dynamic loss optimization, the adaptive curriculum fosters a precise and scalable approach to language learning tailored for individual progress.</p></sec>
<sec>
<title>3.4.2 Dynamic Feedback Systems</title>
<p>CALS employs a sophisticated multi-channel feedback system designed to provide learners with actionable insights and maintain their engagement throughout the learning process (as shown in <xref ref-type="fig" rid="F4">Figure 4</xref>). Feedback at any time <italic>t</italic> is generated as:</p>
<disp-formula id="E26"><label>(26)</label><mml:math id="M30"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B7;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>104</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M105"><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> represents accuracy-based feedback derived from the learner&#x00027;s performance metrics, <inline-formula><mml:math id="M106"><mml:mstyle mathvariant="script"><mml:mi>E</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> captures engagement-driven feedback reflecting effort and persistence, and &#x003B7; is a scaling factor dynamically calibrated to balance between cognitive and affective dimensions of learning. The accuracy-based feedback <inline-formula><mml:math id="M107"><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> is computed as:</p>
<disp-formula id="E27"><label>(27)</label><mml:math id="M31"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>&#x003B4;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>108</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B4;<sub><italic>i</italic></sub> denotes correctness for item <italic>i</italic>, and <inline-formula><mml:math id="M109"><mml:mstyle mathvariant="script"><mml:mi>E</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>i</italic></sub> accounts for effort normalized across all items. This ensures that learners receive constructive feedback, even on partially correct attempts. Engagement-driven feedback <inline-formula><mml:math id="M110"><mml:mstyle mathvariant="script"><mml:mi>E</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> is modeled using learner-specific persistence scores and activity patterns:</p>
<disp-formula id="E28"><label>(28)</label><mml:math id="M32"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>111</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003C1;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>A</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>112</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M113"><mml:mstyle mathvariant="script"><mml:mi>A</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> is the learner&#x00027;s active time during the session, <inline-formula><mml:math id="M114"><mml:mstyle mathvariant="script"><mml:mi>T</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> is the total allotted session time, and &#x003C1; is a coefficient capturing the learner&#x00027;s historical engagement trends. Feedback is delivered in three distinct forms: immediate, delayed, and aggregated. Immediate feedback involves corrective signals provided in real-time, such as hints or explanations for errors detected during assessments. For instance, when a vocabulary error is identified, the system suggests alternative words or usage contexts to reinforce understanding. Delayed feedback is delivered post-session, offering a comprehensive summary of the learner&#x00027;s performance across dimensions such as vocabulary (<inline-formula><mml:math id="M115"><mml:mstyle mathvariant="script"><mml:mi>X</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub>), grammar (<inline-formula><mml:math id="M116"><mml:mstyle mathvariant="script"><mml:mi>Y</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub>), and cultural understanding (<inline-formula><mml:math id="M117"><mml:mstyle mathvariant="script"><mml:mi>Z</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub>), modeled as:</p>
<disp-formula id="E29"><label>(29)</label><mml:math id="M33"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>118</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>119</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>120</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>K</italic> is the number of completed tasks. Aggregated feedback spans multiple learning sessions, providing long-term trends and progress insights. This aggregated feedback leverages predictive analytics to forecast learning trajectories:</p>
<disp-formula id="E30"><label>(30)</label><mml:math id="M34"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B3;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mo>&#x02207;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>F</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B3; is a learning rate for predictive adjustments and &#x02207;<inline-formula><mml:math id="M121"><mml:mstyle mathvariant="script"><mml:mi>F</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> is the gradient of feedback improvements over time. To enhance personalization, CALS implements a feedback adaptation mechanism using a reinforcement learning-based policy <inline-formula><mml:math id="M35"><mml:msup><mml:mrow><mml:mi>&#x003C0;</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, where <italic>s</italic><sub><italic>t</italic></sub> represents the learner&#x00027;s state. The optimal policy is defined as:</p>
<disp-formula id="E31"><label>(31)</label><mml:math id="M36"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>&#x003C0;</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mo class="qopname">arg</mml:mo><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo class="qopname">max</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x003C0;</mml:mi></mml:mrow></mml:msub></mml:mstyle><mml:mstyle mathvariant="double-struck"><mml:mi>E</mml:mi></mml:mstyle><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msup><mml:mrow><mml:mi>&#x003B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msup><mml:mo>&#x000B7;</mml:mo><mml:mi>R</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>&#x003C0;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>R</italic>(<italic>s</italic><sub><italic>t</italic></sub>, &#x003C0;(<italic>s</italic><sub><italic>t</italic></sub>)) is the reward function reflecting the efficacy of the feedback. By integrating real-time assessments, engagement analytics, and adaptive policies, CALS ensures that feedback mechanisms not only address cognitive gaps but also sustain learner motivation, fostering a holistic and responsive educational experience.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Dynamic Feedback Systems in CALS: The diagram visualizes the multi-channel feedback mechanism, highlighting the interaction of accuracy-based (Cov_opt) and engagement-driven (Cov_att) feedback components. Matrix product and Hadamard product operations process query (Q), key (K), and value (V) tensors, enabling real-time adaptation between low and high optimization states (Opt_sat). The system dynamically integrates learner engagement and performance metrics to provide immediate, delayed, and aggregated feedback for a personalized learning experience.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1540539-g0004.tif"/>
</fig>
</sec>
<sec>
<title>3.4.3 Gamified Engagement Frameworks</title>
<p>To sustain learner interest and motivation, CALS integrates an advanced gamified engagement framework that employs dynamic challenges, point systems, and adaptive reward mechanisms. This framework transforms the learning experience into an interactive and rewarding journey by tailoring engagement elements to the learner&#x00027;s proficiency and progress. At any time <italic>t</italic>, the task difficulty <inline-formula><mml:math id="M122"><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> is computed as a weighted combination of baseline difficulty <inline-formula><mml:math id="M123"><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:math></inline-formula><sub>base</sub> and adaptive difficulty <inline-formula><mml:math id="M124"><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:math></inline-formula><sub>adaptive</sub>, represented as:</p>
<disp-formula id="E32"><label>(32)</label><mml:math id="M37"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003BA;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">base</mml:mtext></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>&#x003BA;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">adaptive</mml:mtext></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003BA;&#x02208;[0, 1] is a dynamic balancing parameter adjusted based on the learner&#x00027;s state <bold>L</bold><sub><italic>t</italic></sub>, which includes dimensions such as vocabulary proficiency, grammar comprehension, and cultural understanding. The adaptive difficulty <inline-formula><mml:math id="M125"><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:math></inline-formula><sub>adaptive</sub> ensures that challenges are neither too easy nor overly complex, maintaining optimal engagement and cognitive effort. Points and rewards are structured hierarchically, with achievements linked to milestone completions. Let <inline-formula><mml:math id="M126"><mml:mstyle mathvariant="script"><mml:mi>R</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> denote the reward function at time <italic>t</italic>, defined as:</p>
<disp-formula id="E33"><label>(33)</label><mml:math id="M38"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>127</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x003C8;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>128</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003BE;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>129</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M130"><mml:mstyle mathvariant="script"><mml:mi>P</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> represents the points accrued through task completion, <inline-formula><mml:math id="M131"><mml:mstyle mathvariant="script"><mml:mi>T</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>t</italic></sub> accounts for the time spent on challenging tasks, and &#x003C8;, &#x003BE; are scaling factors emphasizing productivity and persistence. These rewards are tiered, with higher tiers unlocked as learners achieve predefined proficiency thresholds:</p>
<disp-formula id="E34"><label>(34)</label><mml:math id="M39"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>T</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>R</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M132"><mml:mstyle mathvariant="script"><mml:mi>T</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>i</italic></sub> is the cumulative reward for tier <italic>i</italic> and <italic>M</italic> denotes the number of completed subtasks at that tier. Gamified challenges are further personalized using adaptive algorithms that analyze learner trajectories. For instance, adaptive challenges are designed to maintain a consistent engagement level by predicting learner fatigue or overconfidence. The probability of assigning a specific challenge type <inline-formula><mml:math id="M133"><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:math></inline-formula><sub><italic>k</italic></sub> at time <italic>t</italic> is modeled as:</p>
<disp-formula id="E35"><label>(35)</label><mml:math id="M40"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>C</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>L</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mi>&#x003D5;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="false"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mi>&#x003D5;</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>D</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003D5; is an adjustment parameter controlling challenge diversity, and <italic>N</italic> is the total number of available challenges. Social engagement elements amplify the impact of gamification by fostering community-driven learning. Peer comparisons and collaborative tasks encourage learners to benchmark their performance against others, promoting healthy competition and teamwork. Leaderboards are dynamically updated to reflect achievements across groups, calculated as:</p>
<disp-formula id="E36"><label>(36)</label><mml:math id="M41"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">rank</mml:mtext></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">rank</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>P</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mstyle mathvariant="script"><mml:mi>G</mml:mi></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M134"><mml:mstyle mathvariant="script"><mml:mi>G</mml:mi></mml:mstyle></mml:math></inline-formula> represents the group of peers. Collaborative challenges integrate shared goals, incentivizing learners to collectively achieve milestones.</p></sec></sec>
</sec>
<sec id="s4">
<title>4 Experimental setup</title>
<sec>
<title>4.1 Dataset</title>
<p>The Sleep-EDF Dataset (Wang et al., <xref ref-type="bibr" rid="B37">2024</xref>) is a comprehensive collection of sleep recordings designed for research in sleep stage classification and related studies. It includes polysomnographic (PSG) data, encompassing electroencephalogram (EEG), electrooculogram (EOG), and electromyogram (EMG) signals from healthy individuals and patients with sleep disorders. The dataset spans multiple nights for some subjects, offering insights into inter-night variability. The detailed annotations and long-term recordings make it a valuable resource for sleep pattern analysis and machine learning applications in health monitoring. The EEGEyeNet Dataset (Modesitt et al., <xref ref-type="bibr" rid="B22">2023</xref>) focuses on eye movement classification using EEG signals. It consists of recordings from subjects performing controlled eye movements, such as fixations and saccades, under well-defined experimental conditions. The dataset includes high-resolution EEG data and corresponding event markers, providing a robust foundation for developing models that link neural activity to ocular dynamics. Its emphasis on eye movement makes it uniquely suited for advancing research in brain-computer interfaces and cognitive neuroscience. The CHB-MIT Dataset (Duan et al., <xref ref-type="bibr" rid="B10">2021</xref>) is a widely used resource for seizure detection and prediction studies, offering long-term EEG recordings from pediatric epilepsy patients. The dataset includes scalp EEG data annotated with seizure events, recorded over extended periods to capture both ictal and interictal states. The comprehensive annotations and real-world variability make it an essential benchmark for developing and evaluating algorithms in epilepsy diagnosis and management, particularly in clinical and ambulatory settings. The PhyAAt Dataset (Ahuja and Setia, <xref ref-type="bibr" rid="B2">2022</xref>) is a multi-modal collection designed for physical activity analysis and assessment. It integrates accelerometer, gyroscope, and physiological data, such as heart rate, captured during various physical activities and rest states. The dataset includes diverse demographic information, ensuring its applicability across different populations. Its multi-modal nature enables the exploration of relationships between physiological and physical signals, making it a key resource for wearable technology development and health monitoring systems.</p>
<p>While none of the datasets used in this study were collected under natural high-altitude or hypoxic conditions, they were selected for their high signal quality, extensive annotations, and task diversity&#x02013;making them well-suited for controlled evaluation of EEG-based cognitive modeling frameworks. The Sleep-EDF dataset captures physiological brain states during cognitive transitions such as sleep stage changes; CHB-MIT contains EEG recordings under clinical stress settings, including epileptic seizure episodes; and EEGEyeNet includes tasks involving attentional shifts and oculomotor coordination. Although these contexts differ from altitude-induced stress, they share critical cognitive stress features such as fluctuating attention, increased working memory demands, and altered neurophysiological baselines. To approximate real-world cognitive stressors associated with hypoxia, we designed our task stimuli and preprocessing strategy to simulate conditions of high mental load. For example, auditory comprehension inputs were structured with temporally compressed, semantically rich materials to elevate processing demands. These interventions elicit EEG dynamics (e.g., elevated theta and suppressed alpha power) that closely align with prior studies on acute hypoxic exposure. Consequently, while our current data does not originate from high-altitude populations or explicitly track participants&#x00027; native language profiles, it provides a valid simulation environment for benchmarking the DLEM and CALS framework. We fully acknowledge the importance of ecological validity. Future extensions of this work will involve targeted EEG data collection from individuals residing in high-altitude regions or within hypobaric chamber conditions. This will allow for stratified model validation and domain-specific adaptation. At the current stage, however, our goal is to demonstrate the architectural generalizability of our model under controlled, stress-emulated settings, laying a foundation for field-deployable applications.</p>
</sec>
<sec>
<title>4.2 Experimental details</title>
<p>The experiments were conducted on a server equipped with an NVIDIA RTX 3090 GPU and 128 GB RAM to ensure computational efficiency. For model training, PyTorch was utilized as the primary deep learning framework. The Adam optimizer was chosen for its adaptive learning rate properties, set to an initial learning rate of 1 &#x000D7; 10<sup>&#x02212;3</sup> with a cosine annealing scheduler to gradually reduce the learning rate during training. The batch size was set to 64, balancing memory constraints and training speed. All models were trained for 100 epochs to ensure convergence while avoiding overfitting. For preprocessing, the EEG signals were band-pass filtered between 0.5 Hz and 50 Hz to remove artifacts and focus on the relevant frequency bands. We applied a band-pass filter between 0.5 Hz and 50 Hz to all EEG signals, which is a widely accepted standard in cognitive and neuropsychological studies. This frequency window was chosen to preserve the core EEG components known to reflect cognitive processes&#x02013;such as theta and alpha rhythms associated with working memory and attention, and beta/gamma rhythms linked to cognitive control and perceptual integration. Frequencies below 0.5 Hz were excluded to eliminate slow baseline drifts and electrodermal artifacts, while frequencies above 50 Hz were removed to suppress powerline interference and muscle-related artifacts. The preserved bands (0.5&#x02013;50 Hz) include delta (0.5&#x02013;4 Hz), theta (4&#x02013;8 Hz), alpha (8&#x02013;13 Hz), beta (13&#x02013;30 Hz), and low gamma (30&#x02013;50 Hz), all of which have been shown to be modulated under hypoxic conditions in existing EEG literature. While some ultra-high frequency activity (&#x0003E;60 Hz) has been reported in invasive or high-density EEG contexts, such ranges are more susceptible to environmental noise in scalp recordings, particularly under mobile or multi-site experimental setups. Therefore, the selected band range represents a practical and physiologically meaningful trade-off to support consistent signal processing across datasets with different recording conditions. Future work may explore dynamic filtering or high-frequency EEG analysis in closed-loop neurofeedback systems under extended hypoxic exposure. Data augmentation techniques, including random cropping and noise injection, were applied to increase model robustness. Each dataset was split into 80% training, 10% validation, and 10% testing sets, ensuring a balanced evaluation. Cross-validation was employed where applicable to ensure consistency across splits. The neural network architecture comprised a combination of convolutional and recurrent layers. The model included a convolutional feature extractor followed by bidirectional LSTMs to capture temporal dependencies. Dropout layers with a rate of 0.5 were used to mitigate overfitting, and a softmax activation function was applied at the output layer for multi-class classification tasks. Metrics used for evaluation included accuracy, precision, recall, F1-score, and area under the ROC curve (AUC). These metrics were computed for each dataset to enable a thorough assessment of the model&#x00027;s performance across diverse scenarios. Gradient class activation maps (Grad-CAMs) were employed to visualize model decision-making, offering interpretability for the deep learning predictions. Hyperparameter tuning was conducted using grid search, varying learning rates, batch sizes, and dropout rates. The optimal configuration was selected based on validation performance. Regularization techniques, such as <italic>L</italic>2 regularization with a weight decay factor of 1 &#x000D7; 10<sup>&#x02212;4</sup>, were incorporated to prevent overfitting. The models were implemented with mixed precision training to accelerate computation without compromising numerical stability. For datasets with imbalanced class distributions, techniques like oversampling and class-specific weighting were applied during training to ensure fair representation. All experiments were repeated three times to account for randomness, and results were reported as mean values with standard deviations (<xref ref-type="fig" rid="F9">Algorithm 1</xref>).</p>
<fig id="F9" position="float">
<label>Algorithm 1</label>
<caption><p>Training process of DLEM on multi-dataset framework.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1540539-g0009.tif"/>
</fig>
</sec>
<sec>
<title>4.3 Comparison with SOTA methods</title>
<p>The performance of our proposed model was assessed against various state-of-the-art (SOTA) methods, including CLIP (Zhang et al., <xref ref-type="bibr" rid="B43">2025</xref>), ViT (Touvron et al., <xref ref-type="bibr" rid="B36">2022</xref>), I3D (Peng et al., <xref ref-type="bibr" rid="B23">2023</xref>), BLIP (Wattasseril et al., <xref ref-type="bibr" rid="B38">2023</xref>), Wav2Vec 2.0 (Chen and Rudnicky, <xref ref-type="bibr" rid="B7">2023</xref>), and T5 (Grover et al., <xref ref-type="bibr" rid="B12">2021</xref>), across diverse datasets such as Sleep-EDF, EEGEyeNet, CHB-MIT, and PhyAAt. Comprehensive results are summarized in <xref ref-type="table" rid="T1">Tables 1</xref>, <xref ref-type="table" rid="T2">2</xref>, highlighting key metrics like accuracy and recall.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Comparison of ours with SOTA methods on sleep-EDF and EEGEyeNet datasets for emotion analysis.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center" colspan="4"><bold>Sleep-EDF dataset</bold></th>
<th valign="top" align="center" colspan="4"><bold>EEGEyeNet dataset</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="center"><bold>Accuracy</bold></td>
<td valign="top" align="center"><bold>Recall</bold></td>
<td valign="top" align="center"><bold>F1 Score</bold></td>
<td valign="top" align="center"><bold>AUC</bold></td>
<td valign="top" align="center"><bold>Accuracy</bold></td>
<td valign="top" align="center"><bold>Recall</bold></td>
<td valign="top" align="center"><bold>F1 Score</bold></td>
<td valign="top" align="center"><bold>AUC</bold></td>
</tr>
<tr>
<td valign="top" align="left">CLIP (Zhang et al., <xref ref-type="bibr" rid="B43">2025</xref>)</td>
<td valign="top" align="center">83.74 &#x000B1; 0.02</td>
<td valign="top" align="center">81.22 &#x000B1; 0.03</td>
<td valign="top" align="center">80.89 &#x000B1; 0.02</td>
<td valign="top" align="center">84.30 &#x000B1; 0.02</td>
<td valign="top" align="center">85.60 &#x000B1; 0.02</td>
<td valign="top" align="center">82.47 &#x000B1; 0.02</td>
<td valign="top" align="center">84.12 &#x000B1; 0.03</td>
<td valign="top" align="center">87.90 &#x000B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">ViT (Touvron et al., <xref ref-type="bibr" rid="B36">2022</xref>)</td>
<td valign="top" align="center">85.12 &#x000B1; 0.03</td>
<td valign="top" align="center">82.48 &#x000B1; 0.02</td>
<td valign="top" align="center">84.77 &#x000B1; 0.03</td>
<td valign="top" align="center">86.55 &#x000B1; 0.02</td>
<td valign="top" align="center">86.80 &#x000B1; 0.03</td>
<td valign="top" align="center">83.05 &#x000B1; 0.02</td>
<td valign="top" align="center">85.11 &#x000B1; 0.02</td>
<td valign="top" align="center">88.63 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">I3D (Peng et al., <xref ref-type="bibr" rid="B23">2023</xref>)</td>
<td valign="top" align="center">84.65 &#x000B1; 0.02</td>
<td valign="top" align="center">80.98 &#x000B1; 0.03</td>
<td valign="top" align="center">83.49 &#x000B1; 0.02</td>
<td valign="top" align="center">85.76 &#x000B1; 0.03</td>
<td valign="top" align="center">84.93 &#x000B1; 0.02</td>
<td valign="top" align="center">81.88 &#x000B1; 0.02</td>
<td valign="top" align="center">83.62 &#x000B1; 0.03</td>
<td valign="top" align="center">86.70 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">BLIP (Wattasseril et al., <xref ref-type="bibr" rid="B38">2023</xref>)</td>
<td valign="top" align="center">86.30 &#x000B1; 0.02</td>
<td valign="top" align="center">83.75 &#x000B1; 0.03</td>
<td valign="top" align="center">84.91 &#x000B1; 0.03</td>
<td valign="top" align="center">88.45 &#x000B1; 0.02</td>
<td valign="top" align="center">88.15 &#x000B1; 0.02</td>
<td valign="top" align="center">85.99 &#x000B1; 0.02</td>
<td valign="top" align="center">86.34 &#x000B1; 0.02</td>
<td valign="top" align="center">89.22 &#x000B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">Wav2Vec 2.0 (Chen and Rudnicky, <xref ref-type="bibr" rid="B7">2023</xref>)</td>
<td valign="top" align="center">84.21 &#x000B1; 0.03</td>
<td valign="top" align="center">81.05 &#x000B1; 0.02</td>
<td valign="top" align="center">82.90 &#x000B1; 0.02</td>
<td valign="top" align="center">85.10 &#x000B1; 0.03</td>
<td valign="top" align="center">85.50 &#x000B1; 0.02</td>
<td valign="top" align="center">82.70 &#x000B1; 0.02</td>
<td valign="top" align="center">84.45 &#x000B1; 0.03</td>
<td valign="top" align="center">87.05 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">T5 (Grover et al., <xref ref-type="bibr" rid="B12">2021</xref>)</td>
<td valign="top" align="center">87.45 &#x000B1; 0.02</td>
<td valign="top" align="center">84.80 &#x000B1; 0.02</td>
<td valign="top" align="center">85.55 &#x000B1; 0.02</td>
<td valign="top" align="center">89.33 &#x000B1; 0.03</td>
<td valign="top" align="center">88.90 &#x000B1; 0.02</td>
<td valign="top" align="center">86.12 &#x000B1; 0.03</td>
<td valign="top" align="center">87.00 &#x000B1; 0.02</td>
<td valign="top" align="center">90.25 &#x000B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center">90.12 &#x000B1; 0.02</td>
<td valign="top" align="center">87.95 &#x000B1; 0.03</td>
<td valign="top" align="center">88.65 &#x000B1; 0.02</td>
<td valign="top" align="center">91.40 &#x000B1; 0.02</td>
<td valign="top" align="center">91.80 &#x000B1; 0.03</td>
<td valign="top" align="center">89.12 &#x000B1; 0.02</td>
<td valign="top" align="center">90.45 &#x000B1; 0.03</td>
<td valign="top" align="center">93.10 &#x000B1; 0.02</td>
</tr></tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Comparison of ours with SOTA methods on CHB-MIT and PhyAAt datasets for emotion analysis.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center" colspan="4"><bold>CHB-MIT dataset</bold></th>
<th valign="top" align="center" colspan="4"><bold>PhyAAt dataset</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="center"><bold>Accuracy</bold></td>
<td valign="top" align="center"><bold>Recall</bold></td>
<td valign="top" align="center"><bold>F1 Score</bold></td>
<td valign="top" align="center"><bold>AUC</bold></td>
<td valign="top" align="center"><bold>Accuracy</bold></td>
<td valign="top" align="center"><bold>Recall</bold></td>
<td valign="top" align="center"><bold>F1 Score</bold></td>
<td valign="top" align="center"><bold>AUC</bold></td>
</tr>
<tr>
<td valign="top" align="left">CLIP (Zhang et al., <xref ref-type="bibr" rid="B43">2025</xref>)</td>
<td valign="top" align="center">80.45 &#x000B1; 0.02</td>
<td valign="top" align="center">78.32 &#x000B1; 0.02</td>
<td valign="top" align="center">79.20 &#x000B1; 0.03</td>
<td valign="top" align="center">82.10 &#x000B1; 0.03</td>
<td valign="top" align="center">81.78 &#x000B1; 0.03</td>
<td valign="top" align="center">79.12 &#x000B1; 0.02</td>
<td valign="top" align="center">80.22 &#x000B1; 0.02</td>
<td valign="top" align="center">83.45 &#x000B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">ViT (Touvron et al., <xref ref-type="bibr" rid="B36">2022</xref>)</td>
<td valign="top" align="center">82.30 &#x000B1; 0.03</td>
<td valign="top" align="center">80.12 &#x000B1; 0.03</td>
<td valign="top" align="center">81.90 &#x000B1; 0.02</td>
<td valign="top" align="center">84.40 &#x000B1; 0.03</td>
<td valign="top" align="center">84.60 &#x000B1; 0.03</td>
<td valign="top" align="center">82.45 &#x000B1; 0.02</td>
<td valign="top" align="center">83.50 &#x000B1; 0.03</td>
<td valign="top" align="center">85.80 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">I3D (Peng et al., <xref ref-type="bibr" rid="B23">2023</xref>)</td>
<td valign="top" align="center">81.10 &#x000B1; 0.02</td>
<td valign="top" align="center">78.95 &#x000B1; 0.02</td>
<td valign="top" align="center">80.15 &#x000B1; 0.02</td>
<td valign="top" align="center">82.80 &#x000B1; 0.02</td>
<td valign="top" align="center">82.30 &#x000B1; 0.02</td>
<td valign="top" align="center">80.88 &#x000B1; 0.03</td>
<td valign="top" align="center">81.75 &#x000B1; 0.02</td>
<td valign="top" align="center">84.00 &#x000B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">BLIP (Wattasseril et al., <xref ref-type="bibr" rid="B38">2023</xref>)</td>
<td valign="top" align="center">83.55 &#x000B1; 0.03</td>
<td valign="top" align="center">81.30 &#x000B1; 0.03</td>
<td valign="top" align="center">82.70 &#x000B1; 0.02</td>
<td valign="top" align="center">85.90 &#x000B1; 0.02</td>
<td valign="top" align="center">86.20 &#x000B1; 0.03</td>
<td valign="top" align="center">83.95 &#x000B1; 0.02</td>
<td valign="top" align="center">84.75 &#x000B1; 0.02</td>
<td valign="top" align="center">87.50 &#x000B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">Wav2Vec 2.0 (Chen and Rudnicky, <xref ref-type="bibr" rid="B7">2023</xref>)</td>
<td valign="top" align="center">80.90 &#x000B1; 0.03</td>
<td valign="top" align="center">79.50 &#x000B1; 0.02</td>
<td valign="top" align="center">79.80 &#x000B1; 0.02</td>
<td valign="top" align="center">83.10 &#x000B1; 0.02</td>
<td valign="top" align="center">83.00 &#x000B1; 0.02</td>
<td valign="top" align="center">81.60 &#x000B1; 0.03</td>
<td valign="top" align="center">82.00 &#x000B1; 0.03</td>
<td valign="top" align="center">85.10 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">T5 (Grover et al., <xref ref-type="bibr" rid="B12">2021</xref>)</td>
<td valign="top" align="center">84.30 &#x000B1; 0.02</td>
<td valign="top" align="center">82.25 &#x000B1; 0.03</td>
<td valign="top" align="center">83.10 &#x000B1; 0.03</td>
<td valign="top" align="center">86.40 &#x000B1; 0.03</td>
<td valign="top" align="center">87.10 &#x000B1; 0.02</td>
<td valign="top" align="center">85.20 &#x000B1; 0.02</td>
<td valign="top" align="center">85.90 &#x000B1; 0.02</td>
<td valign="top" align="center">88.70 &#x000B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center">88.45 &#x000B1; 0.03</td>
<td valign="top" align="center">86.90 &#x000B1; 0.02</td>
<td valign="top" align="center">87.80 &#x000B1; 0.03</td>
<td valign="top" align="center">90.20 &#x000B1; 0.03</td>
<td valign="top" align="center">89.60 &#x000B1; 0.02</td>
<td valign="top" align="center">87.85 &#x000B1; 0.03</td>
<td valign="top" align="center">88.45 &#x000B1; 0.02</td>
<td valign="top" align="center">91.10 &#x000B1; 0.02</td>
</tr></tbody>
</table>
</table-wrap>
<p>On the Sleep-EDF dataset, our approach demonstrated remarkable effectiveness, achieving an accuracy of 90.12% and a recall of 87.95%, showcasing its capability in discerning intricate patterns for sleep stage classification. Similarly, the EEGEyeNet dataset results underscored the method&#x00027;s robustness in modeling temporal and multimodal embeddings, with accuracy and recall exceeding 91% and 89%, respectively. These outcomes align with Grad-CAM visualizations, which reveal the method&#x00027;s ability to focus on salient temporal features.</p>
<p>For the CHB-MIT dataset, crucial for seizure detection, the model achieved an accuracy of 88.45%, underscoring its reliability in high-stakes clinical applications. On the PhyAAt dataset, leveraging both physical and physiological data, the model maintained high performance, with accuracy reaching 89.60%. These results collectively affirm the adaptability of our architecture across domains.</p>
<p>The proposed framework&#x00027;s integration of convolutional and recurrent components, coupled with tailored augmentations and regularization strategies, distinguishes it from existing SOTA approaches. For instance, while ViT (Touvron et al., <xref ref-type="bibr" rid="B36">2022</xref>) and BLIP (Wattasseril et al., <xref ref-type="bibr" rid="B38">2023</xref>) excel in certain contexts, their lack of recurrent layers limits their capacity for temporal modeling. Similarly, CLIP (Zhang et al., <xref ref-type="bibr" rid="B43">2025</xref>) and Wav2Vec 2.0 (Chen and Rudnicky, <xref ref-type="bibr" rid="B7">2023</xref>), relying on static embeddings, underperform in dynamic feature extraction tasks. This comparative analysis, complemented by <xref ref-type="fig" rid="F5">Figures 5</xref>, <xref ref-type="fig" rid="F6">6</xref>, illustrates the consistency and superior generalization of our model across diverse applications.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Performance comparison of SOTA methods on sleep-EDF dataset and EEGEyeNet dataset datasets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1540539-g0005.tif"/>
</fig>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Performance comparison of SOTA methods on CHB-MIT dataset and PhyAAt dataset datasets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1540539-g0006.tif"/>
</fig>
<p>The comparison across these benchmarks indicates that our proposed architecture&#x00027;s combination of convolutional and recurrent components, along with advanced data augmentation and regularization techniques, effectively generalizes across diverse domains. <xref ref-type="fig" rid="F5">Figures 5</xref>, <xref ref-type="fig" rid="F6">6</xref> illustrates the comparative metrics visually, affirming the consistency and robustness of the proposed model across multiple datasets. Notably, the superior performance of our model, especially in terms of AUC, emphasizes its reliability in high-stakes applications like medical diagnostics and human-computer interaction. The SOTA models, while competitive, did not incorporate domain-specific augmentations or the temporal modeling precision facilitated by our architecture. For instance, ViT (Touvron et al., <xref ref-type="bibr" rid="B36">2022</xref>) and BLIP (Wattasseril et al., <xref ref-type="bibr" rid="B38">2023</xref>) perform well but lack the recurrent layers necessary to fully exploit temporal dependencies in EEG and physiological data. Models like CLIP (Zhang et al., <xref ref-type="bibr" rid="B43">2025</xref>) and Wav2Vec 2.0 (Chen and Rudnicky, <xref ref-type="bibr" rid="B7">2023</xref>), while effective for certain tasks, rely primarily on static embeddings, which may explain their lower performance on datasets requiring dynamic feature extraction.</p>
</sec>
<sec>
<title>4.4 Ablation study</title>
<p>To evaluate the contributions of individual components in our model, an ablation study was conducted on the Sleep-EDF, EEGEyeNet, CHB-MIT, and PhyAAt datasets. The results are summarized in <xref ref-type="table" rid="T3">Tables 3</xref>, <xref ref-type="table" rid="T4">4</xref>, which present the metrics for Accuracy, Recall, F1 Score, and AUC under different configurations. The configurations examined include the removal of specific components, denoted as &#x0201C;w./o. Hybrid Grammar Contextualization Engine&#x0201D;, &#x0201C;w./o. Adaptive Curriculum Design&#x0201D;, and &#x0201C;w./o. Dynamic Feedback Systems&#x0201D;, as well as the full model (&#x0201C;Ours&#x0201D;).</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Ablation study results for ours on sleep-EDF and EEGEyeNet datasets for emotion analysis.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center" colspan="4"><bold>Sleep-EDF dataset</bold></th>
<th valign="top" align="center" colspan="4"><bold>EEGEyeNet dataset</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="center"><bold>Accuracy</bold></td>
<td valign="top" align="center"><bold>Recall</bold></td>
<td valign="top" align="center"><bold>F1 Score</bold></td>
<td valign="top" align="center"><bold>AUC</bold></td>
<td valign="top" align="center"><bold>Accuracy</bold></td>
<td valign="top" align="center"><bold>Recall</bold></td>
<td valign="top" align="center"><bold>F1 Score</bold></td>
<td valign="top" align="center"><bold>AUC</bold></td>
</tr>
<tr>
<td valign="top" align="left">w./o. Hybrid Grammar Contextualization Engine</td>
<td valign="top" align="center">86.12 &#x000B1; 0.03</td>
<td valign="top" align="center">84.20 &#x000B1; 0.03</td>
<td valign="top" align="center">85.05 &#x000B1; 0.02</td>
<td valign="top" align="center">88.70 &#x000B1; 0.03</td>
<td valign="top" align="center">87.80 &#x000B1; 0.02</td>
<td valign="top" align="center">85.90 &#x000B1; 0.03</td>
<td valign="top" align="center">86.65 &#x000B1; 0.02</td>
<td valign="top" align="center">90.10 &#x000B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">w./o. Adaptive Curriculum Design</td>
<td valign="top" align="center">87.30 &#x000B1; 0.02</td>
<td valign="top" align="center">85.75 &#x000B1; 0.02</td>
<td valign="top" align="center">86.40 &#x000B1; 0.03</td>
<td valign="top" align="center">89.55 &#x000B1; 0.02</td>
<td valign="top" align="center">88.95 &#x000B1; 0.03</td>
<td valign="top" align="center">87.15 &#x000B1; 0.02</td>
<td valign="top" align="center">87.50 &#x000B1; 0.03</td>
<td valign="top" align="center">91.00 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">w./o. Dynamic Feedback Systems</td>
<td valign="top" align="center">88.05 &#x000B1; 0.03</td>
<td valign="top" align="center">86.70 &#x000B1; 0.02</td>
<td valign="top" align="center">87.30 &#x000B1; 0.02</td>
<td valign="top" align="center">90.10 &#x000B1; 0.03</td>
<td valign="top" align="center">89.45 &#x000B1; 0.02</td>
<td valign="top" align="center">87.70 &#x000B1; 0.02</td>
<td valign="top" align="center">88.10 &#x000B1; 0.03</td>
<td valign="top" align="center">91.50 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center">90.12 &#x000B1; 0.02</td>
<td valign="top" align="center">87.95 &#x000B1; 0.03</td>
<td valign="top" align="center">88.65 &#x000B1; 0.02</td>
<td valign="top" align="center">91.40 &#x000B1; 0.02</td>
<td valign="top" align="center">91.80 &#x000B1; 0.03</td>
<td valign="top" align="center">89.12 &#x000B1; 0.02</td>
<td valign="top" align="center">90.45 &#x000B1; 0.03</td>
<td valign="top" align="center">93.10 &#x000B1; 0.02</td>
</tr></tbody>
</table>
</table-wrap>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Ablation study results for ours on CHB-MIT and PhyAAt datasets for emotion analysis.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center" colspan="4"><bold>CHB-MIT dataset</bold></th>
<th valign="top" align="center" colspan="4"><bold>PhyAAt dataset</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="center"><bold>Accuracy</bold></td>
<td valign="top" align="center"><bold>Recall</bold></td>
<td valign="top" align="center"><bold>F1 Score</bold></td>
<td valign="top" align="center"><bold>AUC</bold></td>
<td valign="top" align="center"><bold>Accuracy</bold></td>
<td valign="top" align="center"><bold>Recall</bold></td>
<td valign="top" align="center"><bold>F1 Score</bold></td>
<td valign="top" align="center"><bold>AUC</bold></td>
</tr>
<tr>
<td valign="top" align="left">w./o. Hybrid Grammar Contextualization Engine</td>
<td valign="top" align="center">84.10 &#x000B1; 0.03</td>
<td valign="top" align="center">81.95 &#x000B1; 0.02</td>
<td valign="top" align="center">83.00 &#x000B1; 0.03</td>
<td valign="top" align="center">86.20 &#x000B1; 0.02</td>
<td valign="top" align="center">85.50 &#x000B1; 0.02</td>
<td valign="top" align="center">83.10 &#x000B1; 0.03</td>
<td valign="top" align="center">84.00 &#x000B1; 0.02</td>
<td valign="top" align="center">87.10 &#x000B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">w./o. Adaptive Curriculum Design</td>
<td valign="top" align="center">85.30 &#x000B1; 0.02</td>
<td valign="top" align="center">83.25 &#x000B1; 0.03</td>
<td valign="top" align="center">84.10 &#x000B1; 0.02</td>
<td valign="top" align="center">87.40 &#x000B1; 0.03</td>
<td valign="top" align="center">87.20 &#x000B1; 0.02</td>
<td valign="top" align="center">85.05 &#x000B1; 0.02</td>
<td valign="top" align="center">86.00 &#x000B1; 0.03</td>
<td valign="top" align="center">88.50 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">w./o. Dynamic Feedback Systems</td>
<td valign="top" align="center">86.45 &#x000B1; 0.03</td>
<td valign="top" align="center">84.50 &#x000B1; 0.02</td>
<td valign="top" align="center">85.40 &#x000B1; 0.03</td>
<td valign="top" align="center">88.30 &#x000B1; 0.03</td>
<td valign="top" align="center">88.30 &#x000B1; 0.03</td>
<td valign="top" align="center">86.45 &#x000B1; 0.03</td>
<td valign="top" align="center">87.30 &#x000B1; 0.02</td>
<td valign="top" align="center">89.50 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center">88.45 &#x000B1; 0.03</td>
<td valign="top" align="center">86.90 &#x000B1; 0.02</td>
<td valign="top" align="center">87.80 &#x000B1; 0.03</td>
<td valign="top" align="center">90.20 &#x000B1; 0.03</td>
<td valign="top" align="center">89.60 &#x000B1; 0.02</td>
<td valign="top" align="center">87.85 &#x000B1; 0.03</td>
<td valign="top" align="center">88.45 &#x000B1; 0.02</td>
<td valign="top" align="center">91.10 &#x000B1; 0.02</td>
</tr></tbody>
</table>
</table-wrap>
<p>For the Sleep-EDF dataset, the removal of Hybrid Grammar Contextualization Engine resulted in a drop in accuracy from 90.12% to 86.12%, indicating the importance of this component in capturing essential sleep features. Similarly, the absence of Adaptive Curriculum Design reduced the AUC from 91.40% to 89.55%, highlighting its role in enhancing class separability. The full model outperformed all variations, achieving an F1 score of 88.65% and a recall of 87.95%, which underscores the synergistic effect of all components working in concert. A similar pattern was observed on the EEGEyeNet dataset, where the full model achieved an accuracy of 91.80% and an AUC of 93.10%, with noticeable declines when any component was excluded. These results demonstrate the importance of comprehensive feature extraction and temporal modeling strategies. For the CHB-MIT dataset, the removal of Hybrid Grammar Contextualization Engine caused a decrease in accuracy from 88.45% to 84.10% and a drop in recall from 86.90% to 81.95%. This finding suggests that Hybrid Grammar Contextualization Engine significantly contributes to identifying seizure events, likely by capturing critical temporal dynamics. Removing Dynamic Feedback Systems, which is designed to integrate multi-modal features, resulted in a reduction in AUC from 90.20% to 88.30%. This emphasizes the importance of multi-modal embeddings in achieving robust performance in seizure detection. On the PhyAAt dataset, the absence of Adaptive Curriculum Design reduced the recall from 87.85% to 85.05%, revealing its role in refining activity-specific features. The full model consistently achieved the best results, with an AUC of 91.10%, demonstrating its effectiveness in leveraging both physiological and physical signals.</p>
<p>The findings from the ablation study affirm that each component in our model architecture plays a critical role in optimizing performance. Hybrid Grammar Contextualization Engine likely enhances temporal feature extraction, while Adaptive Curriculum Design contributes to fine-grained feature refinement. Dynamic Feedback Systems integrates multi-modal inputs, enabling the model to learn complex relationships across signal domains. The superior performance of the full model validates the design decisions made in the architecture, highlighting its potential for applications in emotion analysis, seizure detection, and activity recognition. <xref ref-type="fig" rid="F7">Figures 7</xref>, <xref ref-type="fig" rid="F8">8</xref> visually compare the ablation study metrics, providing further insights into the impact of each component. These visualizations illustrate the consistent advantage of the full model across all datasets, reinforcing its robustness and generalizability in diverse contexts.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Ablation study of our method on sleep-EDF dataset and EEGEyeNet dataset datasets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1540539-g0007.tif"/>
</fig>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Ablation study of our method on CHB-MIT dataset and PhyAAt dataset datasets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1540539-g0008.tif"/>
</fig>
<p>To evaluate the cross-linguistic generalizability of our proposed framework, we conducted additional experiments using the SEED dataset, which consists of EEG recordings from Mandarin-speaking participants performing language comprehension tasks. <xref ref-type="table" rid="T5">Table 5</xref> presents the comparative performance of our model against six state-of-the-art baselines, consistent with those used in the main experiments. Our framework achieves the highest accuracy (88.75%), F1 score (87.20%), and AUC (89.95%) across all models, outperforming both audio-language models such as Wav2Vec 2.0 and T5, as well as vision-language models like CLIP and BLIP adapted to textual features. The robust performance observed on a Mandarin-language EEG dataset indicates that the architecture&#x00027;s multimodal alignment and cultural embedding mechanisms are effective beyond English, supporting its broader application to multilingual and culturally diverse populations. These findings further validate the adaptability of the DLEM and CALS components to non-English contexts, reinforcing the model&#x00027;s potential for global deployment in language-related cognitive modeling under environmental stressors.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Comparison of ours with SOTA methods on SEED dataset for Chinese listening task.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="left" colspan="4"><bold>SEED dataset (Mandarin listening)</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="left"><bold>Accuracy</bold></td>
<td valign="top" align="left"><bold>Recall</bold></td>
<td valign="top" align="left"><bold>F1 Score</bold></td>
<td valign="top" align="left"><bold>AUC</bold></td>
</tr>
<tr>
<td valign="top" align="left">CLIP (Zhang et al., <xref ref-type="bibr" rid="B43">2025</xref>)</td>
<td valign="top" align="left">83.35 &#x000B1; 0.02</td>
<td valign="top" align="left">80.20 &#x000B1; 0.03</td>
<td valign="top" align="left">81.85 &#x000B1; 0.02</td>
<td valign="top" align="left">85.40 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">ViT (Touvron et al., <xref ref-type="bibr" rid="B36">2022</xref>)</td>
<td valign="top" align="left">84.12 &#x000B1; 0.03</td>
<td valign="top" align="left">81.55 &#x000B1; 0.02</td>
<td valign="top" align="left">82.90 &#x000B1; 0.02</td>
<td valign="top" align="left">86.10 &#x000B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">I3D (Peng et al., <xref ref-type="bibr" rid="B23">2023</xref>)</td>
<td valign="top" align="left">82.80 &#x000B1; 0.02</td>
<td valign="top" align="left">79.85 &#x000B1; 0.02</td>
<td valign="top" align="left">81.30 &#x000B1; 0.03</td>
<td valign="top" align="left">84.70 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">BLIP (Wattasseril et al., <xref ref-type="bibr" rid="B38">2023</xref>)</td>
<td valign="top" align="left">85.10 &#x000B1; 0.03</td>
<td valign="top" align="left">82.30 &#x000B1; 0.03</td>
<td valign="top" align="left">83.45 &#x000B1; 0.02</td>
<td valign="top" align="left">87.00 &#x000B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">Wav2Vec 2.0 (Chen and Rudnicky, <xref ref-type="bibr" rid="B7">2023</xref>)</td>
<td valign="top" align="left">82.60 &#x000B1; 0.02</td>
<td valign="top" align="left">79.90 &#x000B1; 0.03</td>
<td valign="top" align="left">81.05 &#x000B1; 0.03</td>
<td valign="top" align="left">84.80 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">T5 (Grover et al., <xref ref-type="bibr" rid="B12">2021</xref>)</td>
<td valign="top" align="left">85.50 &#x000B1; 0.03</td>
<td valign="top" align="left">82.60 &#x000B1; 0.02</td>
<td valign="top" align="left">83.90 &#x000B1; 0.02</td>
<td valign="top" align="left">87.30 &#x000B1; 0.03</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="left">88.75 &#x000B1; 0.02</td>
<td valign="top" align="left">86.40 &#x000B1; 0.03</td>
<td valign="top" align="left">87.20 &#x000B1; 0.02</td>
<td valign="top" align="left">89.95 &#x000B1; 0.02</td>
</tr></tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s5">
<title>5 Conclusions and future work</title>
<p>Exploring the EEG Representation of English Listening Comprehension Under Hypoxic ConditionsAll the files uploaded by the user have been fully loaded. Searching won&#x00027;t provide additional information. This study investigates the impact of hypoxic conditions on English listening comprehension, an area of growing relevance for cognitive performance in high-altitude environments. By addressing the limitations of traditional behavioral and physiological approaches, which often lack depth in capturing neural responses, the research introduces an innovative framework. This framework integrates EEG-based neural decoding with the Dynamic Linguistic Enhancement Model (DLEM), which enhances linguistic analysis through adaptive vocabulary, contextual grammar application, and cultural embedding. Using real-time EEG feedback, the study further employs the Contextual Augmented Learning Strategy (CALS) to adaptively optimize curriculum delivery. Experimental results confirm that this integrative approach improves comprehension accuracy and reduces cognitive load, offering significant implications for advancing education and cognitive resilience under environmental stressors. The findings underscore the potential of leveraging physiological insights for scalable educational strategies in hypoxic conditions.</p>
<p>Despite these promising results, the study acknowledges two primary limitations. The generalizability of the findings is constrained by the controlled experimental settings, which may not fully replicate the complexity of real-world high-altitude environments. Future research should explore longitudinal field studies to validate the framework across diverse contexts. One of the limitations of the current study is the absence of EEG data obtained from native English speakers who are long-term residents of high-altitude environments. The publicly available datasets we employed, while robust in terms of signal quality and annotation, do not provide metadata regarding participants&#x00027; environmental exposure or geographic location. This limits our ability to compare EEG patterns across populations with different degrees of acclimatization to hypoxia. Consequently, the observed neural responses primarily reflect the effects of acute hypoxic conditions simulated in laboratory environments. It is possible that individuals who have adapted to chronic high-altitude exposure exhibit distinct electrophysiological characteristics, such as altered baseline oxygenation, neurovascular coupling, or cognitive compensation mechanisms. These adaptations could modulate EEG markers of linguistic processing in ways not captured by our current experimental design. We recognize this as a valuable future direction and plan to conduct targeted EEG data collection in high-altitude regions, focusing on native English-speaking populations. Such an extension would allow for stratified comparisons and could validate the generalizability of our findings to real-world high-altitude educational and occupational contexts. Incorporating this demographic would enhance the ecological validity of our framework and provide a more comprehensive understanding of cognitive resilience under hypoxia. While the EEG-based approach provides valuable granularity, its reliance on advanced technological infrastructure poses challenges for widespread implementation in resource-limited settings. Future work could focus on developing more accessible and cost-effective EEG technologies or alternative biomarkers to ensure broader applicability.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>YS: Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing, Data curation, Methodology, Supervision, Conceptualization, Formal analysis, Project administration, Validation, Investigation, Funding acquisition, Resources, Visualization, Software. YY: Data curation, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing, Visualization, Supervision, Funding acquisition.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research and/or publication of this article.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p></sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Agung</surname> <given-names>A. S. S. N.</given-names></name> <name><surname>Surtikanti</surname> <given-names>M. W.</given-names></name></person-group> (<year>2020</year>). <article-title>Students&#x00027; perception of online learning during covid-19 pandemic: a case study on the English students of STKIP Pamane Talino</article-title>. <source>SOSHUM</source> : <italic>Jurnal Sosial dan Humaniora</italic>. <volume>10</volume>:<fpage>2</fpage>. <pub-id pub-id-type="doi">10.31940/soshum.v10i2.1316</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ahuja</surname> <given-names>C.</given-names></name> <name><surname>Setia</surname> <given-names>D.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Measuring human auditory attention with EEG,&#x0201D;</article-title> in <source>2022 14th International Conference on COMmunication Systems</source> &#x00026; <italic>NETworkS (COMSNETS)</italic> (Bangalore: IEEE), <fpage>774</fpage>&#x02013;<lpage>778</lpage>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alfallaj</surname> <given-names>F.</given-names></name> <name><surname>Al-Ma&#x00027;amari</surname> <given-names>A. A. H.</given-names></name> <name><surname>Aldhali</surname> <given-names>F. I. A.</given-names></name></person-group> (<year>2021</year>). <article-title>Retracted: education of university students &#x02013; cultural perceptions on technology of English learning</article-title>. <source>Int. J. Elect. Eng. Educ</source>. <volume>60</volume>:<fpage>1</fpage>. <pub-id pub-id-type="doi">10.1177/0020720920984323</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aoyama</surname> <given-names>R.</given-names></name></person-group> (<year>2021</year>). <article-title>Language teacher identity and english education policy in japan: competing discourses surrounding &#x0201C;non-native&#x0201D; English-speaking teachers</article-title>. <source>RELC J</source>. <pub-id pub-id-type="doi">10.1177/00336882211032999</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ariastuti</surname> <given-names>M. D.</given-names></name> <name><surname>Wahyudin</surname> <given-names>A. Y.</given-names></name></person-group> (<year>2022</year>). <article-title>Exploring academic performance and learning style of undergraduate students in English education program</article-title>. <source>J. English lang. Teach. Learn</source>. <volume>3</volume>, <fpage>67</fpage>&#x02013;<lpage>73</lpage>. <pub-id pub-id-type="doi">10.33365/jeltl.v3i1.1817</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bae</surname> <given-names>S.</given-names></name> <name><surname>Park</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>Investing in the future: Korean early English education as neoliberal management of youth</article-title>. <source>Multilingua</source> 39. <pub-id pub-id-type="doi">10.1515/multi-2019-0009</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>L.-W.</given-names></name> <name><surname>Rudnicky</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Exploring wav2vec 2.0 fine tuning for improved speech emotion recognition,&#x0201D;</article-title> in <source>ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</source> (<publisher-loc>Rhodes Island</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>Y.</given-names></name></person-group> (<year>2021</year>). <article-title>College English teaching quality evaluation system based on information fusion and optimized RBF neural network decision algorithm</article-title>. <source>J. Sensors</source>. <pub-id pub-id-type="doi">10.1155/2021/6178569</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Coleman</surname> <given-names>J. J.</given-names></name></person-group> (<year>2021</year>). <article-title>Research: Affective reader response: using ordinary affects to repair literacy normativities in ELA and English education</article-title>. <source>English Educ</source>. <volume>53</volume>, <fpage>254</fpage>&#x02013;<lpage>276</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://publicationsncte.org/content/journals/10.58680/ee202131482?crawler=true">https://publicationsncte.org/content/journals/10.58680/ee202131482?crawler=true</ext-link></citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Duan</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Qiao</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Huang</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>B.</given-names></name></person-group> (<year>2021</year>). <article-title>An automatic method for epileptic seizure detection based on deep metric learning</article-title>. <source>IEEE J. Biomed. Health Inform</source>. <volume>26</volume>, <fpage>2147</fpage>&#x02013;<lpage>2157</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2021.3138852</pub-id><pub-id pub-id-type="pmid">34962890</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elliott</surname> <given-names>V.</given-names></name> <name><surname>Hodgson</surname> <given-names>J.</given-names></name></person-group> (<year>2021</year>). <article-title>Setting an agenda for English education research</article-title>. <source>English Educ</source>. <volume>55</volume>, <fpage>369</fpage>&#x02013;<lpage>374</lpage>. <pub-id pub-id-type="doi">10.1080/04250494.2021.1978737</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Grover</surname> <given-names>K.</given-names></name> <name><surname>Kaur</surname> <given-names>K.</given-names></name> <name><surname>Tiwari</surname> <given-names>K.</given-names></name> <name><surname>Rupali</surname> <given-names>Kumar, P.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Deep learning based question generation using t5 transformer,&#x0201D;</article-title> in <source>Advanced Computing: 10th International Conference, IACC 2020</source> (<publisher-loc>Panaji</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>243</fpage>&#x02013;<lpage>255</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hendriks-Balk</surname> <given-names>M. C.</given-names></name> <name><surname>Megdiche</surname> <given-names>F.</given-names></name> <name><surname>Pezzi</surname> <given-names>L.</given-names></name> <name><surname>Reynaud</surname> <given-names>O.</given-names></name> <name><surname>Da Costa</surname> <given-names>S.</given-names></name> <name><surname>Bueti</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Brainstem correlates of a cold pressor test measured by ultra-high field fmri</article-title>. <source>Front. Neurosci</source>. <volume>14</volume>:<fpage>39</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2020.00039</pub-id><pub-id pub-id-type="pmid">32082112</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>L.</given-names></name> <name><surname>Yao</surname> <given-names>W.</given-names></name></person-group> (<year>2021</year>). <article-title>Retracted: Design and implementation of college English multimedia aided teaching resources</article-title>. <source>Int. J. Elect. Eng. Educ</source>. <volume>60</volume>:<fpage>1</fpage>. <pub-id pub-id-type="doi">10.1177/0020720920983517</pub-id></citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Iturriaga</surname> <given-names>R.</given-names></name></person-group> (<year>2023</year>). <article-title>Carotid body contribution to the physio-pathological consequences of intermittent hypoxia: role of nitro-oxidative stress and inflammation</article-title>. <source>J. Physiol</source>. <volume>601</volume>, <fpage>5495</fpage>&#x02013;<lpage>5507</lpage>. <pub-id pub-id-type="doi">10.1113/JP284112</pub-id><pub-id pub-id-type="pmid">37119020</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Iturriaga</surname> <given-names>R.</given-names></name> <name><surname>Castillo-Gal&#x000E1;n</surname> <given-names>S.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;The beneficial effect of the blockade of stim-activated trpc-orai channels on vascular remodeling and pulmonary hypertension induced by intermittent hypoxia is independent of oxidative stress,&#x0201D;</article-title> in <source>International Society for Arterial Chemoreception</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>53</fpage>&#x02013;<lpage>60</lpage>.<pub-id pub-id-type="pmid">37322335</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Iturriaga</surname> <given-names>R.</given-names></name> <name><surname>Pereyra</surname> <given-names>K.</given-names></name> <name><surname>Las Heras</surname> <given-names>A.</given-names></name> <name><surname>Diaz-Jara</surname> <given-names>E.</given-names></name> <name><surname>Del Rio</surname> <given-names>R.</given-names></name></person-group> (<year>2023</year>). <article-title>Carotid body ablation reduced the hypertension and the astrocyte activation in the nts induced by long-term exposure to chronic intermittent hypoxia</article-title>. <source>IBRO Neurosci. Rep</source>. <volume>15</volume>, <fpage>S689</fpage>&#x02013;<lpage>S690</lpage>. <pub-id pub-id-type="doi">10.1016/j.ibneur.2023.08.1388</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Karlen-Amarante</surname> <given-names>M.</given-names></name> <name><surname>Glovak</surname> <given-names>Z. T.</given-names></name> <name><surname>Huff</surname> <given-names>A.</given-names></name> <name><surname>Oliveira</surname> <given-names>L. M.</given-names></name> <name><surname>Ramirez</surname> <given-names>J.-M.</given-names></name></person-group> (<year>2024</year>). <article-title>Postinspiratory and preb&#x000F6;tzinger complexes contribute to respiratory-sympathetic coupling in mice before and after chronic intermittent hypoxia</article-title>. <source>Front. Neurosci</source>. <volume>18</volume>:<fpage>1386737</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2024.1386737</pub-id><pub-id pub-id-type="pmid">38774786</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kashinathan</surname> <given-names>S.</given-names></name> <name><surname>Aziz</surname> <given-names>A. A.</given-names></name></person-group> (<year>2021</year>). <article-title>ESL learners&#x00027; challenges in speaking English in malaysian classroom</article-title>. <source>Int. J. Acad. Res. Prog. Educ. Dev</source>. <volume>10</volume>:<fpage>2</fpage>. <pub-id pub-id-type="doi">10.6007/IJARPED/v10-i2/10355</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>H.</given-names></name> <name><surname>Hwang</surname> <given-names>Y.</given-names></name></person-group> (<year>2022</year>). <article-title>Technology-enhanced education through VR-making and metaverse-linking to foster teacher readiness and sustainable learning</article-title>. <source>Sustainability</source>. <volume>14</volume>:<fpage>4786</fpage>. <pub-id pub-id-type="doi">10.3390/su14084786</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liang</surname> <given-names>Y.</given-names></name></person-group> (<year>2021</year>). <article-title>Retracted: College business english teaching in the context of multimedia network</article-title>. <source>Int. J. Elect. Eng. Educ</source>. <volume>60</volume>:<fpage>1</fpage>. <pub-id pub-id-type="doi">10.1177/00207209211007768</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Modesitt</surname> <given-names>E.</given-names></name> <name><surname>Yang</surname> <given-names>R.</given-names></name> <name><surname>Liu</surname> <given-names>Q.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Two heads are better than one: A bio-inspired method for improving classification on EEG-ET data,&#x0201D;</article-title> in <source>International Conference on Human-Computer Interaction</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>382</fpage>&#x02013;<lpage>390</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Peng</surname> <given-names>Y.</given-names></name> <name><surname>Lee</surname> <given-names>J.</given-names></name> <name><surname>Watanabe</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;I3D: Transformer architectures with input-dependent dynamic depth for speech recognition,&#x0201D;</article-title> in <source>ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</source> (<publisher-loc>Rhodes Island</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Renganathan</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>English language education in rural schools in Malaysia: a systematic review of research</article-title>. <source>Educ. Rev</source>. <volume>75</volume>, <fpage>787</fpage>&#x02013;<lpage>804</lpage>. <pub-id pub-id-type="doi">10.1080/00131911.2021.1931041</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Richards</surname> <given-names>J.</given-names></name> <name><surname>Pun</surname> <given-names>J. K. H.</given-names></name></person-group> (<year>2021</year>). <article-title>A typology of english-medium instruction</article-title>. <source>RELC J</source>. <pub-id pub-id-type="doi">10.1177/0033688220968584</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rusmiyanto</surname> <given-names>R.</given-names></name> <name><surname>Huriati</surname> <given-names>N.</given-names></name> <name><surname>Fitriani</surname> <given-names>N.</given-names></name> <name><surname>Tyas</surname> <given-names>N. K.</given-names></name> <name><surname>Rofi&#x00027;i</surname> <given-names>A.</given-names></name> <name><surname>Sari</surname> <given-names>M. N.</given-names></name></person-group> (<year>2023</year>). <article-title>The role of artificial intelligence (AI) in developing english language learner&#x00027;s communication skills</article-title>. <source>J. Educ</source>. <volume>6</volume>, <fpage>750</fpage>&#x02013;<lpage>757</lpage>. <pub-id pub-id-type="doi">10.31004/joe.v6i1.2990</pub-id><pub-id pub-id-type="pmid">37842718</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sallam</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>ChatGPT utility in healthcare education, research, and practice: Systematic review on the promising perspectives and valid concerns</article-title>. <source>Healthcare</source>. <volume>11</volume>:<fpage>887</fpage>. <pub-id pub-id-type="doi">10.3390/healthcare11060887</pub-id><pub-id pub-id-type="pmid">36981544</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Seo</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>An emerging trend in english education in Korea: &#x02013; maternal English education&#x00027; (eommapyo yeongeo)</article-title>. <source>English Today</source>. <volume>37</volume>, <fpage>163</fpage>&#x02013;<lpage>168</lpage>. <pub-id pub-id-type="doi">10.1017/S0266078420000048</pub-id></citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Septiyanti</surname> <given-names>M.</given-names></name> <name><surname>Inderawati</surname> <given-names>R.</given-names></name> <name><surname>Vianty</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>Technological pedagogical and content knowledge (TPACK) perception of English education students</article-title>. <source>Engl. Rev</source>. <volume>8</volume>, <fpage>165</fpage>&#x02013;<lpage>174</lpage>. <pub-id pub-id-type="doi">10.25134/erjee.v8i2.2114</pub-id><pub-id pub-id-type="pmid">35651576</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shaikh</surname> <given-names>S.</given-names></name> <name><surname>Yildirim</surname> <given-names>S. Y.</given-names></name> <name><surname>Klimova</surname> <given-names>B.</given-names></name> <name><surname>Pikhart</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>Assessing the usability of chatgpt for formal English language learning</article-title>. <source>Eur. J. Investigat. Health, Psychol. Educ</source>. <volume>13</volume>, <fpage>1937</fpage>&#x02013;<lpage>1960</lpage>. <pub-id pub-id-type="doi">10.3390/ejihpe13090140</pub-id><pub-id pub-id-type="pmid">37754479</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sihn</surname> <given-names>D.</given-names></name> <name><surname>Kim</surname> <given-names>S.-P.</given-names></name></person-group> (<year>2022</year>). <article-title>Brain infraslow activity correlates with arousal levels</article-title>. <source>Front. Neurosci</source>. <volume>16</volume>:<fpage>765585</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2022.765585</pub-id><pub-id pub-id-type="pmid">35281492</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Simamora</surname> <given-names>M. W. B.</given-names></name> <name><surname>Oktaviani</surname> <given-names>L.</given-names></name></person-group> (<year>2020</year>). <article-title>What is your favorite movie: a strategy of English education students to improve English vocabulary</article-title>. <source>J. Engl. Lang. Teach. Learn</source>. <volume>1</volume>:<fpage>2</fpage>. <pub-id pub-id-type="doi">10.33365/jeltl.v1i2.604</pub-id></citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sofyan</surname> <given-names>N.</given-names></name></person-group> (<year>2021</year>). <article-title>The role of English as global language</article-title>. <source>Edukasi</source>. <volume>3</volume>:<fpage>2</fpage>. <pub-id pub-id-type="doi">10.33387/j.edu.v19i1.3200</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>Z.</given-names></name> <name><surname>Anbarasan</surname> <given-names>M.</given-names></name> <name><surname>Kumar</surname> <given-names>D. P.</given-names></name> <name><surname>Kumar</surname> <given-names>P.</given-names></name></person-group> (<year>2020</year>). <article-title>Design of online intelligent English teaching platform based on artificial intelligence techniques</article-title>. <source>Int. Conf. Climate Inform</source>. <volume>37</volume>, <fpage>1166</fpage>&#x02013;<lpage>1180</lpage>. <pub-id pub-id-type="doi">10.1111/coin.12351</pub-id></citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Syakur</surname> <given-names>A.</given-names></name> <name><surname>Fanani</surname> <given-names>Z.</given-names></name> <name><surname>Ahmadi</surname> <given-names>R.</given-names></name></person-group> (<year>2020</year>). <article-title>The effectiveness of reading English learning process based on blended learning through &#x0201C;absyak&#x0201D; website media in higher education</article-title>. <source>Budapest Int. Res. Critics Linguist. Educ. J</source>. <volume>3</volume>, <fpage>763</fpage>&#x02013;<lpage>772</lpage>. <pub-id pub-id-type="doi">10.33258/birle.v3i2.927</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Touvron</surname> <given-names>H.</given-names></name> <name><surname>Cord</surname> <given-names>M.</given-names></name> <name><surname>J&#x000E9;gou</surname> <given-names>H.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Deit III: Revenge of the VIT,&#x0201D;</article-title> in <source>European Conference on Computer Vision</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>516</fpage>&#x02013;<lpage>533</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name></person-group> (<year>2024</year>). <article-title>&#x0201C;A multi-modal framework with contrastive learning and sequential encoding for enhanced sleep stage detection,&#x0201D;</article-title> in <source>Chinese Conference on Pattern Recognition and Computer Vision (PRCV)</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>3</fpage>&#x02013;<lpage>17</lpage>.</citation>
</ref>
<ref id="B38">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wattasseril</surname> <given-names>J. I.</given-names></name> <name><surname>Shekhar</surname> <given-names>S.</given-names></name> <name><surname>D&#x000F6;llner</surname> <given-names>J.</given-names></name> <name><surname>Trapp</surname> <given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Zero-shot video moment retrieval using blip-based models,&#x0201D;</article-title> in <source>International Symposium on Visual Computing</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>160</fpage>&#x02013;<lpage>171</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>F.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Han</surname> <given-names>D.</given-names></name></person-group> (<year>2022</year>). <article-title>Development countermeasures of college English education based on deep learning and artificial intelligence</article-title>. <source>Mobile Inform. Syst</source>. <pub-id pub-id-type="doi">10.1155/2022/8389800</pub-id></citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yao</surname> <given-names>Y.</given-names></name> <name><surname>Ma</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>Retracted: A multimedia network English listening teaching model based on confidence learning algorithm of speech recognition</article-title>. <source>Int. J. Elect. Eng. Educ</source>. <volume>60</volume>, <fpage>52</fpage>&#x02013;<lpage>59</lpage>. <pub-id pub-id-type="doi">10.1177/0020720920984678</pub-id></citation>
</ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yunita</surname> <given-names>W.</given-names></name> <name><surname>Maisarah</surname> <given-names>I.</given-names></name></person-group> (<year>2020</year>). <article-title>Students&#x00027; perception on learning language at the graduate program of English education amids the covid 19 pandemic</article-title>. <source>Linguists: J. Linguist. Lang. Teach</source>. <volume>6</volume>:<fpage>6</fpage>. <pub-id pub-id-type="doi">10.29300/ling.v6i2.3718</pub-id></citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zein</surname> <given-names>S.</given-names></name> <name><surname>Sukyadi</surname> <given-names>D.</given-names></name> <name><surname>Hamied</surname> <given-names>F. A.</given-names></name> <name><surname>Lengkanawati</surname> <given-names>N.</given-names></name></person-group> (<year>2020</year>). <article-title>English language education in Indonesia: A review of research (2011&#x02013;2019)</article-title>. <source>Lang. Teach</source>. <volume>53</volume>, <fpage>491</fpage>&#x02013;<lpage>523</lpage>. <pub-id pub-id-type="doi">10.1017/S0261444820000208</pub-id></citation>
</ref>
<ref id="B43">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>B.</given-names></name> <name><surname>Zhang</surname> <given-names>P.</given-names></name> <name><surname>Dong</surname> <given-names>X.</given-names></name> <name><surname>Zang</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name></person-group> (<year>2025</year>). <article-title>&#x0201C;Long-clip: unlocking the long-text capability of clip,&#x0201D;</article-title> in <source>European Conference on Computer Vision</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>310</fpage>&#x02013;<lpage>325</lpage>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>S.</given-names></name> <name><surname>Su</surname> <given-names>Z.</given-names></name> <name><surname>Miao</surname> <given-names>G.</given-names></name></person-group> (<year>2020</year>). <article-title>Retracted: Application of English education information management system based on convolution neural network classification algorithm</article-title>. <source>Int. J. Elect. Eng. Educ</source>. <volume>60</volume>:<fpage>1</fpage>. <pub-id pub-id-type="doi">10.1177/0020720920940614</pub-id></citation>
</ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>P.</given-names></name> <name><surname>Jiang</surname> <given-names>J.</given-names></name> <name><surname>Jiang</surname> <given-names>T.</given-names></name></person-group> (<year>2021</year>). <article-title>Retracted: A contrastive study of Chinese and English values in English film teaching</article-title>. <source>Int. J. Elect. Eng. Educ</source>. <volume>60</volume>:<fpage>1</fpage>. <pub-id pub-id-type="doi">10.1177/0020720920983540</pub-id></citation>
</ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname> <given-names>C.</given-names></name> <name><surname>Li</surname> <given-names>P.</given-names></name> <name><surname>Jin</surname> <given-names>L.</given-names></name></person-group> (<year>2021</year>). <article-title>Online college english education in Wuhan against the covid-19 pandemic: student and teacher readiness, challenges and implications</article-title>. <source>PLoS ONE</source>. <volume>16</volume>:<fpage>e0258137</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0258137</pub-id><pub-id pub-id-type="pmid">34597337</pub-id></citation></ref>
</ref-list>
</back>
</article>