<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Educ.</journal-id>
<journal-title>Frontiers in Education</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Educ.</abbrev-journal-title>
<issn pub-type="epub">2504-284X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/feduc.2025.1614673</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Education</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>AI vs. teacher feedback on EFL argumentative writing: a quantitative study</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Alnemrat</surname> <given-names>Areen</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Aldamen</surname> <given-names>Hesham</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2159352/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Almashour</surname> <given-names>Mohamad</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2332087/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Al-Deaibes</surname> <given-names>Mutasim</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2086236/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>AlSharefeen</surname> <given-names>Rami</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/3067478/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of English Language and Literature, Yarmouk University</institution>, <addr-line>Irbid</addr-line>, <country>Jordan</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of English Language and Literature, The University of Jordan</institution>, <addr-line>Amman</addr-line>, <country>Jordan</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of English, American University of Sharjah</institution>, <addr-line>Sharjah</addr-line>, <country>United Arab Emirates</country></aff>
<aff id="aff4"><sup>4</sup><institution>Rabdan Academy</institution>, <addr-line>Abu Dhabi</addr-line>, <country>United Arab Emirates</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Siv Gamlem, Volda University College, Norway</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Irina Engeness, &#x00D8;stfold University College, Norway</p>
<p>Mohammad Mahyoob Albuhairy, Taibah University, Saudi Arabia</p></fn>
<corresp id="c001">&#x002A;Correspondence: Mutasim Al-Deaibes, <email>deaibesm@gmail.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>29</day>
<month>07</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>10</volume>
<elocation-id>1614673</elocation-id>
<history>
<date date-type="received">
<day>22</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>06</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Alnemrat, Aldamen, Almashour, Al-Deaibes and AlSharefeen.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Alnemrat, Aldamen, Almashour, Al-Deaibes and AlSharefeen</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>This study investigates the effectiveness of AI-generated feedback compared to teacher-generated feedback on the argumentative writing performance of English as a Foreign Language (EFL) learners at different proficiency levels.</p>
</sec>
<sec>
<title>Methods</title>
<p>Sixty undergraduate students from a writing-focused EFL course in Jordan participated in a quasi-experimental, pretest-posttest study. Participants were stratified into two ACTFL proficiency levels (Intermediate-Low and Advanced-Low) and assigned to either an AI feedback group or a teacher feedback group. Students completed an argumentative writing task, received feedback based on their group, and revised their essays accordingly. An analytic rubric was used to assess writing performance, and inter-rater reliability was established on a stratified 30% subsample to support the validity of the scoring process, with pre- and post-test scores analyzed for gains.</p>
</sec>
<sec>
<title>Results</title>
<p>Results showed significant improvement in writing performance across all groups, regardless of feedback source or proficiency level. Importantly, no statistically significant difference was found between the AI and teacher feedback groups, and the effect size for this comparison was small (Cohen&#x2019;s d = 0.10). A two-way ANOVA revealed a significant main effect for proficiency level but no significant interaction between feedback type and proficiency. Intermediate-Low learners demonstrated the greatest within-group gains, suggesting that both feedback types were particularly impactful for lower-proficiency students.</p>
</sec>
<sec>
<title>Discussion</title>
<p>The findings underscore the potential of large language models (LLMs), when carefully scaffolded and ethically deployed, to support writing development in EFL contexts. AI-generated feedback may serve as a scalable complement to teacher feedback in large, mixed-proficiency classrooms, particularly when guided by well-developed prompts and pedagogical oversight.</p>
</sec>
</abstract>
<kwd-group>
<kwd>AI-generated feedback</kwd>
<kwd>argumentative writing</kwd>
<kwd>EFL learners</kwd>
<kwd>large language models (LLMs)</kwd>
<kwd>second language writing</kwd>
<kwd>mixed-proficiency classrooms</kwd>
</kwd-group>
<counts>
<fig-count count="2"/>
<table-count count="6"/>
<equation-count count="0"/>
<ref-count count="41"/>
<page-count count="11"/>
<word-count count="7725"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Assessment, Testing and Applied Measurement</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>1 Introduction</title>
<p>Argumentative writing proficiency constitutes a fundamental component of academic literacy acquisition for learners of English as a Foreign Language (<xref ref-type="bibr" rid="B41">Zhu, 2001</xref>), involving not only formulating texts characterized with linguistic accuracy but also constructing coherent and persuasive evidence-based arguments, integrating counterarguments, and maintaining a logical flow (<xref ref-type="bibr" rid="B9">Ferretti and Graham, 2019</xref>; <xref ref-type="bibr" rid="B31">Su et al., 2023</xref>; <xref ref-type="bibr" rid="B32">Su et al., 2021</xref>). Effective feedback represents a pivotal mechanism in the developmental trajectory of argumentative writing competence, with instructor-mediated evaluative commentary traditionally serving as the primary intervention modality for facilitating rhetorical enhancement and structural coherence (<xref ref-type="bibr" rid="B2">Banihashem et al., 2022</xref>; <xref ref-type="bibr" rid="B40">Zhang and Hyland, 2018</xref>). However, providing effective feedback to learners is significantly constrained by structural barriers inherent in contemporary educational settings characterized with high student-to-teacher ratios and heterogeneous proficiency distribution within the same classroom. These systemic limitations transform personalized evaluative feedback into a resource-intensive pedagogical intervention, frequently resulting in delayed feedback and diminished instructional efficacy (<xref ref-type="bibr" rid="B10">Guo et al., 2024</xref>; <xref ref-type="bibr" rid="B21">Liu et al., 2024</xref>; <xref ref-type="bibr" rid="B25">Ouahidi, 2021</xref>; <xref ref-type="bibr" rid="B36">Wisniewski et al., 2020</xref>). The advent of Generative Artificial Intelligence (GenAI) in educational contexts has introduced novel avenues for feedback delivery. AI-powered tools, such as ChatGPT, can generate immediate, detailed, and personalized feedback on student writing, potentially alleviating the workload of educators and providing timely assistance to learners (<xref ref-type="bibr" rid="B10">Guo et al., 2024</xref>; <xref ref-type="bibr" rid="B19">Lee and Moore, 2024</xref>). However, there is limited empirical research on how GenAI tools may assist in scaling feedback, especially in large English as a Foreign Language (EFL) writing classes (<xref ref-type="bibr" rid="B20">Li et al., 2024</xref>). For example, <xref ref-type="bibr" rid="B34">Wang and Dang&#x2019;s (2024)</xref> systematic review pointed to a significant dearth of empirical investigations examining applications of GenAI in EFL settings. Within the limited scholarly discourse, studies have begun to explore the efficacy of AI-generated feedback in comparison to traditional teacher feedback. For instance, research indicates that GenAI feedback can be as effective as teacher feedback in improving certain aspects of writing, such as coherence and cohesion (<xref ref-type="bibr" rid="B39">Yoon et al., 2023</xref>). However, concerns persist regarding the depth, accuracy, and contextual appropriateness of AI-generated feedback, particularly in addressing higher-order writing skills (<xref ref-type="bibr" rid="B6">Chan and Hu, 2023</xref>; <xref ref-type="bibr" rid="B39">Yoon et al., 2023</xref>). Moreover, the integration of AI into educational feedback mechanisms raises ethical considerations, notably regarding data privacy, the potential for overreliance on technology and plagiarism (<xref ref-type="bibr" rid="B30">S&#x00E1;nchez-Vera et al., 2024</xref>). Ensuring that AI tools are used responsibly, and that student data is protected is paramount (<xref ref-type="bibr" rid="B38">Yan et al., 2024</xref>). In the Arab world, very few studies looked at the impact of GenAI on higher education settings (<xref ref-type="bibr" rid="B8">Fadlelmula and Qadhi, 2024</xref>). This study aims to fill this gap by investigating the comparative effectiveness of AI-generated feedback and teacher-generated feedback on the argumentative writing performance of EFL students in Jordanian settings. Specifically, it seeks to determine whether AI feedback can match or surpass the quality and impact of traditional teacher feedback in enhancing students&#x2019; abilities to construct and refine arguments, integrate counterarguments, and maintain logical coherence in their writing. The questions this paper aim to answer are as follows:</p>
<list list-type="simple">
<list-item>
<label>1.</label>
<p>To what extent does the type of feedback (AI-generated vs. teacher-generated) influence EFL students&#x2019; improvement in argumentative writing performance?</p>
</list-item>
<list-item>
<label>2.</label>
<p>To what extent does language proficiency level (Intermediate-Low vs. Advanced-Low) affect writing performance after revision?</p>
</list-item>
<list-item>
<label>3.</label>
<p>Is there an interaction between feedback type and proficiency level in determining EFL students&#x2019; post-revision writing outcomes?</p>
</list-item>
</list>
<p>The findings of this study hold significant implications for EFL writing instruction. By elucidating the effectiveness of AI-generated feedback, educators can make informed decisions about integrating AI tools into their pedagogical practices. Understanding the comparative advantages and limitations of AI and teacher feedback can aid in designing hybrid feedback models that leverage the strengths of both approaches. Furthermore, addressing the ethical considerations associated with AI in education will contribute to the development of guidelines and policies that ensure responsible and effective use of technology in language learning contexts.</p>
</sec>
<sec id="S2">
<title>2 Literature review</title>
<p>Argumentative writing is widely recognized as one of the most cognitively and linguistically demanding genres in EFL instruction (<xref ref-type="bibr" rid="B9">Ferretti and Graham, 2019</xref>). It requires learners to formulate a clear stance, develop logical reasoning, integrate counterarguments, and adhere to academic discourse conventions (<xref ref-type="bibr" rid="B9">Ferretti and Graham, 2019</xref>; <xref ref-type="bibr" rid="B31">Su et al., 2023</xref>; <xref ref-type="bibr" rid="B32">Su et al., 2021</xref>). These requirements make argumentative writing particularly difficult for students whose linguistic proficiency is still developing (<xref ref-type="bibr" rid="B26">Pelenkahu et al., 2024</xref>). In the EFL context, students often struggle not only with surface-level features such as grammar and vocabulary but also with deeper genre-related demands, such as organizing their arguments logically and integrating rebuttals effectively (<xref ref-type="bibr" rid="B23">Mallahi, 2024</xref>). This dual challenge underscores the need for scaffolding that supports both linguistic and rhetorical development. Several studies emphasize that EFL learners commonly exhibit a &#x201C;one-sided&#x201D; argument structure, with minimal integration of opposing viewpoints (cf. <xref ref-type="bibr" rid="B14">He and Du, 2024</xref>; <xref ref-type="bibr" rid="B28">Qin and Karabacak, 2010</xref>; <xref ref-type="bibr" rid="B33">Wagner et al., 2017</xref>). Moreover, as learners tend to rely heavily on formulaic expressions, they demonstrate limited use of critical thinking strategies during the planning and revision stages of writing (<xref ref-type="bibr" rid="B9">Ferretti and Graham, 2019</xref>; <xref ref-type="bibr" rid="B18">Kuhn, 1991</xref>). These limitations are often exacerbated in mixed-ability classrooms, where less proficient students may lack the metacognitive strategies necessary to revise content meaningfully, while more proficient learners still require structured support for advanced rhetorical moves such as counterargument and rebuttal (<xref ref-type="bibr" rid="B13">Hazaea, 2023</xref>).</p>
<p>Feedback has been identified as one of the most pivotal pedagogical tools in improving argumentative writing skills. According to <xref ref-type="bibr" rid="B12">Hattie and Timperley</xref>&#x2019;s (<xref ref-type="bibr" rid="B12">2007</xref>, p. 86) model, feedback clarifies goals (<italic>feed up</italic>), points out progress (<italic>feed back</italic>), and closes gaps between current achievement and learning outcomes (<italic>feed forward</italic>). Numerous studies confirm that high-quality, formative feedback significantly enhances learners&#x2019; writing performance, especially when it is timely, specific, and focused on meaning-level aspects (<xref ref-type="bibr" rid="B4">Bitchener and Ferris, 2012</xref>; <xref ref-type="bibr" rid="B27">Peltzer et al., 2024</xref>; <xref ref-type="bibr" rid="B40">Zhang and Hyland, 2018</xref>). In argumentative writing, feedback is particularly crucial for helping learners recognize logical gaps, improve coherence, and refine argumentative strategies (<xref ref-type="bibr" rid="B10">Guo et al., 2024</xref>). However, providing individualized feedback in large and mixed-ability classrooms remains a persistent challenge, especially in resource-constrained settings (<xref ref-type="bibr" rid="B21">Liu et al., 2024</xref>; <xref ref-type="bibr" rid="B25">Ouahidi, 2021</xref>; <xref ref-type="bibr" rid="B36">Wisniewski et al., 2020</xref>).</p>
<p>Recent advances in large language models (LLMs) such as ChatGPT have opened new possibilities for AI-supported writing instruction. Unlike traditional automated writing evaluation (AWE) tools, which primarily target grammar and syntax, LLMs can provide nuanced, contextualized, and semi-structured feedback on content-level aspects of writing (<xref ref-type="bibr" rid="B39">Yoon et al., 2023</xref>). ChatGPT, in particular, has shown promise in supporting learners with argument structure, organization, and even idea generation, especially when guided by carefully designed prompts (<xref ref-type="bibr" rid="B10">Guo et al., 2024</xref>). Studies by <xref ref-type="bibr" rid="B11">Guo et al. (2022)</xref>, <xref ref-type="bibr" rid="B22">Mahapatra (2024)</xref> show that AI-supported scaffolding can improve feedback quality, revision depth, and learner motivation. These tools are especially effective when students are trained to use structured prompts that focus AI output on genre-specific goals, such as argument strength and counterargument inclusion (<xref ref-type="bibr" rid="B24">Mollick and Mollick, 2023</xref>). In a study on chatbot-assisted argumentative writing, <xref ref-type="bibr" rid="B11">Guo et al. (2022)</xref> found that students who received AI support produced stronger arguments and engaged in deeper revision than those relying solely on peer feedback. These findings are echoed in research by <xref ref-type="bibr" rid="B22">Mahapatra (2024)</xref>, who found GenAI feedback to have a significant positive impact on student learning coupled with positive student perceptions (p. 13).</p>
<p>Feedback effectiveness in writing instruction is not solely determined by its quality or timing but also by how learners engage with, interpret, and apply the feedback they receive; a process often referred to as feedback uptake (<xref ref-type="bibr" rid="B5">Carless and Boud, 2018</xref>). The Feedback Engagement Model proposed by <xref ref-type="bibr" rid="B35">Winstone et al. (2017)</xref> emphasizes that feedback is inherently dialogic, requiring active learner agency for it to be pedagogically impactful. This model outlines key stages in feedback engagement, including noticing, sense-making, and implementation each of which can vary significantly depending on whether the feedback originates from a human or an AI source.</p>
<p>When interacting with teacher-generated feedback, learners often benefit from interpersonal trust, contextual knowledge, and nuanced scaffolding that is aligned with classroom dynamics. However, such feedback may be delayed due to time constraints and can sometimes be inconsistent in tone or focus (<xref ref-type="bibr" rid="B40">Zhang and Hyland, 2018</xref>). By contrast, AI-generated feedback (e.g., from ChatGPT) offers immediacy and consistency, and can be tailored through prompt engineering to target specific rhetorical or genre-related issues (<xref ref-type="bibr" rid="B10">Guo et al., 2024</xref>). Yet, studies show that learners may interact with AI feedback more passively, often accepting suggestions without critical evaluation or reflection (<xref ref-type="bibr" rid="B39">Yoon et al., 2023</xref>). This passive uptake raises concerns about surface-level revision, overreliance, and reduced metacognitive engagement. Effective uptake of AI feedback, therefore, depends not only on the technical quality of the output but also on how learners are trained to critically engage with it. Furthermore, scaffolded reflection such as asking students to justify how they revised based on AI feedback can mitigate overreliance and encourage critical thinking (<xref ref-type="bibr" rid="B37">Woo et al., 2024</xref>). Research suggests that embedding reflective prompts (e.g., &#x201C;Which of these AI suggestions will you use, and why?&#x201D;) and requiring justification for revisions can enhance cognitive engagement and support deeper learning (<xref ref-type="bibr" rid="B24">Mollick and Mollick, 2023</xref>; <xref ref-type="bibr" rid="B35">Winstone et al., 2017</xref>). Ultimately, while both AI and teacher feedback can be effective, their pedagogical value is mediated by how learners perceive their credibility and interact with the feedback in the revision process.</p>
<p>However, the limitations of AI feedback are also widely acknowledged. <xref ref-type="bibr" rid="B39">Yoon et al. (2023)</xref> caution that while ChatGPT can generate plausible feedback on coherence and logic, it may also &#x201C;hallucinate&#x201D; critiques, offer generic advice, or miss contextually nuanced issues. There is also the risk that students may accept AI suggestions uncritically, leading to overreliance and potential plagiarism (cf. <xref ref-type="bibr" rid="B1">Alshurafat et al., 2024</xref>; <xref ref-type="bibr" rid="B7">Esmaeil et al., 2023</xref>; <xref ref-type="bibr" rid="B15">Hostetter et al., 2024</xref>; <xref ref-type="bibr" rid="B30">S&#x00E1;nchez-Vera et al., 2024</xref>). These limitations underscore the importance of prompt design, feedback framing, and teacher mediation in AI-supported writing instruction.</p>
</sec>
<sec id="S3">
<title>3 Methodology</title>
<sec id="S3.SS1">
<title>3.1 Research design</title>
<p>Following the recommendations outlined in <xref ref-type="bibr" rid="B29">Rose et al. (2019)</xref>, this study employed a quasi-experimental, pretest-posttest, between-subjects design to examine the impact of AI-generated and teacher-generated feedback on EFL students&#x2019; argumentative writing performance. The two independent variables were Feedback Type (AI-only vs. teacher-only) and Proficiency Level (Intermediate-Low vs. Advanced-Low). The dependent variable was the total score on an analytic rubric evaluating students&#x2019; performance on a single argumentative writing task. A 2 &#x00D7; 2 factorial design was used to assess both main effects and their interaction. Participants were stratified by proficiency level and then assigned to feedback conditions based on intact class groupings to maintain ecological validity. This design was chosen to reflect the realities of classroom implementation while preserving sufficient experimental control for statistical analysis.</p>
</sec>
<sec id="S3.SS2">
<title>3.2 Participants</title>
<p>The participants were 120 (83 females and 37 males) undergraduate EFL students enrolled in a writing-focused course at a large public university in Jordan. All were native speakers of Arabic and had received a minimum of 4 years of formal English instruction at the university level. Based on prior institutional placement procedures, including Oral Proficiency Interviews aligned with ACTFL guidelines, participants were classified as either Intermediate-Low or Advanced-Low in proficiency. To ensure balanced representation, stratified sampling was used to assign participants to two feedback conditions: AI-generated feedback and teacher-generated feedback, with equal distribution across proficiency levels. This resulted in four subgroups: AI/Intermediate-Low, AI/Advanced-Low, Teacher/Intermediate-Low, and Teacher/Advanced-Low each containing 30 students. Group assignment was not randomized but followed intact classroom sections to maintain ecological validity. Students were informed about the nature of the study, assured of confidentiality, and provided informed consent. The study received ethical approval from the university&#x2019;s Institutional Review Board.</p>
</sec>
<sec id="S3.SS3">
<title>3.3 Writing task</title>
<p>All students completed the same argumentative writing task, responding to the prompt: &#x201C;Should university education be free for all students?&#x201D; This topic was selected for its relevance, accessibility, and capacity to elicit critical reasoning, supporting the integration of claims, counterclaims, and rebuttals. Students were required to produce an essay of 250&#x2013;300 words within a 45 min time limit. Prior to the task, they received brief instruction on the structural expectations of argumentative writing, including thesis formulation, body development, and conclusion. Students submitted a first draft (Draft 1), received feedback according to group assignment, and then submitted a revised version (Draft 2) within 1 week.</p>
</sec>
<sec id="S3.SS4">
<title>3.4 Feedback conditions</title>
<sec id="S3.SS4.SSS1">
<title>3.4.1 AI feedback group</title>
<p>Participants in the AI group received feedback from ChatGPT (GPT-4) using a structured, piloted prompt designed to elicit genre-specific, rhetorical-level feedback on argumentative writing (see <xref ref-type="supplementary-material" rid="DS1">Supplementary Appendix C</xref> for representative samples of feedback). The AI prompt guided the model to focus on argument structure, clarity, use of counterarguments, and revision support without rewriting any part of the student&#x2019;s text. Students were trained to input their essays in privacy mode (&#x201C;chat history off&#x201D;) and instructed to apply the feedback independently. The full version of the tested AI mentor prompt, including the analytic rubric used to guide feedback interpretation, is provided in <xref ref-type="supplementary-material" rid="DS1">Supplementary Appendix A</xref>. The prompt was developed iteratively and grounded in recent AI pedagogy literature (e.g., <xref ref-type="bibr" rid="B24">Mollick and Mollick, 2023</xref>). The same rubric was used to guide feedback delivery in both the AI and teacher conditions to ensure consistency in focus, expectations, and assessment criteria (see <xref ref-type="supplementary-material" rid="DS1">Supplementary Appendix B</xref> for full rubric). The revised AI prompt was designed to simulate an interactive, step-by-step mentoring session, fostering a collaborative learning environment between the AI and the student. The process begins with the mentor gathering key contextual information about the student&#x2019;s writing goals, proficiency level, and specific areas of concern. By asking one question at a time and pausing for a response, the prompt supports a focused and responsive dialog, ensuring that feedback is tailored to the student&#x2019;s needs and aligned with their current level of development.</p>
<p>A central feature of the prompt is its emphasis on balanced, scaffolded feedback. The AI mentor is instructed to begin by identifying the strengths of the student&#x2019;s work, thereby establishing a supportive tone and recognizing effort. Constructive feedback follows, targeting rhetorical features such as argument clarity, evidence use, organization, and counterarguments. The student is then guided through the revision process, encouraged to apply feedback thoughtfully and to reflect on the changes made promoting metacognitive awareness and deeper engagement with the writing task. Moreover, the prompt supports iterative learning by allowing for follow-up feedback after revision. The AI mentor reviews the revised section, invites further reflection, and provides additional suggestions if needed. Whether the student feels ready to finalize the work or seeks continued support, the session concludes with affirming, forward-looking encouragement. This mentoring model not only reinforces writing development but also fosters learner autonomy, confidence, and a growth-oriented mindset.</p>
</sec>
<sec id="S3.SS4.SSS2">
<title>3.4.2 Teacher feedback group</title>
<p>Participants in the teacher group received individualized feedback from their course instructor. Comments were handwritten on printed copies of Draft 1, guided by a feedback checklist aligned with the rubric dimensions: argument clarity, supporting evidence, counterargument integration, organization, and language use. Feedback was formative, non-evaluative, and provided within 48 hours of submission. All students were encouraged to reflect on their feedback and revise their drafts accordingly.</p>
</sec>
</sec>
<sec id="S3.SS5">
<title>3.5 Instruments</title>
<sec id="S3.SS5.SSS1">
<title>3.5.1 Analytic writing rubric</title>
<p>An analytic rubric was developed and validated to assess argumentative writing performance. It included five equally weighted dimensions adapted from established frameworks used in prior EFL writing research (e.g., <xref ref-type="bibr" rid="B9">Ferretti and Graham, 2019</xref>; <xref ref-type="bibr" rid="B28">Qin and Karabacak, 2010</xref>):</p>
<list list-type="simple">
<list-item>
<label>1.</label>
<p>Claim Clarity and Relevance</p>
</list-item>
<list-item>
<label>2.</label>
<p>Support and Evidence</p>
</list-item>
<list-item>
<label>3.</label>
<p>Counterarguments and Rebuttal</p>
</list-item>
<list-item>
<label>4.</label>
<p>Organization and Coherence</p>
</list-item>
<list-item>
<label>5.</label>
<p>Language Use (Grammar and Vocabulary)</p>
</list-item>
</list>
<p>Each category was scored on a five-point scale (1 = very weak; 5 = excellent), with a maximum total score of 25. The rubric (see <xref ref-type="supplementary-material" rid="DS1">Supplementary Appendix B</xref> for full rubric) was reviewed by two L2 writing experts for content validity.</p>
</sec>
<sec id="S3.SS5.SSS2">
<title>3.5.2 Data collection procedure</title>
<p>The study protocol comprised six sequential phases designed to ensure methodological rigor and data integrity. Initially, all participants underwent a comprehensive orientation session during which they were informed of the study objectives, feedback mechanisms, and ethical protections governing their participation. Following this briefing, students completed their initial writing task (Draft 1) under standardized classroom conditions to ensure consistency across all participants. The intervention phase involved the systematic delivery of feedback according to predetermined group assignments, with participants receiving either AI-generated feedback through ChatGPT or traditional instructor feedback. Subsequently, students were afforded a 1 week period to independently revise their compositions and submit their final drafts (Draft 2), allowing for adequate reflection and incorporation of the provided feedback. Assessment procedures employed a validated analytic rubric administered by a primary rater who evaluated both initial and revised drafts. To establish inter-rater reliability and minimize scoring bias, a second trained evaluator independently assessed a stratified random sample comprising 36 essays (30% of the total corpus) using identical rubric criteria while remaining blind to group assignment conditions. Finally, all scoring data were systematically recorded in a structured Excel database, with entries organized by participant identification number, experimental group assignment, proficiency level classification, and pre-test (Draft 1) and post-test (Draft 2) performance scores.</p>
</sec>
<sec id="S3.SS5.SSS3">
<title>3.5.3 Inter-rater reliability procedures</title>
<p>To ensure robust assessment of rater agreement, both Pearson&#x2019;s correlation (r) and the Intra-Class Correlation Coefficient (ICC) were used, recognizing ICC as the preferred metric for ordinal scoring in educational research (<xref ref-type="bibr" rid="B17">Koo and Li, 2016</xref>). Pearson&#x2019;s <italic>r</italic> was calculated for initial comparison, yielding moderate agreement for pre-test scores (r = 0.653; MAD = 2.11) and lower agreement for post-test scores (r = 0.316; MAD = 2.25). However, ICC (2,1) estimates provided a more robust evaluation of inter-rater consistency, with the post-test ICC = 0.61, indicating moderate agreement. These results support the overall reliability of the scoring procedure, though some variability remained particularly at higher score ranges. Rater discrepancies were discussed <italic>post hoc</italic> to calibrate interpretations and ensure consistent rubric application (see <xref ref-type="table" rid="T1">Table 1</xref>).</p>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p>Inter-rater reliability metrics for writing scores (<italic>n</italic> = 36).</p></caption>
<table cellspacing="5" cellpadding="5" frame="box" rules="all">
<thead>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Metric</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Pre-test</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Post-test</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Pearson correlation (r)</td>
<td valign="top" align="left">0.653</td>
<td valign="top" align="left">0.316</td>
</tr>
<tr>
<td valign="top" align="left">Mean absolute difference</td>
<td valign="top" align="left">2.11</td>
<td valign="top" align="left">2.25</td>
</tr>
</tbody>
</table></table-wrap>
<p>As shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, Pre-test scores show moderate alignment between Rater 1 and Rater 2, clustering along the diagonal. <xref ref-type="fig" rid="F2">Figure 2</xref>, on the other hand, show more dispersion, especially at higher score ranges, indicating lower agreement.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>Pre-test scores.</p></caption>
<alt-text>Scatter plot titled &#x201C;Inter-Rater Agreement: Pre-Test Scores&#x201D; showing scores from two raters. The x-axis represents Rater 1 scores, and the y-axis represents Rater 2 scores. Data points are scattered around a dashed diagonal line indicating perfect agreement. Scores range from 10 to 24.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="feduc-10-1614673-g001.tif"/>
</fig>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Post-test scores.</p></caption>
<alt-text>Scatter plot showing inter-rater agreement on pre-test scores, with Rater 1 scores on the x-axis ranging from 10 to 24 and Rater 2 scores on the y-axis ranging from 10 to 24. Data points are scattered, and a dotted diagonal line indicates perfect agreement.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="feduc-10-1614673-g002.tif"/>
</fig>
</sec>
</sec>
<sec id="S3.SS6">
<title>3.6 Data analysis</title>
<p>Data analysis was conducted using Python programming language with specialized statistical libraries including pandas for data manipulation, scipy for statistical computations, and statsmodels for advanced statistical modeling. The analytical framework encompassed multiple complementary approaches to comprehensively evaluate the intervention effects. Descriptive analyses were initially performed to characterize the dataset, including computation of means, standard deviations, and score ranges stratified by experimental group and proficiency level classifications. To examine within-group performance changes, paired-sample <italic>t</italic>-tests were employed to assess the statistical significance of improvement from pre-intervention (Draft 1) to post-intervention (Draft 2) scores within each feedback condition. Between-group comparisons utilized independent-sample <italic>t</italic>-tests to evaluate differences in revision gains, calculated as the difference between post-test and pre-test scores, across the two feedback modalities. Additionally, a two-way analysis of variance (ANOVA) was implemented to simultaneously examine the main effects of feedback type and proficiency level, as well as their potential interaction, on post-intervention writing performance. Effect size calculations accompanied all inferential tests to assess practical significance beyond statistical significance, with Cohen&#x2019;s d computed for <italic>t</italic>-test comparisons and partial eta squared (&#x03B7;<sup>2</sup>) calculated for ANOVA results. Prior to conducting parametric analyses, fundamental statistical assumptions were rigorously evaluated through Shapiro-Wilk tests for normality of distributions and Levene&#x2019;s tests for homogeneity of variance across groups, ensuring the appropriateness of the selected analytical procedures.</p>
</sec>
<sec id="S3.SS7">
<title>3.7 Ethical considerations</title>
<p>This study complied with institutional and international guidelines for ethical research in education. Informed consent was obtained from all participants. No personally identifiable information was collected or shared with the AI tool. All AI interactions occurred with &#x201C;chat history&#x201D; disabled to avoid data retention. To mitigate risks associated with AI-generated feedback (e.g., hallucinations, generic responses), prompts were standardized and tested extensively prior to deployment. Students were explicitly instructed to critically evaluate the feedback they received and revise their work accordingly. Teacher support was available for clarification. All writing samples and scores were anonymized prior to analysis, and data were stored securely with access limited to the research team.</p>
</sec>
</sec>
<sec id="S4" sec-type="results">
<title>4 Results</title>
<sec id="S4.SS1">
<title>4.1 Descriptive statistics</title>
<p>A total of 120 participants were recruited for this study and included in the final analysis. The sample was systematically stratified to ensure balanced representation across two key variables: feedback modality (artificial intelligence-generated vs. teacher-provided feedback) and initial language proficiency level (Intermediate-Low vs. Advanced-Low, as determined by standardized placement assessments). This balanced factorial design resulted in equal cell sizes (<italic>n</italic> = 30 per condition), thereby optimizing statistical power and enabling robust between-group comparisons. <xref ref-type="table" rid="T2">Tables 2</xref>&#x2013;<xref ref-type="table" rid="T4">4</xref> present comprehensive descriptive statistics including means, standard deviations, and score ranges for pre-test performance, post-test performance, and gain scores (calculated as post-test minus pre-test scores) disaggregated by experimental condition and proficiency level. The descriptive data reveal several noteworthy patterns that warrant detailed examination. Across all experimental conditions and proficiency levels, participants demonstrated measurable improvement in writing performance from the initial draft (pre-test) to the revised draft (post-test). This universal pattern of improvement suggests that both AI-generated and teacher-provided feedback were effective in facilitating writing enhancement, regardless of students&#x2019; initial proficiency levels. The consistency of this finding across all subgroups provides preliminary evidence for the general efficacy of corrective feedback in second language writing contexts. The data also reveal a compelling interaction between initial proficiency level and learning outcomes. Intermediate-Low proficiency students, while demonstrating lower absolute scores on both pre-test and post-test measures compared to their Advanced-Low counterparts, exhibited notably higher average gain scores across both feedback conditions. This pattern suggests that students with lower initial proficiency may derive greater benefit from corrective feedback interventions, potentially due to greater room for improvement or increased sensitivity to explicit error correction at earlier stages of language development. Conversely, Advanced-Low proficiency students, despite achieving higher absolute performance scores, showed more modest gains from pre-test to post-test. This ceiling effect phenomenon is consistent with previous research in second language acquisition, which suggests that learners at higher proficiency levels may require more sophisticated or targeted interventions to achieve measurable improvement (cf. <xref ref-type="bibr" rid="B3">Biber et al., 2011</xref>).</p>
<table-wrap position="float" id="T2">
<label>TABLE 2</label>
<caption><p>Pre- and post-test means and standard deviations.</p></caption>
<table cellspacing="5" cellpadding="5" frame="box" rules="all">
<thead>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Group</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Proficiency</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Pre M (SD)</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Post M (SD)</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Gain M (SD)</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">AI</td>
<td valign="top" align="left">Advanced-low</td>
<td valign="top" align="left">17.23 (1.76)</td>
<td valign="top" align="left">20.4 (1.65)</td>
<td valign="top" align="left">3.17 (2.09)</td>
</tr>
<tr>
<td valign="top" align="left">AI</td>
<td valign="top" align="left">Intermediate-low</td>
<td valign="top" align="left">12.5 (1.72)</td>
<td valign="top" align="left">18.1 (1.54)</td>
<td valign="top" align="left">5.6 (2.18)</td>
</tr>
<tr>
<td valign="top" align="left">Teacher</td>
<td valign="top" align="left">Advanced-low</td>
<td valign="top" align="left">17.2 (1.71)</td>
<td valign="top" align="left">20.43 (1.65)</td>
<td valign="top" align="left">3.23 (2.33)</td>
</tr>
<tr>
<td valign="top" align="left">Teacher</td>
<td valign="top" align="left">Intermediate-low</td>
<td valign="top" align="left">12.4 (1.67)</td>
<td valign="top" align="left">17.43 (1.87)</td>
<td valign="top" align="left">5.03 (2.53)</td>
</tr>
</tbody>
</table></table-wrap>
<table-wrap position="float" id="T3">
<label>TABLE 3</label>
<caption><p>Score ranges.</p></caption>
<table cellspacing="5" cellpadding="5" frame="box" rules="all">
<thead>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Group</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Proficiency</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Pre min&#x2013;max</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Post min&#x2013;max</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Gain min&#x2013;max</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">AI</td>
<td valign="top" align="left">Advanced-low</td>
<td valign="top" align="left">15&#x2013;20</td>
<td valign="top" align="left">18&#x2013;23</td>
<td valign="top" align="left">&#x2212;1 to 7</td>
</tr>
<tr>
<td valign="top" align="left">AI</td>
<td valign="top" align="left">Intermediate-low</td>
<td valign="top" align="left">10&#x2013;15</td>
<td valign="top" align="left">15&#x2013;20</td>
<td valign="top" align="left">1&#x2013;10</td>
</tr>
<tr>
<td valign="top" align="left">Teacher</td>
<td valign="top" align="left">Advanced-low</td>
<td valign="top" align="left">15&#x2013;20</td>
<td valign="top" align="left">18&#x2013;22</td>
<td valign="top" align="left">&#x2212;1 to 7</td>
</tr>
<tr>
<td valign="top" align="left">Teacher</td>
<td valign="top" align="left">Intermediate-low</td>
<td valign="top" align="left">10&#x2013;15</td>
<td valign="top" align="left">15&#x2013;20</td>
<td valign="top" align="left">0&#x2013;10</td>
</tr>
</tbody>
</table></table-wrap>
<table-wrap position="float" id="T4">
<label>TABLE 4</label>
<caption><p>Descriptive statistics by group and proficiency.</p></caption>
<table cellspacing="5" cellpadding="5" frame="box" rules="all">
<thead>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Group</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Proficiency</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Pre_total_<break/>mean</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Pre_total_<break/>std</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Pre_total_<break/>min</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Pre_total_<break/>max</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Pre_total_<break/>count</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Post_total_<break/>mean</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Post_total_<break/>std</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Post_total_<break/>min</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Post_total_<break/>max</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Post_total_<break/>count</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Gain_<break/>mean</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Gain_<break/>std</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Gain_<break/>min</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Gain_<break/>max</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Gain_<break/>count</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">AI</td>
<td valign="top" align="left">Advanced-low</td>
<td valign="top" align="left">17.23</td>
<td valign="top" align="left">1.76</td>
<td valign="top" align="left">15.0</td>
<td valign="top" align="left">20.0</td>
<td valign="top" align="left">30.0</td>
<td valign="top" align="left">20.4</td>
<td valign="top" align="left">1.65</td>
<td valign="top" align="left">18.0</td>
<td valign="top" align="left">23.0</td>
<td valign="top" align="left">30.0</td>
<td valign="top" align="left">3.17</td>
<td valign="top" align="left">2.09</td>
<td valign="top" align="left">&#x2212;1.0</td>
<td valign="top" align="left">7.0</td>
<td valign="top" align="left">30.0</td>
</tr>
<tr>
<td valign="top" align="left">AI</td>
<td valign="top" align="left">Intermediate-low</td>
<td valign="top" align="left">12.5</td>
<td valign="top" align="left">1.72</td>
<td valign="top" align="left">10.0</td>
<td valign="top" align="left">15.0</td>
<td valign="top" align="left">30.0</td>
<td valign="top" align="left">18.1</td>
<td valign="top" align="left">1.54</td>
<td valign="top" align="left">15.0</td>
<td valign="top" align="left">20.0</td>
<td valign="top" align="left">30.0</td>
<td valign="top" align="left">5.6</td>
<td valign="top" align="left">2.18</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">10.0</td>
<td valign="top" align="left">30.0</td>
</tr>
<tr>
<td valign="top" align="left">Teacher</td>
<td valign="top" align="left">Advanced-low</td>
<td valign="top" align="left">17.2</td>
<td valign="top" align="left">1.71</td>
<td valign="top" align="left">15.0</td>
<td valign="top" align="left">20.0</td>
<td valign="top" align="left">30.0</td>
<td valign="top" align="left">20.43</td>
<td valign="top" align="left">1.65</td>
<td valign="top" align="left">18.0</td>
<td valign="top" align="left">22.0</td>
<td valign="top" align="left">30.0</td>
<td valign="top" align="left">3.23</td>
<td valign="top" align="left">2.33</td>
<td valign="top" align="left">&#x2212;1.0</td>
<td valign="top" align="left">7.0</td>
<td valign="top" align="left">30.0</td>
</tr>
<tr>
<td valign="top" align="left">Teacher</td>
<td valign="top" align="left">Intermediate-low</td>
<td valign="top" align="left">12.4</td>
<td valign="top" align="left">1.67</td>
<td valign="top" align="left">10.0</td>
<td valign="top" align="left">15.0</td>
<td valign="top" align="left">30.0</td>
<td valign="top" align="left">17.43</td>
<td valign="top" align="left">1.87</td>
<td valign="top" align="left">15.0</td>
<td valign="top" align="left">20.0</td>
<td valign="top" align="left">30.0</td>
<td valign="top" align="left">5.03</td>
<td valign="top" align="left">2.53</td>
<td valign="top" align="left">0.0</td>
<td valign="top" align="left">10.0</td>
<td valign="top" align="left">30.0</td>
</tr>
</tbody>
</table></table-wrap>
</sec>
<sec id="S4.SS2">
<title>4.2 Within-group comparisons (paired samples <italic>t</italic>-tests)</title>
<p>To evaluate whether the feedback conditions led to statistically significant improvement in writing, paired samples <italic>t</italic>-tests were conducted within each of the four subgroups. Cohen&#x2019;s d values were calculated to assess the magnitude of the change, as shown in <xref ref-type="table" rid="T5">Table 5</xref>. <xref ref-type="table" rid="T5">Table 5</xref> reveals that all groups demonstrated statistically significant improvement in writing scores from pre- to post-feedback, with large or very large effect sizes observed across conditions.</p>
<table-wrap position="float" id="T5">
<label>TABLE 5</label>
<caption><p>Paired <italic>t</italic>-test results and effect sizes by group and proficiency.</p></caption>
<table cellspacing="5" cellpadding="5" frame="box" rules="all">
<thead>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Group</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Proficiency</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">t-value</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;"><italic>P</italic>-value</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Cohen&#x2019;s d</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">AI</td>
<td valign="top" align="left">Advanced-low</td>
<td valign="top" align="left">8.32</td>
<td valign="top" align="left">0.0</td>
<td valign="top" align="left">1.518</td>
</tr>
<tr>
<td valign="top" align="left">AI</td>
<td valign="top" align="left">Intermediate-low</td>
<td valign="top" align="left">14.1</td>
<td valign="top" align="left">0.0</td>
<td valign="top" align="left">2.575</td>
</tr>
<tr>
<td valign="top" align="left">Teacher</td>
<td valign="top" align="left">Advanced-low</td>
<td valign="top" align="left">7.6</td>
<td valign="top" align="left">0.0</td>
<td valign="top" align="left">1.388</td>
</tr>
<tr>
<td valign="top" align="left">Teacher</td>
<td valign="top" align="left">Intermediate-low</td>
<td valign="top" align="left">10.92</td>
<td valign="top" align="left">0.0</td>
<td valign="top" align="left">1.993</td>
</tr>
</tbody>
</table></table-wrap>
</sec>
<sec id="S4.SS3">
<title>4.3 Between-Group comparison (independent samples <italic>t</italic>-test)</title>
<p>To assess whether AI or teacher feedback led to greater gains, an independent samples <italic>t</italic>-test was conducted on the gain scores across feedback groups: t(118) = 0.55, <italic>p</italic> = 0.586, Cohen&#x2019;s d = 0.10. The difference in gain scores between the AI and teacher feedback groups was not statistically significant, and the effect size was very small, suggesting practical equivalence.</p>
</sec>
<sec id="S4.SS4">
<title>4.4 Two-way ANOVA</title>
<p>A two-way ANOVA was conducted to test the main effects of Feedback Type and Proficiency Level, and their interaction on Post-Test Total Score, as shown in <xref ref-type="table" rid="T6">Table 6</xref>. The two-way analysis of variance yielded a statistically significant main effect for proficiency level. Post-revision performance scores were significantly higher among Advanced-Low participants compared to Intermediate-Low participants, indicating that language proficiency level was a meaningful predictor of writing quality following revision. With respect to feedback modality, the main effect of feedback type (AI-mediated versus instructor-provided) did not reach statistical significance. This finding suggests comparable efficacy between artificial intelligence and human instructor feedback on learners&#x2019; writing performance. The analysis further revealed no significant interaction effect between proficiency level and feedback type. The absence of a significant interaction indicates that the relative effectiveness of AI-mediated versus instructor-provided feedback remained consistent across proficiency levels.</p>
<table-wrap position="float" id="T6">
<label>TABLE 6</label>
<caption><p>Two-way analysis of variance (ANOVA) results.</p></caption>
<table cellspacing="5" cellpadding="5" frame="box" rules="all">
<thead>
<tr>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">Source</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">SS</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">df</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;">F</td>
<td valign="top" align="left" style="color:#ffffff;background-color: #7f8080;"><italic>P</italic></td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">C (group)</td>
<td valign="top" align="left">3.00833333333<break/>3277</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.06187240085<break/>19935</td>
<td valign="top" align="left">0.30493230613<break/>04629</td>
</tr>
<tr>
<td valign="top" align="left">C (proficiency)</td>
<td valign="top" align="left">210.675000000<break/>00322</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">74.3634242823<break/>8272</td>
<td valign="top" align="left">3.89081944533<break/>2791e-14</td>
</tr>
<tr>
<td valign="top" align="left">C (group):C (proficiency)</td>
<td valign="top" align="left">3.67500000000<break/>0084</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.29719038442<break/>03565</td>
<td valign="top" align="left">0.25707366633<break/>367534</td>
</tr>
<tr>
<td valign="top" align="left">Residual</td>
<td valign="top" align="left">328.633333333<break/>3333</td>
<td valign="top" align="left">116.0</td>
<td valign="top" align="left">Nan</td>
<td valign="top" align="left">Nan</td>
</tr>
</tbody>
</table></table-wrap>
</sec>
<sec id="S4.SS5">
<title>4.5 Summary of findings</title>
<p>Both AI-generated and teacher-generated feedback led to statistically significant improvements in students&#x2019; argumentative writing scores. However, no significant difference was observed in the gain scores between the two feedback groups, indicating comparable effectiveness. Proficiency level exerted a strong main effect, with Advanced-Low learners outperforming Intermediate-Low learners on the post-test. No significant interaction was found between feedback type and proficiency level. Effect size calculations underscored the practical relevance of these findings, particularly for Intermediate-Low learners who demonstrated the largest within-group gains. Although all essays were scored by the primary researcher, inter-rater reliability was established on a stratified 30% subsample. The results revealed moderate to acceptable agreement, especially for pre-test score supporting the reliability and validity of the scoring process.</p>
</sec>
</sec>
<sec id="S5" sec-type="discussion">
<title>5 Discussion</title>
<sec id="S5.SS1">
<title>5.1 Summary of main findings</title>
<p>This study set out to examine the impact of AI-generated versus teacher-generated feedback on the argumentative writing performance of EFL learners with different proficiency levels. Results revealed that both feedback types led to statistically significant improvement in writing scores across all participants, regardless of proficiency. However, no significant differences were found between the AI and teacher feedback groups in terms of overall writing gains. This finding disagrees with those drawn by <xref ref-type="bibr" rid="B20">Li et al. (2024)</xref>, who found that students who relied on AI feedback (experiment group) demonstrated more improvement than that relied on teacher feedback (control group). In contrast, proficiency level had a significant main effect, with Advanced-Low learners outperforming Intermediate-Low learners in the post-test. This finding comports with the systematic review conducted by <xref ref-type="bibr" rid="B3">Biber et al. (2011)</xref>, which demonstrated an inverse relationship between initial proficiency level and magnitude of performance gains, whereby students with lower baseline competencies exhibited proportionally greater improvement than their more proficient counterparts. Effect sizes were large across all groups, indicating substantial practical improvement, especially among Intermediate-Low learners who benefited most from the revision process. These findings contribute to a growing body of evidence suggesting that AI feedback, when structured and scaffolded effectively, can match the effectiveness of traditional teacher feedback in supporting meaning-level revisions in argumentative writing (e.g., <xref ref-type="bibr" rid="B10">Guo et al., 2024</xref>).</p>
</sec>
<sec id="S5.SS2">
<title>5.2 AI Feedback and writing development</title>
<p>The finding that AI-generated feedback performed comparably to teacher-generated feedback supports prior research on the instructional potential of large language models (LLMs) in EFL writing (<xref ref-type="bibr" rid="B11">Guo et al., 2022</xref>). Importantly, students in the AI group used tested prompts that directed the model to provide focused, rhetorical-level commentary that is targeting argument clarity, evidence, counterarguments, and coherence. This aligns with recent work emphasizing the critical role of prompt engineering in shaping the relevance and usefulness of AI feedback (<xref ref-type="bibr" rid="B24">Mollick and Mollick, 2023</xref>). The non-significant difference in gains between AI and Teacher groups, coupled with the small effect size (Cohen&#x2019;s d = 0.10), is particularly relevant for scalability in writing instruction. In contexts where teacher feedback is constrained by class size, workload, or time limitations, structured AI feedback can serve as a viable supplement or alternative, especially if embedded within a pedagogically sound writing process.</p>
</sec>
<sec id="S5.SS3">
<title>5.3 Proficiency level as a moderator</title>
<p>The significant effect of proficiency level highlights the importance of learner characteristics in shaping writing outcomes. While all groups improved significantly, Intermediate-Low learners exhibited the largest effect sizes in both feedback conditions. This suggests that when given access to clear, actionable feedback (whether from a human or an AI assistant) lower-proficiency learners can make dramatic improvements in their writing performance. These findings support research that emphasizes the role of feedback scaffolding in mixed-proficiency classrooms (<xref ref-type="bibr" rid="B16">Knoch et al., 2015</xref>). Interestingly, the absence of an interaction effect between feedback type and proficiency level suggests that AI feedback did not disadvantage less proficient learners, a concern raised in earlier research (<xref ref-type="bibr" rid="B39">Yoon et al., 2023</xref>). This outcome may be attributed to the training and guidance students received on how to interpret and apply AI feedback, as well as the structured prompts used to limit off-topic or generic responses.</p>
</sec>
</sec>
<sec id="S6">
<title>6 Conclusion, implications, limitations, and future research</title>
<p>This study explored the comparative effectiveness of AI-generated and teacher-generated feedback on EFL students&#x2019; argumentative writing performance across two ACTFL proficiency levels. Findings revealed that both feedback types led to statistically significant gains in writing scores, with no significant difference between the AI and teacher groups. This result, coupled with a small effect size for the between-group comparison, suggests that well-structured AI feedback (delivered through prompt engineering) can serve as a scalable and pedagogically meaningful alternative to traditional teacher feedback. Notably, proficiency level emerged as a significant predictor of post-revision performance, with Advanced-Low learners outperforming Intermediate-Low learners. However, the largest within-group effect sizes were observed among Intermediate-Low students, indicating their high responsiveness to structured revision support, regardless of the feedback source. The results underscore the growing potential of large language models (LLMs) in second language writing instruction, especially in contexts where teacher feedback is limited by class size or time constraints. At the same time, the findings reinforce the importance of embedding AI use within guided, ethical, and pedagogically sound frameworks. This study is not without limitations. Essays were scored by a single rater, and only immediate revision gains were assessed. Future studies should include multiple raters, examine long-term effects of AI-assisted revision, and incorporate learner reflections to better understand feedback uptake. As educational technologies evolve, it is imperative to continue evaluating how AI can complement rather than replace the human element in language teaching and learning. Despite the single-rater design, the inclusion of inter-rater reliability metrics for a representative subsample provides reasonable confidence in scoring validity and transparency.</p>
<p>The findings have significant implications for EFL writing instruction across diverse educational contexts. Large language models can serve as effective feedback partners, delivering genre-specific guidance that supports student revision while freeing instructors for higher-order pedagogical activities. Standardized prompt design focusing on genre conventions can mitigate AI inaccuracies and misdirection, while the scalable nature of AI feedback enables differentiated instruction in large, mixed-proficiency classrooms. However, AI should complement rather than replace teacher feedback, functioning as a preliminary revision tool that prepares students for subsequent instructor guidance. This tiered approach preserves essential human pedagogical expertise while leveraging technological capabilities to enhance instructional effectiveness and accessibility across varied resource contexts.</p>
<p>While this study offers valuable insights into the comparative effects of AI and teacher feedback, several limitations must be acknowledged. First, although most essays were scored by a single trained rater, inter-rater reliability was established through a dual-rating process on a stratified subsample of 36 essays. The inclusion of a second rater, use of a validated analytic rubric, and high transparency in scoring procedures help mitigate concerns of bias and subjectivity. Pre-test scores showed moderate agreement (r = 0.653), while post-test scores revealed more variability (r = 0.316), a common pattern in subjective assessment of revised writing. Several factors may have contributed to this issue. First, the nature of the post-test responses, produced after exposure to individualized feedback and revision, may have led to more diverse writing structures and strategies, increasing subjectivity in rating. Second, although raters used a shared rubric, differences in interpretation may have emerged when evaluating revisions. Discrepancies were resolved through discussion to ensure shared understanding of rubric dimensions. Nonetheless, future studies should consider full-scale dual scoring or use intra-class correlation (ICC) to assess agreement more robustly. To address this in future research, we recommend the implementation of full dual scoring for all writing samples rather than a subset, alongside rigorous rater calibration sessions prior to and during scoring. These practices can help improve consistency and minimize subjectivity in evaluating student writing, especially in post-intervention contexts where performance tends to be more heterogeneous.</p>
<p>Second, the study examined short-term effects of feedback on a single revision cycle. Longitudinal data would be needed to evaluate the durability of learning gains and the impact of sustained feedback engagement over time. Third, another limitation of this study is the absence of direct measurement of student feedback uptake. While our approach was guided by theoretical models of feedback engagement (e.g., <xref ref-type="bibr" rid="B35">Winstone et al., 2017</xref>), we did not collect empirical data on how students interpreted or applied the formative feedback provided. As a result, the study cannot account for individual differences in feedback engagement or clarify the specific ways in which feedback influenced revisions. Future research should consider incorporating qualitative and process-oriented methods, such as think-aloud protocols, student interviews, or digital revision tracking, to capture how learners interact with feedback and make use of it during revision. Lastly, a further limitation of this study lies in its use of a single argumentative writing prompt as the basis for data collection. While argumentative writing is an important academic genre, relying on a single task restricts the extent to which findings can be generalized to other types of writing, such as narrative, expository, or reflective genres. Moreover, genre-specific features may influence how students interpret and apply feedback, meaning that the observed effects of the intervention may not transfer uniformly across contexts. Future research should replicate this study using a broader range of writing genres and disciplinary tasks to assess whether the effectiveness of rubric-aligned, AI-generated feedback varies by genre or subject area.</p>
<p>Although post-test scores showed lower inter-rater correlation, the inclusion of a second rater on a stratified subsample mitigates the risk of single-rater bias and supports the reliability of scoring. While ICC estimates indicated moderate agreement (ICC = 0.61), full-scale dual scoring across all essays would further enhance the reliability and generalizability of the findings, particularly in the context of post-revision assessment where scoring tends to be more variable.</p>
<p>Future research endeavors should investigate several critical dimensions to advance understanding of AI-assisted writing instruction. Longitudinal studies examining the cumulative effects of iterative AI-supported writing cycles would provide valuable insights into sustained performance trajectories and skill development patterns over extended periods. Additionally, comparative analyses of differentiated AI prompt strategies, including structure-oriented, error-correction focused, and tone-modification approaches, would elucidate the relative efficacy of targeted feedback modalities in facilitating specific aspects of revision behavior. Investigations into student attitudes and confidence levels regarding AI-generated versus instructor-provided feedback represent another essential research direction, particularly given the implications for pedagogical acceptance and implementation. Furthermore, the incorporation of qualitative methodologies would substantially enhance the depth of understanding regarding student engagement with AI feedback systems. Specifically, stimulated recall interviews and reflective journaling protocols could illuminate the cognitive processes underlying students&#x2019; interpretation, evaluation, and application of AI-generated suggestions, a domain that remains significantly underexplored in current literature. Such mixed-methods approaches would provide crucial insights into the mechanisms through which AI feedback influences writing development and inform evidence-based best practices for educational implementation.</p>
</sec>
</body>
<back>
<sec id="S7" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in this study are included in this article/<xref ref-type="supplementary-material" rid="DS1">Supplementary material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="S8" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The studies involving humans were approved by University of Jordan Board of Ethics. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study. Written informed consent was obtained from the individual(s) for the publication of any potentially identifiable images or data included in this article.</p>
</sec>
<sec id="S9" sec-type="author-contributions">
<title>Author contributions</title>
<p>AA: Writing &#x2013; review and editing, Writing &#x2013; original draft. HA: Writing &#x2013; review and editing, Writing &#x2013; original draft. MA: Writing &#x2013; review and editing, Writing &#x2013; original draft. MA-D: Writing &#x2013; original draft, Writing &#x2013; review and editing. RA: Writing &#x2013; original draft, Writing &#x2013; review and editing.</p>
</sec>
<sec id="S10" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research and/or publication of this article.</p>
</sec>
<sec id="S11" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="S12" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The authors declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="S13" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="S14" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/feduc.2025.1614673/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/feduc.2025.1614673/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.docx" id="DS1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alshurafat</surname> <given-names>H.</given-names></name> <name><surname>Al Shbail</surname> <given-names>M. O.</given-names></name> <name><surname>Hamdan</surname> <given-names>A.</given-names></name> <name><surname>Al-Dmour</surname> <given-names>A.</given-names></name> <name><surname>Ensour</surname> <given-names>W.</given-names></name></person-group> (<year>2024</year>). <article-title>Factors affecting accounting students&#x2019; misuse of chatgpt: An application of the fraud triangle theory.</article-title> <source><italic>J. Financial Report. Account.</italic></source> <volume>22</volume> <fpage>274</fpage>&#x2013;<lpage>288</lpage>. <pub-id pub-id-type="doi">10.1108/JFRA-04-2023-0182</pub-id></citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Banihashem</surname> <given-names>S. K.</given-names></name> <name><surname>Noroozi</surname> <given-names>O.</given-names></name> <name><surname>Van Ginkel</surname> <given-names>S.</given-names></name> <name><surname>Macfadyen</surname> <given-names>L. P.</given-names></name> <name><surname>Biemans</surname> <given-names>H. J.</given-names></name></person-group> (<year>2022</year>). <article-title>A systematic review of the role of learning analytics in enhancing feedback practices in higher education.</article-title> <source><italic>Educ. Res. Rev.</italic></source> <volume>37</volume>:<fpage>100489</fpage>. <pub-id pub-id-type="doi">10.1016/j.edurev.2022.100489</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Biber</surname> <given-names>D.</given-names></name> <name><surname>Nekrasova</surname> <given-names>T.</given-names></name> <name><surname>Horn</surname> <given-names>B.</given-names></name></person-group> (<year>2011</year>). <article-title>The effectiveness of feedback for L1-English and L2-writing development: A meta-analysis.</article-title> <source><italic>ETS Res. Rep. Ser.</italic></source> <volume>2011</volume> <fpage>i</fpage>&#x2013;<lpage>99</lpage>. <pub-id pub-id-type="doi">10.1002/j.2333-8504.2011.tb02241.x</pub-id></citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bitchener</surname> <given-names>J.</given-names></name> <name><surname>Ferris</surname> <given-names>D. R.</given-names></name></person-group> (<year>2012</year>). <source><italic>Written corrective feedback in second language acquisition and writing.</italic></source> <publisher-loc>London</publisher-loc>: <publisher-name>Routledge</publisher-name>.</citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Carless</surname> <given-names>D.</given-names></name> <name><surname>Boud</surname> <given-names>D.</given-names></name></person-group> (<year>2018</year>). <article-title>The development of student feedback literacy: Enabling uptake of feedback.</article-title> <source><italic>Assess. Eval. High. Educ.</italic></source> <volume>43</volume> <fpage>1315</fpage>&#x2013;<lpage>1325</lpage>. <pub-id pub-id-type="doi">10.1080/02602938.2018.1463354</pub-id></citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chan</surname> <given-names>C. K. Y.</given-names></name> <name><surname>Hu</surname> <given-names>W.</given-names></name></person-group> (<year>2023</year>). <article-title>Students&#x2019; voices on generative AI: Perceptions, benefits, and challenges in higher education.</article-title> <source><italic>Int. J. Educ. Technol. High. Educ.</italic></source> <volume>20</volume> <fpage>43</fpage>&#x2013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1186/s41239-023-00411-8</pub-id></citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Esmaeil</surname> <given-names>A.-A.-A.</given-names></name> <name><surname>Dzulkifli</surname> <given-names>D.</given-names></name> <name><surname>Maakip</surname> <given-names>I.</given-names></name> <name><surname>Matanluk</surname> <given-names>O.</given-names></name> <name><surname>Marshall</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <article-title>Understanding student perception regarding the use of ChatGPT in their argumentative writing: A qualitative inquiry.</article-title> <source><italic>J. Komunikasi Malays. J. Commun.</italic></source> <volume>39</volume> <fpage>150</fpage>&#x2013;<lpage>165</lpage>. <pub-id pub-id-type="doi">10.17576/JKMJC-2023-3904-08</pub-id> <pub-id pub-id-type="pmid">39963415</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fadlelmula</surname> <given-names>F.</given-names></name> <name><surname>Qadhi</surname> <given-names>S.</given-names></name></person-group> (<year>2024</year>). <article-title>A systematic review of research on artificial intelligence in higher education: Practice, gaps, and future directions in the GCC.</article-title> <source><italic>J. Univ. Teach. Learn. Pract.</italic></source> <volume>21</volume> <fpage>146</fpage>&#x2013;<lpage>173</lpage>. <pub-id pub-id-type="doi">10.53761/pswgbw82</pub-id></citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ferretti</surname> <given-names>R. P.</given-names></name> <name><surname>Graham</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>Argumentative writing: Theory, assessment, and instruction.</article-title> <source><italic>Read. Writ.</italic></source> <volume>32</volume> <fpage>1345</fpage>&#x2013;<lpage>1357</lpage>. <pub-id pub-id-type="doi">10.1007/s11145-019-09950-x</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>K.</given-names></name> <name><surname>Pan</surname> <given-names>M.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Lai</surname> <given-names>C.</given-names></name></person-group> (<year>2024</year>). <article-title>Effects of an AI-supported approach to peer feedback on university EFL students&#x2019; feedback quality and writing ability.</article-title> <source><italic>Int. High. Educ.</italic></source> <volume>63</volume>:<fpage>100962</fpage>. <pub-id pub-id-type="doi">10.1016/j.iheduc.2024.100962</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>K.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Chu</surname> <given-names>S. K. W.</given-names></name></person-group> (<year>2022</year>). <article-title>Using chatbots to scaffold EFL students&#x2019; argumentative writing.</article-title> <source><italic>Assess. Writ.</italic></source> <volume>54</volume>:<fpage>100666</fpage>. <pub-id pub-id-type="doi">10.1016/j.asw.2022.100666</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hattie</surname> <given-names>J.</given-names></name> <name><surname>Timperley</surname> <given-names>H.</given-names></name></person-group> (<year>2007</year>). <article-title>The power of feedback.</article-title> <source><italic>Rev. Educ. Res.</italic></source> <volume>77</volume> <fpage>81</fpage>&#x2013;<lpage>112</lpage>. <pub-id pub-id-type="doi">10.3102/003465430298487</pub-id> <pub-id pub-id-type="pmid">38293548</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hazaea</surname> <given-names>A. N.</given-names></name></person-group> (<year>2023</year>). <article-title>Process-Genre approach in mixed-ability classes: Correlational study between EFL academic paragraph reading and writing.</article-title> <source><italic>Novitas-Royal (Research on Youth and Language)</italic></source> <volume>17</volume> <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.5281/zenodo.10015742</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>H.</given-names></name> <name><surname>Du</surname> <given-names>Y.</given-names></name></person-group> (<year>2024</year>). &#x201C;<article-title>The effectiveness of dialogical argumentation in supporting low-level EAP learners&#x2019; evidence-based writing: A longitudinal study</article-title>,&#x201D; in <source><italic>English for academic purposes in the EMI context in Asia: XJTLU impact</italic></source>, <role>eds</role> <person-group person-group-type="editor"><name><surname>Zou</surname> <given-names>B.</given-names></name> <name><surname>Mahy</surname> <given-names>T.</given-names></name></person-group> (<publisher-loc>Switzerland</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>45</fpage>&#x2013;<lpage>75</lpage>.</citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hostetter</surname><given-names>A. B.</given-names></name> <name><surname>Call</surname> <given-names>N.</given-names></name> <name><surname>Frazier</surname> <given-names>G.</given-names></name> <name><surname>James</surname> <given-names>T.</given-names></name> <name><surname>Linnertz</surname> <given-names>C.</given-names></name> <name><surname>Nestle</surname> <given-names>E.</given-names></name><etal/></person-group> (<year>2024</year>). <article-title>Student and faculty perceptions of generative artificial intelligence in student writing.</article-title> <source><italic>Teach. Psychol.</italic></source> <volume>52</volume> <fpage>319</fpage>&#x2013;<lpage>329</lpage>. <pub-id pub-id-type="doi">10.1177/00986283241279401</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Knoch</surname> <given-names>U.</given-names></name> <name><surname>Rouhshad</surname> <given-names>A.</given-names></name> <name><surname>Oon</surname> <given-names>S. P.</given-names></name> <name><surname>Storch</surname> <given-names>N.</given-names></name></person-group> (<year>2015</year>). <article-title>What happens to ESL students&#x2019; writing after three years of study at an English medium university?</article-title> <source><italic>J. Sec. Lang. Writ.</italic></source> <volume>28</volume> <fpage>39</fpage>&#x2013;<lpage>52</lpage>. <pub-id pub-id-type="doi">10.1016/j.jslw.2015.02.005</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Koo</surname> <given-names>T. K.</given-names></name> <name><surname>Li</surname> <given-names>M. Y.</given-names></name></person-group> (<year>2016</year>). <article-title>A guideline of selecting and reporting intraclass correlation coefficients for reliability research.</article-title> <source><italic>J. Chiropract. Med.</italic></source> <volume>15</volume> <fpage>155</fpage>&#x2013;<lpage>163</lpage>. <pub-id pub-id-type="doi">10.1016/j.jcm.2016.02.012</pub-id> <pub-id pub-id-type="pmid">27330520</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kuhn</surname> <given-names>D.</given-names></name></person-group> (<year>1991</year>). <source><italic>The skills of argument.</italic></source> <publisher-loc>Cambridge</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>.</citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>S. S.</given-names></name> <name><surname>Moore</surname> <given-names>R. L.</given-names></name></person-group> (<year>2024</year>). <article-title>Harnessing Generative AI (GenAI) for automated feedback in higher education: A systematic review.</article-title> <source><italic>Online Learn.</italic></source> <volume>28</volume> <fpage>82</fpage>&#x2013;<lpage>106</lpage>. <pub-id pub-id-type="doi">10.24059/olj.v28i3.4593</pub-id> <pub-id pub-id-type="pmid">33692645</pub-id></citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Luo</surname> <given-names>S.</given-names></name> <name><surname>Huang</surname> <given-names>C.</given-names></name></person-group> (<year>2024</year>). <article-title>The influence of GenAI on the effectiveness of argumentative writing in higher education: Evidence from a quasi-experimental study in China.</article-title> <source><italic>J. Asian Public Pol.</italic></source> <volume>18</volume>, <fpage>405</fpage>&#x2013;<lpage>430</lpage>. <pub-id pub-id-type="doi">10.1080/17516234.2024.2363128</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Xiong</surname> <given-names>W.</given-names></name> <name><surname>Xiong</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>Y.-F. B.</given-names></name></person-group> (<year>2024</year>). <article-title>Generating timely individualized feedback to support student learning of conceptual knowledge in Writing-To-Learn activities.</article-title> <source><italic>J. Comp. Educ.</italic></source> <volume>11</volume> <fpage>367</fpage>&#x2013;<lpage>399</lpage>. <pub-id pub-id-type="doi">10.1007/s40692-023-00261-3</pub-id></citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mahapatra</surname> <given-names>S.</given-names></name></person-group> (<year>2024</year>). <article-title>Impact of ChatGPT on ESL students&#x2019; academic writing skills: A mixed methods intervention study.</article-title> <source><italic>Smart Learn. Environ.</italic></source> <volume>11</volume>:<fpage>9</fpage>. <pub-id pub-id-type="doi">10.1186/s40561-024-00295-9</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mallahi</surname> <given-names>O.</given-names></name></person-group> (<year>2024</year>). <article-title>Exploring the status of argumentative essay writing strategies and problems of Iranian EFL learners.</article-title> <source><italic>Asian-Pacific J. Sec. For. Lang. Educ.</italic></source> <volume>9</volume>:<fpage>19</fpage>. <pub-id pub-id-type="doi">10.1186/s40862-023-00241-1</pub-id></citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mollick</surname> <given-names>E.</given-names></name> <name><surname>Mollick</surname> <given-names>L.</given-names></name></person-group> (<year>2023</year>). <article-title>Assigning AI: Seven approaches for students, with prompts.</article-title> <source><italic>arXiv [Preprint]</italic></source> <pub-id pub-id-type="doi">10.48550/arXiv.2306.10052</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ouahidi</surname> <given-names>L. M.</given-names></name></person-group> (<year>2021</year>). <article-title>Teaching writing to tertiary EFL large classes: Challenges and prospects.</article-title> <source><italic>Int. J. Linguist. Literat. Trans.</italic></source> <volume>4</volume> <fpage>28</fpage>&#x2013;<lpage>35</lpage>. <pub-id pub-id-type="doi">10.32996/ijllt.2021.4.6.5</pub-id></citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pelenkahu</surname> <given-names>N.</given-names></name> <name><surname>Ali</surname> <given-names>M. I.</given-names></name> <name><surname>Tatipang</surname> <given-names>D. P.</given-names></name> <name><surname>Wuntu</surname> <given-names>C. N.</given-names></name> <name><surname>Rorintulus</surname> <given-names>O. A.</given-names></name></person-group> (<year>2024</year>). <article-title>Metacognitive strategies and critical thinking in elevating EFL argumentative writing proficiency: Practical insights.</article-title> <source><italic>Stud. Eng. Lang. Educ.</italic></source> <volume>11</volume> <fpage>873</fpage>&#x2013;<lpage>892</lpage>. <pub-id pub-id-type="doi">10.24815/siele.v11i2.35832</pub-id></citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peltzer</surname> <given-names>K.</given-names></name> <name><surname>Lorca</surname> <given-names>A. L.</given-names></name> <name><surname>Krause</surname> <given-names>U.-M.</given-names></name> <name><surname>Busse</surname> <given-names>V.</given-names></name></person-group> (<year>2024</year>). <article-title>Effects of formative feedback on argumentative writing in English and cross-linguistic transfer to German.</article-title> <source><italic>Learn. Instruct.</italic></source> <volume>92</volume>:<fpage>101935</fpage>. <pub-id pub-id-type="doi">10.1016/j.learninstruc.2024.101935</pub-id></citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qin</surname> <given-names>J.</given-names></name> <name><surname>Karabacak</surname> <given-names>E.</given-names></name></person-group> (<year>2010</year>). <article-title>The analysis of Toulmin elements in Chinese EFL university argumentative writing.</article-title> <source><italic>System (Link&#x00F6;ping)</italic></source> <volume>38</volume> <fpage>444</fpage>&#x2013;<lpage>456</lpage>. <pub-id pub-id-type="doi">10.1016/j.system.2010.06.012</pub-id></citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rose</surname> <given-names>H.</given-names></name> <name><surname>McKinley</surname> <given-names>J.</given-names></name> <name><surname>Baffoe-Djan</surname> <given-names>J. B.</given-names></name></person-group> (<year>2019</year>). <source><italic>Data collection research methods in applied linguistics.</italic></source> <publisher-loc>London</publisher-loc>: <publisher-name>Bloomsbury Academic</publisher-name>.</citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>S&#x00E1;,nchez-Vera</surname> <given-names>F.</given-names></name> <name><surname>Reyes</surname> <given-names>I. P.</given-names></name> <name><surname>Cedeo</surname> <given-names>B. E.</given-names></name></person-group> (<year>2024</year>). &#x201C;<article-title>Impact of artificial intelligence on academic integrity: Perspectives of faculty members in Spain</article-title>,&#x201D; in <source><italic>Artificial intelligence and education: Enhancing human capabilities, protecting rights, and fostering effective collaboration between humans and machines in life, learning, and work</italic></source>, <role>ed.</role> M. D. D&#x00ED;az-Noguera (<publisher-loc>Spain</publisher-loc>: <publisher-name>Octaedro</publisher-name>).</citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Su</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>Y.</given-names></name> <name><surname>Lai</surname> <given-names>C.</given-names></name></person-group> (<year>2023</year>). <article-title>Collaborating with ChatGPT in argumentative writing classrooms.</article-title> <source><italic>Assess. Writ.</italic></source> <volume>57</volume>:<fpage>100752</fpage>. <pub-id pub-id-type="doi">10.1016/j.asw.2023.100752</pub-id></citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Su</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>K.</given-names></name> <name><surname>Lai</surname> <given-names>C.</given-names></name> <name><surname>Jin</surname> <given-names>T.</given-names></name></person-group> (<year>2021</year>). <article-title>The progression of collaborative argumentation among English learners: A qualitative study.</article-title> <source><italic>System</italic></source> <volume>98</volume>:<fpage>102471</fpage>. <pub-id pub-id-type="doi">10.1016/j.system.2021.102471</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wagner</surname> <given-names>C. J.</given-names></name> <name><surname>Parra</surname> <given-names>M. O.</given-names></name> <name><surname>Proctor</surname> <given-names>C. P.</given-names></name></person-group> (<year>2017</year>). <article-title>The interplay between student-led discussions and argumentative writing.</article-title> <source><italic>TESOL Quar.</italic></source> <volume>51</volume> <fpage>438</fpage>&#x2013;<lpage>449</lpage>. <pub-id pub-id-type="doi">10.1002/tesq.340</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Dang</surname> <given-names>A.</given-names></name></person-group> (<year>2024</year>). <article-title>Enhancing L2 writing with generative ai: A systematic review of pedagogical integration and outcomes.</article-title> <source><italic>Preprint</italic></source> <volume>2</volume> <fpage>1</fpage>&#x2013;<lpage>30</lpage>. <pub-id pub-id-type="doi">10.13140/RG.2.2.19572.16005</pub-id></citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Winstone</surname> <given-names>N. E.</given-names></name> <name><surname>Nash</surname> <given-names>R. A.</given-names></name> <name><surname>Parker</surname> <given-names>M.</given-names></name> <name><surname>Rowntree</surname> <given-names>J.</given-names></name></person-group> (<year>2017</year>). <article-title>Supporting learners&#x2019; agentic engagement with feedback: A systematic review and a taxonomy of recipience processes.</article-title> <source><italic>Educ. Psychol.</italic></source> <volume>52</volume> <fpage>17</fpage>&#x2013;<lpage>37</lpage>. <pub-id pub-id-type="doi">10.1080/00461520.2016.1207538</pub-id></citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wisniewski</surname> <given-names>B.</given-names></name> <name><surname>Zierer</surname> <given-names>K.</given-names></name> <name><surname>Hattie</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>The power of feedback revisited: A meta-analysis of educational feedback research.</article-title> <source><italic>Front. Psychol.</italic></source> <volume>10</volume>:<fpage>3087</fpage>. <pub-id pub-id-type="doi">10.3389/fpsyg.2019.03087</pub-id> <pub-id pub-id-type="pmid">32038429</pub-id></citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Woo</surname> <given-names>D. J.</given-names></name> <name><surname>Wang</surname> <given-names>D.</given-names></name> <name><surname>Guo</surname> <given-names>K.</given-names></name> <name><surname>Susanto</surname> <given-names>H.</given-names></name></person-group> (<year>2024</year>). <article-title>Teaching EFL students to write with ChatGPT: Students&#x2019; motivation to learn, cognitive load, and satisfaction with the learning process.</article-title> <source><italic>Educ. Inf. Technol.</italic></source> <volume>29</volume> <fpage>24963</fpage>&#x2013;<lpage>24990</lpage>. <pub-id pub-id-type="doi">10.1007/s10639-024-12819-4</pub-id></citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yan</surname> <given-names>L.</given-names></name> <name><surname>Sha</surname> <given-names>L.</given-names></name> <name><surname>Zhao</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Martinez-Maldonado</surname> <given-names>R.</given-names></name> <name><surname>Chen</surname> <given-names>G.</given-names></name><etal/></person-group> (<year>2024</year>). <article-title>Practical and ethical challenges of large language models in education: A systematic scoping review.</article-title> <source><italic>Br. J. Educ. Technol.</italic></source> <volume>55</volume> <fpage>90</fpage>&#x2013;<lpage>112</lpage>. <pub-id pub-id-type="doi">10.1111/bjet.13370</pub-id></citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yoon</surname> <given-names>S.-Y.</given-names></name> <name><surname>Miszoglad</surname> <given-names>E.</given-names></name> <name><surname>Pierce</surname> <given-names>L. R.</given-names></name></person-group> (<year>2023</year>). <article-title>Evaluation of ChatGPT feedback on ELL writers&#x2019; coherence and cohesion.</article-title> <source><italic>arXiv [Preprint]</italic></source> <pub-id pub-id-type="doi">10.48550/arXiv.2310.06505</pub-id></citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Z. V.</given-names></name> <name><surname>Hyland</surname> <given-names>K.</given-names></name></person-group> (<year>2018</year>). <article-title>Student engagement with teacher and automated feedback on L2 writing.</article-title> <source><italic>Assess. Writ.</italic></source> <volume>36</volume> <fpage>90</fpage>&#x2013;<lpage>102</lpage>. <pub-id pub-id-type="doi">10.1016/j.asw.2018.02.004</pub-id></citation></ref>
<ref id="B41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>W.</given-names></name></person-group> (<year>2001</year>). <article-title>Performing argumentative writing in English: Difficulties, processes, and strategies.</article-title> <source><italic>TESL Can. J.</italic></source> <volume>19</volume> <fpage>34</fpage>&#x2013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.18806/tesl.v19i1.918</pub-id></citation></ref>
</ref-list>
</back>
</article>