<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" article-type="research-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Psychol.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Psychology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Psychol.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">1664-1078</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpsyg.2026.1764712</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Psychometric evaluation of the full and shortened versions of the WGCTA-II in Slovak university students</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>&#x0160;ebokov&#x00E1;</surname>
<given-names>Gabriela</given-names>
</name>
<xref ref-type="aff" rid="aff1"></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1859388"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x0026; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>R&#x00E1;czov&#x00E1;</surname>
<given-names>Lucia</given-names>
</name>
<xref ref-type="aff" rid="aff1"></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1204198"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x0026; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Uhl&#x00E1;rikov&#x00E1;</surname>
<given-names>Jana</given-names>
</name>
<xref ref-type="aff" rid="aff1"></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3248559"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x0026; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Luka&#x010D;kov&#x00E1;</surname>
<given-names>Ema</given-names>
</name>
<xref ref-type="aff" rid="aff1"></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3331350"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x0026; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Soll&#x00E1;r</surname>
<given-names>Tom&#x00E1;&#x0161;</given-names>
</name>
<xref ref-type="aff" rid="aff1"></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1940176"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x0026; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
</contrib>
</contrib-group>
<aff id="aff1"><institution>Department of Psychological Sciences, Faculty of Social Sciences and Health Care, Constantine the Philosopher University in Nitra</institution>, <city>Nitra</city>, <country country="sk">Slovakia</country></aff>
<author-notes>
<corresp id="c001"><label>&#x002A;</label>Correspondence: Gabriela &#x0160;ebokov&#x00E1;, <email xlink:href="mailto:gsebokova@ukf.sk">gsebokova@ukf.sk</email></corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2026-02-02">
<day>02</day>
<month>02</month>
<year>2026</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2026</year>
</pub-date>
<volume>17</volume>
<elocation-id>1764712</elocation-id>
<history>
<date date-type="received">
<day>10</day>
<month>12</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>08</day>
<month>01</month>
<year>2026</year>
</date>
<date date-type="accepted">
<day>12</day>
<month>01</month>
<year>2026</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2026 &#x0160;ebokov&#x00E1;, R&#x00E1;czov&#x00E1;, Uhl&#x00E1;rikov&#x00E1;, Luka&#x010D;kov&#x00E1; and Soll&#x00E1;r.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>&#x0160;ebokov&#x00E1;, R&#x00E1;czov&#x00E1;, Uhl&#x00E1;rikov&#x00E1;, Luka&#x010D;kov&#x00E1; and Soll&#x00E1;r</copyright-holder>
<license>
<ali:license_ref start_date="2026-02-02">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Critical thinking (CT) is a key cognitive skill essential for academic and informed decision-making. Although the Watson-Glaser Critical Thinking Appraisal (WGCTA) has been widely used internationally, its psychometric properties have not yet been systematically evaluated in Slovakia. The present study aimed to examine the reliability and validity of the Slovak WGCTA-II (Form C) and to develop a shortened version using contemporary psychometric methods.</p>
</sec>
<sec>
<title>Method</title>
<p>Slovak university students (<italic>N</italic> =&#x202F;264) completed the WGCTA-II and two versions of the Cognitive Reflection Test (CRT-V, CRT-N) for criterion validity. Reliability analyses, confirmatory factor analyses (CFA), parallel analysis, and Item Response Theory (IRT) models were used to examine internal consistency, dimensionality, and item functioning. Based on theoretical relevance and psychometric performance, two core dimensions&#x2014;Interpretation and Evaluation of Arguments&#x2014;were retained. An independent sample (<italic>N</italic> =&#x202F;137) was used to replicate reliability and model fit.</p>
</sec>
<sec>
<title>Results</title>
<p>The original WGCTA-II showed acceptable overall reliability but limited construct validity at the subscale level. The resulting 9-item unidimensional version demonstrated good IRT model fit (RMSEA&#x202F;=&#x202F;0.042; SRMR&#x202F;=&#x202F;0.076) and satisfactory reliability (<italic>&#x03C9;</italic> =&#x202F;0.66), replicated in the second sample. Criterion validity was supported by correlations with the Cognitive Reflection Test (<italic>r</italic> =&#x202F;0.30&#x2013;0.38).</p>
</sec>
<sec>
<title>Discussion</title>
<p>These findings provide the first psychometric evidence for the Slovak WGCTA-II, demonstrate the utility of combining CTT and IRT for robust test evaluation, and introduce a concise, culturally adapted tool for efficient assessment of critical thinking, contributing to methodological innovation in psychological measurement.</p>
</sec>
</abstract>
<kwd-group>
<kwd>confirmatory factor analysis</kwd>
<kwd>critical thinking</kwd>
<kwd>item response theory</kwd>
<kwd>psychometric evaluation</kwd>
<kwd>Slovak university students</kwd>
<kwd>WGCTA-II</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declared that financial support was received for this work and/or its publication. This work was supported by Scientific Grant Agency of the Ministry of Education, Research, Development and Youth of the Slovak Republic and the Slovak Academy of Sciences, project VEGA no. 1/0336/24 Critical thinking in relation to academic success and decision-making in specific areas of students&#x2019; lives.</funding-statement>
</funding-group>
<counts>
<fig-count count="0"/>
<table-count count="4"/>
<equation-count count="0"/>
<ref-count count="77"/>
<page-count count="11"/>
<word-count count="9074"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Quantitative Psychology and Measurement</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>Critical thinking (CT) is widely recognized as a core 21st-century skill and a key predictor of success in education and society (<xref ref-type="bibr" rid="ref69">Voogt and Roblin, 2012</xref>; <xref ref-type="bibr" rid="ref56">Rusmin et al., 2024</xref>; <xref ref-type="bibr" rid="ref55">Rothinam et al., 2025</xref>). This is largely because CT is associated with key competencies for effective functioning in contemporary society, such as informed decision-making (<xref ref-type="bibr" rid="ref65">Stupple et al., 2017</xref>; <xref ref-type="bibr" rid="ref49">Ren et al., 2020</xref>), employability, and civic engagement (<xref ref-type="bibr" rid="ref20">Facione and Facione, 2001</xref>; <xref ref-type="bibr" rid="ref41">Minnameier and Hermkes, 2020</xref>; <xref ref-type="bibr" rid="ref60">Simonovic et al., 2022</xref>). Moreover, CT is increasingly viewed as a central outcome of higher education, underscoring the importance of fostering this ability among university students, whose academic achievement has also been shown to be linked to CT (<xref ref-type="bibr" rid="ref52">Rivas et al., 2023</xref>).</p>
<p>Although many definitions of CT have been proposed (e.g., <xref ref-type="bibr" rid="ref16">Ennis, 2011</xref>; <xref ref-type="bibr" rid="ref28">Halpern, 2014</xref>; <xref ref-type="bibr" rid="ref42">Moore, 2011</xref>), a particularly comprehensive one was formulated by the American Philosophical Association Delphi panel of 46 experts, defining critical thinking as &#x201C;purposeful, self-regulatory judgment which results in interpretation, analysis, evaluation, and inference, as well as explanation of the evidential, conceptual, methodological, criteriological, or contextual considerations upon which that judgment is based &#x201C;(<xref ref-type="bibr" rid="ref19">Facione, 1990</xref>, p. 3). Although CT may manifest differently across contexts and disciplines, a shared feature across conceptualizations is the ability to analyze and evaluate arguments and the evidence supporting them &#x2013; skills widely recognized as central to critical reasoning (<xref ref-type="bibr" rid="ref14">Dwyer et al., 2014</xref>; <xref ref-type="bibr" rid="ref43">Mueller et al., 2020</xref>; <xref ref-type="bibr" rid="ref39">Liu et al., 2014</xref>; <xref ref-type="bibr" rid="ref63">Stanovich et al., 2016</xref>). Despite variations in definitions, this common emphasis supports viewing CT as a coherent construct centered on analytic and evaluative reasoning.</p>
<p>Given its importance, the accurate assessment of CT is essential for both research and educational practice. Although ministries of education worldwide emphasize the development of CT, only a limited number of assessment tools have been validated for use in different populations (<xref ref-type="bibr" rid="ref29">Hassan and Madhum, 2007</xref>). Several standardized instruments exist, including the Watson&#x2013;Glaser Critical Thinking Appraisal (WGCTA; <xref ref-type="bibr" rid="ref71">Watson and Glaser, 1980</xref>), the Cornell Critical Thinking Test (<xref ref-type="bibr" rid="ref17">Ennis and Millman, 1985</xref>), the California Critical Thinking Skills Test (<xref ref-type="bibr" rid="ref19">Facione, 1990</xref>), and the Halpern Critical Thinking Assessment (<xref ref-type="bibr" rid="ref27">Halpern, 2010</xref>). Despite their widespread use, concerns remain regarding their psychometric quality. Research has revealed inconsistent evidence of validity and reliability, including low reliability coefficients (<xref ref-type="bibr" rid="ref1">Abrami et al., 2008</xref>; <xref ref-type="bibr" rid="ref68">Verburgh et al., 2013</xref>), unstable factor structures (<xref ref-type="bibr" rid="ref1">Abrami et al., 2008</xref>; <xref ref-type="bibr" rid="ref5">Bernard et al., 2008</xref>; <xref ref-type="bibr" rid="ref38">Leach et al., 2020</xref>; <xref ref-type="bibr" rid="ref68">Verburgh et al., 2013</xref>), problematic response formats (<xref ref-type="bibr" rid="ref5">Bernard et al., 2008</xref>; <xref ref-type="bibr" rid="ref36">Ku, 2009</xref>), ambiguous instructions (<xref ref-type="bibr" rid="ref21">Fawkes et al., 2003</xref>; <xref ref-type="bibr" rid="ref45">Possin, 2014</xref>), limited research applications (<xref ref-type="bibr" rid="ref39">Liu et al., 2014</xref>) and cross-cultural equivalence (<xref ref-type="bibr" rid="ref9">Butler et al., 2012</xref>). A persistent challenge in the critical thinking literature concerns how to validly assess such a multifaceted construct, further complicated by the limited empirical evidence supporting the correspondence between test items and their underlying theoretical dimensions (<xref ref-type="bibr" rid="ref38">Leach et al., 2020</xref>). These issues highlight the need for thorough psychometric evaluation of CT assessments in diverse populations. The present study therefore focuses on the Watson&#x2013;Glaser Critical Thinking Appraisal (WGCTA)&#x2014;the oldest, most widely used, and most extensively studied CT test (<xref ref-type="bibr" rid="ref5">Bernard et al., 2008</xref>). Although the WGCTA has been applied across a wide range of educational and professional contexts and is valued for assessing core components of CT (<xref ref-type="bibr" rid="ref75">Wayas et al., 2024</xref>; <xref ref-type="bibr" rid="ref2">Afzal et al., 2024</xref>; <xref ref-type="bibr" rid="ref59">&#x0160;ebokov&#x00E1; et al., 2025</xref>), its continued use depends on robust evidence of its psychometric adequacy, making its evaluation both theoretically and practically important.</p>
<p>The WGCTA defines critical thinking as &#x201C;the ability to identify and analyze problems, seek and evaluate relevant information, and reach an appropriate conclusion&#x201D; (<xref ref-type="bibr" rid="ref74">Watson and Glaser, 2018</xref>). It assesses five interrelated skills in a verbal context: (1) Inference &#x2013; judging the likelihood that conclusions follow from given information; (2) Recognition of assumptions &#x2013; recognizing implied assumptions or presuppositions behind the provided statements; (3) Deduction &#x2013; evaluating whether conclusions logically follow from premises; (4) Interpretation &#x2013; assessing whether conclusions are justified by evidence; and (5) Evaluation of arguments &#x2013; determining the strength and relevance of arguments (<xref ref-type="bibr" rid="ref74">Watson and Glaser, 2018</xref>). Since its introduction, the WGCTA has undergone several revisions. The original 1964 version included 100 items (Forms Ym and Zm). In 1980, Forms A and B (WGCTA-II, 80 items) were published, followed by a UK adaptation of Form B (Form C; <xref ref-type="bibr" rid="ref57">Rust, 2002</xref>). A 40-item Short Form S and later parallel Forms D and E were also developed. The most recent revision, WGCTA-III (<xref ref-type="bibr" rid="ref74">Watson and Glaser, 2018</xref>), introduced updated, business-oriented items and is now available in an online format.</p>
<p>Although the WGCTA has been extensively studied internationally, its psychometric properties have not yet been systematically examined in Slovakia. The Slovak adaptation of WGCTA-II (Form C) exists (<xref ref-type="bibr" rid="ref73">Watson and Glaser, 2000</xref>), but no published data are available regarding its reliability, validity, or factor structure in Slovak university students. Moreover, the administration of the WGCTA-II requires approximately 40&#x202F;min, which may limit its feasibility for large-scale research applications. Examining the Slovak version of the WGCTA-II in this context provides valuable empirical evidence on its psychometric soundness and practical applicability, thereby contributing to a broader understanding of the cross-cultural validity of critical thinking assessment tools, specifically in central-eastern European context.</p>
<p>Over the years, numerous studies have examined the psychometric properties of various WGCTA versions, highlighting concerns about their validity and reliability, particularly regarding the factor structure of the test. Some studies suggest a unidimensional solution rather than separate scores for the five subscales (<xref ref-type="bibr" rid="ref5">Bernard et al., 2008</xref>; <xref ref-type="bibr" rid="ref29">Hassan and Madhum, 2007</xref>). <xref ref-type="bibr" rid="ref5">Bernard et al. (2008)</xref> examined 60 sets of subscale means and 13 sets of intercorrelations reported in published studies with diverse learner groups, identifying a consistent one-factor solution and concluding that the WGCTA measures a general critical thinking skill rather than five distinct dimensions. Similar findings emerged from a study with Lebanese university students, where exploratory factor analysis of the WGCTA-S also indicated a unidimensional structure (<xref ref-type="bibr" rid="ref29">Hassan and Madhum, 2007</xref>). Taken together, these findings challenge the theoretical assumption of five separable dimensions and raise questions about the construct validity of the WGCTA subscales.</p>
<p>Another concern involves the wide range of reliability coefficients, reported between 0.23 and 0.73 (<xref ref-type="bibr" rid="ref5">Bernard et al., 2008</xref>) or 0.17 to 0.74 (<xref ref-type="bibr" rid="ref40">Loo and Thorpe, 1999</xref>). This variability suggests that the reliability of the WGCTA is not a stable property of the instrument but may be strongly context dependent. Previous research suggests that reliability may be influenced by contextual differences across learners and settings (<xref ref-type="bibr" rid="ref5">Bernard et al., 2008</xref>), the multiple-choice format that increases the likelihood of guessing (<xref ref-type="bibr" rid="ref70">Wagner and Harvey, 2006</xref>), and challenges related to cross-cultural applicability and face validity. Additional limitations of the WGCTA include non-comparable test forms, unclear evidence of differential validity across respondent groups, and limited evidence of incremental validity (<xref ref-type="bibr" rid="ref39">Liu et al., 2014</xref>). Together, these issues limit the interpretability and comparability of WGCTA scores across studies and raise concerns about its suitability for cross-cultural research without thorough local validation.</p>
<p>Despite these issues, some studies provide evidence of convergent and criterion validity. For instance, WGCTA-S scores moderately correlated with analysis and problem-solving (<italic>r</italic>&#x202F;=&#x202F;0.52) and judgment and decision-making (<italic>r</italic>&#x202F;=&#x202F;0.52) (<xref ref-type="bibr" rid="ref15">Ejiogu et al., 2006</xref>), as well as with undergraduate course grades in psychology and education (<italic>r</italic>&#x202F;=&#x202F;0.20&#x2013;0.62) (<xref ref-type="bibr" rid="ref25">Gadzella et al., 2006</xref>). Overall, the available evidence indicates that while the WGCTA demonstrates meaningful convergent and criterion validity, its limitations are primarily related to the psychometric functioning of specific items and subscale structures, pointing to the need for systematic refinement rather than rejection, which subsequently motivated revisions of the test.</p>
<p>In response to these considerations, the Red Model was introduced in WGCTA-II, reorganizing items into a simplified three-factor structure: Recognize Assumptions, Evaluate Arguments, and Draw Conclusions (combining Inference, Deduction, and Interpretation). Confirmatory factor analysis supported this structure, with internal consistency ranging from 0.81 to 0.89 (NCS Pearson, Inc., 2009). Convergent and criterion validity were demonstrated through correlations with WAIS-IV (<italic>r</italic>&#x202F;=&#x202F;0.52), Raven&#x2019;s APM (<italic>r</italic>&#x202F;=&#x202F;0.53), and the Advanced Numerical Reasoning Appraisal (<italic>r</italic>&#x202F;=&#x202F;0.68). The RED model formed the basis for WGCTA-III, which features a new, globally applicable item bank. However, the RED model and its three-factor structure have not yet been independently and extensively evaluated. This gap in empirical evidence constrains both the measurement and conceptual understanding of critical thinking within educational settings. As <xref ref-type="bibr" rid="ref36">Ku (2009)</xref> noted in her review of the psychometric properties of CT assessments, test developers and affiliated researchers tend to report more favorable psychometric qualities than independent researchers. This underscores the need for more impartial, externally conducted validations.</p>
<p>Moreover, no comprehensive item-level psychometric investigation has been conducted. IRT has several advantages over the classical test theory (CTT). In CTT, reliability is considered a property of the entire test, typically measured using coefficients such as Cronbach&#x2019;s alpha. This approach necessitates re-evaluation if the test undergoes modifications, such as shortening or adaptation. In contrast, Item Response Theory (IRT) evaluates reliability at the item level, providing a more precise analysis of how each question contributes to assessing critical thinking, thereby enhancing the accuracy and flexibility of the measurement (<xref ref-type="bibr" rid="ref6">Bond, 2015</xref>). IRT provides an accurate and consistent measurement of CT by accounting for both item difficulty and individual ability. Unlike traditional methods, it separates ability from item characteristics, ensuring more valid and reliable estimates while maintaining fairness through probabilistic modeling, where higher-ability students are more likely to answer difficult questions correctly (<xref ref-type="bibr" rid="ref6">Bond, 2015</xref>). Critical thinking comprises multiple subscales, necessitating measurement tools that can assess each ability separately. Empirical research indicates that using the item response theory approach enhances the precision of cognitive and complex thinking skill assessments by providing a more detailed analysis of an instrument&#x2019;s reliability and validity (<xref ref-type="bibr" rid="ref66">Suwita et al., 2024</xref>). Therefore, this methodological framework seems suitable for present study.</p>
<p>Given the unresolved issues with the WGCTA &#x2013; unclear factor structure, inconsistent reliability, cultural applicability concerns, and the absence of item-level psychometric investigations&#x2014;further examination is needed. In addition, the test&#x2019;s considerable length and limited large-scale research applications in higher education contexts point to the need for a shorter yet psychometrically sound version. Building on the limitations, the present study had three main aims. First, we examined the psychometric properties of the Slovak version of the WGCTA &#x2013; II (Form C) in a sample of university students. Specifically, we conducted reliability analyses and confirmatory factor analyses to clarify the underlying structure of the test&#x2019;s five subscales. This step was necessary to determine whether the original multidimensional structure of the WGCTA would be empirically supported in our cultural context and to identify potential sources of psychometric weakness. Second, based on the results of the factor analyses and the item-level diagnostics, we performed an item response theory (IRT) analysis using the Rasch model to evaluate a shortened version of the WGCTA that would retain only the most informative and psychometrically valid items. The goal was to develop a more practical and efficient measure of critical thinking suitable for research and educational settings, without compromising reliability or validity. Third, we examined the reliability and the convergent validity of the shortened version of WGCTA on two samples of university students.</p>
</sec>
<sec sec-type="methods" id="sec2">
<label>2</label>
<title>Methods</title>
<sec id="sec3">
<label>2.1</label>
<title>Research sample</title>
<p>The research sample consisted of 266 university students enrolled in bachelor&#x2019;s degree programs at two universities in Slovakia. Two participants were excluded due to incomplete data, resulting in a final sample of 264 respondents (168 men, 96 women; <italic>M<sub>age</sub></italic>&#x202F;=&#x202F;19.94, <italic>SD<sub>age</sub></italic>&#x202F;=&#x202F;1.36). Of these, 216 students were in their first year and 48 students were in their second year. The sample included 40% students from the humanities (psychology) and 60% students from technical fields (automatic production, control systems). Most participants (79%) were of Slovak nationality.</p>
<p>The second research sample comprised 137 undergraduate students (32 men, 105 women; <italic>M<sub>age</sub></italic>&#x202F;=&#x202F;20.93, <italic>SD<sub>age</sub></italic>&#x202F;=&#x202F;1.45) from universities in Slovakia. This sample included 69% students from the humanities (psychology and social work) and 31% students from technical fields (landscape engineering) studying at two Slovak universities. Most participants (98%) were of Slovak nationality. Regarding academic year, 120 students were in their second year and 17 were in their third year.</p>
</sec>
<sec id="sec4">
<label>2.2</label>
<title>Materials</title>
<p><italic>Watson&#x2013;Glaser Critical Thinking Appraisal</italic> (WGCTA-II, form C; <xref ref-type="bibr" rid="ref72">Watson and Glaser, 1991</xref>): The WGCTA contains statements representing a wide range of written and spoken materials commonly encountered in everyday situations and exists in several versions, including updated editions used internationally. For the purposes of this study, we employ the version that is currently the only standardized version available in our context, as more recent editions, although existing abroad, have not yet been adapted or validated locally. The test comprises 80 items divided into five subscales (each consisting of 16 items): Inference &#x2013; the ability to judge whether conclusions follow from the provided information. Recognition of Assumptions &#x2013; the ability to identify implicit assumptions in statements. Deduction &#x2013; the ability to determine whether a conclusion logically follows from given premises. Induction/Interpretation &#x2013; the ability to evaluate whether evidence and conclusions can be generalized. Evaluation of Arguments &#x2013; the ability to assess the relevance and strength of arguments related to a problem. For the Inference subscale, a 5-point Likert scale is used, whereas the other four subscales employ a dichotomous (1/0) response format. The test includes distinct instructions and practice questions with answers for each of the five subscales to help participants understand the task and familiarize themselves with the test format. This structure makes the test cognitively and procedurally demanding, as participants must undergo brief training or familiarization before each subtest.</p>
<p>Example of the statement: Research on vocabulary development in children from 8&#x202F;months to 6&#x202F;years has shown that the number of words used grows from 0 words at 8&#x202F;months to 2,562 words at 6&#x202F;years.</p>
<p>None of the children who participated in the study could speak at 6&#x202F;months (mark YES, the conclusion follows unequivocally from the statement, since the vocabulary of children at 6&#x202F;months was 0 words).</p>
<p>The growth of vocabulary is slowest during the period when children learn to walk (mark NO, the conclusion does not follow from the statement, as there is no information about the relationship between vocabulary and learning to walk).</p>
<p><italic>Cognitive Reflection Test</italic> (CRT): Convergent validity of the WGCTA (both original and shortened) was evaluated using both the numerical (<xref ref-type="bibr" rid="ref23">Frederick, 2005</xref>) and verbal (<xref ref-type="bibr" rid="ref61">Sirota et al., 2020</xref>) versions of the Cognitive Reflection Test. The CRT assesses the ability to suppress intuitive but incorrect responses in favor of reflective, deliberative reasoning, based on the dual-processing framework of cognition. Type 1 processing is fast and intuitive, often leading to biased responses, whereas Type 2 processing is slower, reflective, and supports logical reasoning (<xref ref-type="bibr" rid="ref23">Frederick, 2005</xref>; <xref ref-type="bibr" rid="ref63">Stanovich et al., 2016</xref>). Critical thinking is conceptually related to Type 2 processing, as reflective thinking facilitates the development and application of rules and strategies necessary for analytical problem solving (<xref ref-type="bibr" rid="ref60">Simonovic et al., 2022</xref>; <xref ref-type="bibr" rid="ref7">Bonnefon, 2018</xref>). Including both versions allows assessment of cognitive reflection while accounting for potential confounding with numeracy skills in the numerical version and focusing on reasoning ability in the verbal version (<xref ref-type="bibr" rid="ref61">Sirota et al., 2020</xref>). This approach ensures a comprehensive and theoretically relevant measure for convergent validation of critical thinking.</p>
<p><italic>Numerical CRT</italic> (<xref ref-type="bibr" rid="ref23">Frederick, 2005</xref>): The test consists of 3 items (e.g., &#x201C;If it takes 3 people 3&#x202F;min to make 3 products, how long would it take 10 people to make 10 products?&#x201D;). Each item has two response alternatives: one representing intuitive thinking and one representing reflective thinking. Participants select one answer, with no time limit imposed. Scores range from 0 (no correct answers) to 3 (all correct), with higher scores indicating greater cognitive reflection.</p>
<p><italic>Verbal CRT</italic> (<xref ref-type="bibr" rid="ref61">Sirota et al., 2020</xref>): The test includes 10 open-ended problems (e.g., &#x201C;Laura&#x2019;s father has 5 daughters but no sons: Nana, Nene, Nini, Nono. What is the fifth daughter&#x2019;s name?&#x201D;). It measures the ability to suppress a default intuitive response in favor of a correct reflective answer. Scores range from 0 (no correct answers) to 10 (all correct), with higher scores indicating greater cognitive reflection.</p>
</sec>
<sec id="sec6">
<label>2.3</label>
<title>Procedure</title>
<p>Data for the first research sample were collected during the winter term of 2023/2024 under supervised, paper-and-pencil conditions. Participants received oral and written feedback on their test scores, and individual reports were provided upon request via email or post. Data for the second research sample were collected in the winter term of 2024/2025, also under supervision. Participants again received both oral and written feedback, and individual reports were emailed upon request.</p>
<p>All procedures followed were in accordance with the ethical standards of the responsible committee on human experimentation (both institutional and national) and with the Declaration of Helsinki of 1975, as revised in 2000. The study was also approved as part of an ongoing project (VEGA 1/0336/24: Critical thinking in relation to academic success and decision-making in specific areas of students&#x00B4; lives) by the ethics committee (UKF/370/2025/191013:024).</p>
</sec>
<sec id="sec7">
<label>2.4</label>
<title>Statistical analysis</title>
<p>All analyses were conducted in R (Version 4.2.2; <xref ref-type="bibr" rid="ref47">R Core Team, 2022</xref>). Data preparation and descriptive statistics were performed using the dplyr package (<xref ref-type="bibr" rid="ref77">Wickham et al., 2023a</xref>) and tidyr (<xref ref-type="bibr" rid="ref78">Wickham et al., 2023b</xref>), while visualizations were created with ggplot2 (<xref ref-type="bibr" rid="ref76">Wickham, 2016</xref>). One item with zero variance (WG_ Assumptions _31) was excluded prior to all analyses.</p>
<p>Internal consistency was evaluated using Cronbach&#x2019;s <italic>&#x03B1;</italic> and McDonald&#x2019;s <italic>&#x03C9;</italic>, computed from the polychoric correlation matrix due to the dichotomous scoring. Reliability indices, item difficulties, and item&#x2013;total correlations were obtained using psych (<xref ref-type="bibr" rid="ref51">Revelle, 2023</xref>) in combination with polycor (<xref ref-type="bibr" rid="ref22">Fox, 2019</xref>).</p>
<p>The dimensional structure of each WGCTA subscale was examined using confirmatory factor analysis (CFA) with a single-factor model per subscale. Our analytic strategy followed a stepwise logic. First, we examined the unidimensionality and fit of each WGCTA subscale separately using first-order CFA models. This approach was necessary to evaluate whether the basic assumptions for testing a higher-order or hierarchical model were met. Given the inadequate fit and low factor loadings observed at the subscale level, higher-order CFA models were not pursued. CFAs were estimated with the WLSMV estimator in lavaan (<xref ref-type="bibr" rid="ref54">Rosseel, 2012</xref>), with additional utilities provided by semTools (<xref ref-type="bibr" rid="ref31">Jorgensen et al., 2022</xref>). Model fit was evaluated using &#x03C7;<sup>2</sup>, RMSEA, CFI, TLI, and SRMR, following recommended APA and structural-equation-modeling guidelines (<xref ref-type="bibr" rid="ref30">Hu and Bentler, 1999</xref>; <xref ref-type="bibr" rid="ref34">Kline, 2016</xref>). Values of CFI and TLI&#x202F;&#x2265;&#x202F;0.90 indicate acceptable fit, and &#x2265; 0.95 indicate excellent fit. RMSEA &#x003C; 0.06 and SRMR &#x003C; 0.08 are generally regarded as indicative of good fit. Because most subscales demonstrated inadequate fit (e.g., low CFI/TLI), higher-order WGCTA models were not tested.</p>
<p>To develop a shortened WGCTA version, items from the Interpretation and Evaluation of Arguments subscales were selected using parallel analysis (via psych, <xref ref-type="bibr" rid="ref51">Revelle, 2023</xref>; TAM, <xref ref-type="bibr" rid="ref53">Robitzsch et al., 2021</xref>) and factor loadings.</p>
<p>The shortened version was evaluated using a Rasch model estimated with mirt (<xref ref-type="bibr" rid="ref10">Chalmers, 2012</xref>). Model fit was assessed using &#x03C7;<sup>2</sup> fit statistics, RMSEA, SRMR, infit/outfit, and S-X<sup>2</sup> indices. Criterion validity was examined via correlations with CRT-N and CRT-V using psych (<xref ref-type="bibr" rid="ref51">Revelle, 2023</xref>). Differences between dependent correlations were tested with Steiger&#x2019;s Z test using cocor (<xref ref-type="bibr" rid="ref13">Diedenhofen and Musch, 2015</xref>). Finally, the nine-item Rasch model was replicated on an independent sample using the same analytic procedures to confirm model stability, item functioning, and criterion validity.</p>
</sec>
</sec>
<sec sec-type="results" id="sec8">
<label>3</label>
<title>Results</title>
<sec id="sec9">
<label>3.1</label>
<title>Psychometric analysis of the full WGCTA-II and its individual dimensions</title>
<p>The analyses were conducted on 79 items of the Slovak version of the WGCTA, covering five subscales: Inference, Recognition of Assumptions, Deduction, Interpretation, and Evaluation of Arguments. One item with zero variance (WG_Assumptions_31) was excluded. All items were scored dichotomously (0&#x202F;=&#x202F;incorrect, 1&#x202F;=&#x202F;correct).</p>
<p>Item difficulty varied widely across the WGCTA subscales (see <xref ref-type="supplementary-material" rid="SM1">Supplementary Table 1</xref>). Inference was the most challenging, Recognition of Assumptions the easiest, Deduction showed moderate difficulty, Interpretation ranged from easy to difficult items, and Evaluation of Arguments exhibited relatively consistent medium difficulty.</p>
<p>Internal consistency was assessed using Cronbach&#x2019;s alpha and McDonald&#x2019;s omega coefficients. The overall WGCTA showed moderate to good reliability (<italic>&#x03B1;</italic>&#x202F;=&#x202F;0.691; <italic>&#x03C9;</italic>&#x202F;=&#x202F;0.846). However, it is important to note that both Cronbach&#x2019;s alpha and McDonald&#x2019;s omega are sensitive to the number of items, and higher values may reflect test length rather than true inter-item homogeneity (<xref ref-type="bibr" rid="ref11">Cortina, 1993</xref>; <xref ref-type="bibr" rid="ref48">Raykov, 1997</xref>). Therefore, the relatively high reliability of the total WGCTA score should not be interpreted as evidence of a well-functioning or internally coherent multidimensional structure. In contrast, reliability estimates for the individual subscales were substantially lower (<italic>&#x03B1;</italic>&#x202F;=&#x202F;0.23&#x2013;0.51; &#x03C9;&#x202F;=&#x202F;0.49&#x2013;0.72; see <xref ref-type="table" rid="tab1">Table 1</xref>), indicating weak internal consistency and limited homogeneity of items within the proposed dimensions. This discrepancy between the relatively higher reliability of the total score and the poor reliability of the individual subscales suggests that the original multidimensional structure of the WGCTA is not psychometrically coherent in the present sample.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Internal consistency coefficients for the total scale and subscales of the WGCTA-II (full version).</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Subscale</th>
<th align="center" valign="top">Number of items</th>
<th align="center" valign="top">Cronbach&#x2019;s &#x03B1;</th>
<th align="center" valign="top">McDonald&#x2019;s <italic>&#x03C9;</italic></th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Total WGCTA</td>
<td align="center" valign="middle">79</td>
<td align="char" valign="middle" char=".">0.691</td>
<td align="char" valign="middle" char=".">0.846</td>
</tr>
<tr>
<td align="left" valign="middle">Inference</td>
<td align="center" valign="middle">16</td>
<td align="char" valign="middle" char=".">0.469</td>
<td align="char" valign="middle" char=".">0.594</td>
</tr>
<tr>
<td align="left" valign="middle">Assumptions</td>
<td align="center" valign="middle">15</td>
<td align="char" valign="middle" char=".">0.262</td>
<td align="char" valign="middle" char=".">0.497</td>
</tr>
<tr>
<td align="left" valign="middle">Deduction</td>
<td align="center" valign="middle">16</td>
<td align="char" valign="middle" char=".">0.234</td>
<td align="char" valign="middle" char=".">0.654</td>
</tr>
<tr>
<td align="left" valign="middle">Interpretation</td>
<td align="center" valign="middle">16</td>
<td align="char" valign="middle" char=".">0.491</td>
<td align="char" valign="middle" char=".">0.719</td>
</tr>
<tr>
<td align="left" valign="middle">Arguments</td>
<td align="center" valign="middle">16</td>
<td align="char" valign="middle" char=".">0.511</td>
<td align="char" valign="middle" char=".">0.667</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Given these findings, subsequent analyses focused on evaluating the dimensional structure of each WGCTA subscale separately before testing the overall model. This stepwise approach is crucial, as assessing individual subscales provides insight into whether they represent coherent latent dimensions. If the subscales fail to demonstrate adequate internal consistency or unidimensionality, testing a comprehensive higher-order model would not be theoretically or empirically justified. Accordingly, a separate confirmatory factor analysis (CFA) was conducted for each WGCTA subscale, testing a single-factor model for each dimension. The results (<xref ref-type="table" rid="tab2">Table 2</xref>) indicated that the single-factor model showed poor fit across most subscales, with CFI and TLI values well below the recommended threshold of 0.90 and in some cases even below 0.70, suggesting that a unidimensional solution was inadequate. The best fit was observed for the Interpretation subscale, whereas Recognition of Assumptions and Evaluation of Arguments showed the poorest fit. The standardized factor loadings of the items are presented in <xref ref-type="supplementary-material" rid="SM1">Supplementary Table 2</xref>. Inspection of the standardized factor loadings revealed that a substantial proportion of items across most subscales exhibited loadings below commonly recommended thresholds (e.g., &#x003C; 0.30), with several items showing very weak or negative loadings. This pattern indicates limited association between many indicators and their intended latent dimensions and suggests that the original subscale structure is not well supported at the item level in the present sample. Given that even the individual subscales did not demonstrate satisfactory fit, we did not proceed to test a higher-order model across the full WGCTA. Testing a higher-order CFA model would require adequate unidimensionality and acceptable fit at the subscale level. Given that most subscales failed to meet these prerequisites, proceeding with a second-order model would have been both theoretically and statistically unjustified.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Confirmatory factor analysis of the WGCTA-II subscales (full version).</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Subscale</th>
<th align="center" valign="top">&#x03C7;<sup>2</sup></th>
<th align="center" valign="top">df</th>
<th align="center" valign="top"><italic>p</italic></th>
<th align="center" valign="top">RMSEA</th>
<th align="center" valign="top">CFI</th>
<th align="center" valign="top">TLI</th>
<th align="center" valign="top">SRMR</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Inference</td>
<td align="char" valign="middle" char=".">158.29</td>
<td align="center" valign="middle">104</td>
<td align="char" valign="middle" char=".">&#x003C; 0.001</td>
<td align="char" valign="middle" char=".">0.045</td>
<td align="char" valign="middle" char=".">0.746</td>
<td align="char" valign="middle" char=".">0.707</td>
<td align="char" valign="middle" char=".">0.112</td>
</tr>
<tr>
<td align="left" valign="middle">Assumptions</td>
<td align="char" valign="middle" char=".">138.81</td>
<td align="center" valign="middle">90</td>
<td align="char" valign="middle" char=".">0.001</td>
<td align="char" valign="middle" char=".">0.045</td>
<td align="char" valign="middle" char=".">0.649</td>
<td align="char" valign="middle" char=".">0.590</td>
<td align="char" valign="middle" char=".">0.118</td>
</tr>
<tr>
<td align="left" valign="middle">Deduction</td>
<td align="char" valign="middle" char=".">230.91</td>
<td align="center" valign="middle">104</td>
<td align="char" valign="middle" char=".">&#x003C; 0.001</td>
<td align="char" valign="middle" char=".">0.068</td>
<td align="char" valign="middle" char=".">0.767</td>
<td align="char" valign="middle" char=".">0.731</td>
<td align="char" valign="middle" char=".">0.121</td>
</tr>
<tr>
<td align="left" valign="middle">Interpretation</td>
<td align="char" valign="middle" char=".">181.02</td>
<td align="center" valign="middle">104</td>
<td align="char" valign="middle" char=".">&#x003C; 0.001</td>
<td align="char" valign="middle" char=".">0.053</td>
<td align="char" valign="middle" char=".">0.809</td>
<td align="char" valign="middle" char=".">0.780</td>
<td align="char" valign="middle" char=".">0.121</td>
</tr>
<tr>
<td align="left" valign="middle">Arguments</td>
<td align="char" valign="middle" char=".">230.40</td>
<td align="center" valign="middle">104</td>
<td align="char" valign="middle" char=".">&#x003C; 0.001</td>
<td align="char" valign="middle" char=".">0.068</td>
<td align="char" valign="middle" char=".">0.603</td>
<td align="char" valign="middle" char=".">0.542</td>
<td align="char" valign="middle" char=".">0.129</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec10">
<label>3.2</label>
<title>Item selection for the shortened WGCTA version</title>
<p>The results of the analyses indicated that the items of the original WGCTA did not contribute evenly to the measurement of the critical thinking construct, with several dimensions showing low reliability and weak construct validity. Accordingly, the second aim of the study was to develop a more efficient and reliable WGCTA version by selecting items with suitable difficulty, higher factor loadings, and positive contributions to internal consistency.</p>
<p>Given the large number of items, we first selected subscales that best represented the critical thinking construct, combining theoretical considerations (RED model) with empirical results. From the Draw Conclusions dimension, only the Interpretation subscale was retained due to superior internal consistency and fit, while Inference and Deduction were excluded for low reliability, poor fit, high difficulty, and complex item formats. Evaluation of Arguments was retained despite lower CFI and TLI values because of acceptable RMSEA, adequate reliability (<italic>&#x03C9;</italic>&#x202F;=&#x202F;0.667), and balanced item difficulty; Recognition of Assumptions was excluded due to very low reliability and negative factor loadings. Parallel analysis revealed that many indicators within the Interpretation and Evaluation of Arguments subscales failed to demonstrate sufficiently strong associations with their latent dimensions, as reflected in standardized factor loadings below commonly recommended thresholds (e.g., &#x003C; 0.30). Parallel analysis within Interpretation and Evaluation of Arguments identified the most psychometrically and content-relevant items, resulting in a 9-item shortened scale (5 Interpretation, 4 Evaluation of Arguments; see <xref ref-type="table" rid="tab3">Table 3</xref>). Items with low factor loadings (below 0.25) and negative item-total correlations were excluded to improve construct validity. The selection combined empirical criteria (loadings, model fit, reliability indices) with theoretical relevance, ensuring that retained items represent the core critical thinking processes. Item selection also accounted for linguistic and cultural appropriateness (e.g., avoiding double negatives and culturally specific expressions) and ensured a balance between strong and weak arguments as well as correct and incorrect answers. Low factor loadings were interpreted not as isolated statistical artifacts, but as systematic indicators of weak construct representation, cultural mismatch, or excessive cognitive and linguistic complexity of certain items. The final set meets psychometric quality standards while encompassing key aspects of critical thinking, such as evidence interpretation and argument evaluation.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Paralel analysis of interpretation and arguments subscales.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Item</th>
<th align="center" valign="top">Loading</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" colspan="2">Interpretation</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_51</td>
<td align="char" valign="middle" char=".">0.548</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_56</td>
<td align="char" valign="middle" char=".">0.485</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_53</td>
<td align="char" valign="middle" char=".">0.420</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_54</td>
<td align="char" valign="middle" char=".">0.414</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_61</td>
<td align="char" valign="middle" char=".">0.403</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_60</td>
<td align="char" valign="middle" char=".">0.401</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_49</td>
<td align="char" valign="middle" char=".">0.305</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_62</td>
<td align="char" valign="middle" char=".">0.304</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_63</td>
<td align="char" valign="middle" char=".">0.246</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_59</td>
<td align="char" valign="middle" char=".">0.237</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_55</td>
<td align="char" valign="middle" char=".">0.216</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_50</td>
<td align="char" valign="middle" char=".">0.044</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_58</td>
<td align="char" valign="middle" char=".">0.040</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_64</td>
<td align="char" valign="middle" char=".">0.024</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_57</td>
<td align="char" valign="middle" char=".">&#x2212;0.054</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_52</td>
<td align="char" valign="middle" char=".">&#x2212;0.153</td>
</tr>
<tr>
<td align="left" valign="middle" colspan="2">Arguments</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_69</td>
<td align="center" valign="middle">0.452</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_71</td>
<td align="char" valign="middle" char=".">0.451</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_78</td>
<td align="char" valign="middle" char=".">0.361</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_68</td>
<td align="char" valign="middle" char=".">0.330</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_66</td>
<td align="char" valign="middle" char=".">0.306</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_80</td>
<td align="char" valign="middle" char=".">0.287</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_76</td>
<td align="char" valign="middle" char=".">0.285</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_79</td>
<td align="char" valign="middle" char=".">0.280</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_67</td>
<td align="char" valign="middle" char=".">0.263</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_65</td>
<td align="char" valign="middle" char=".">0.237</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_70</td>
<td align="char" valign="middle" char=".">0.210</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_73</td>
<td align="char" valign="middle" char=".">0.174</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_75</td>
<td align="char" valign="middle" char=".">0.151</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_74</td>
<td align="char" valign="middle" char=".">0.113</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_72</td>
<td align="char" valign="middle" char=".">0.103</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_77</td>
<td align="char" valign="middle" char=".">0.020</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec11">
<label>3.3</label>
<title>Examination of the shortened WGCTA version using IRT</title>
<p>The correlation between the Interpretation and Evaluation of Arguments dimensions was 0.48, indicating a moderate relationship. This level of correlation led to the consideration of a common factor and the development of a unidimensional scale. The nine-item scale was subsequently evaluated using IRT analysis with the Rasch model. The results indicated that the nine-item model demonstrated an acceptable fit to the data (&#x03C7;<sup>2</sup>(35)&#x202F;=&#x202F;50.996, <italic>p</italic>&#x202F;=&#x202F;0.039; RMSEA&#x202F;=&#x202F;0.042; SRMR&#x202F;=&#x202F;0.076). RMSEA values below 0.06 and SRMR values below 0.08 suggest good model fit (<xref ref-type="bibr" rid="ref30">Hu and Bentler, 1999</xref>), while the chi-square test result points to only minor deviations, which are acceptable given the small number of items.</p>
<p>The 9-item scale demonstrated moderate internal consistency (<italic>&#x03B1;</italic>&#x202F;=&#x202F;0.65; <italic>&#x03C9;</italic>&#x202F;=&#x202F;0.66), which is acceptable given the shortened length and suitable for practical use. Item difficulty estimates ranged from &#x2212;2.40 (WG_interpretacia_54) to 0.41 (WG_interpretacia_62; <xref ref-type="table" rid="tab4">Table 4</xref>), indicating that the combination of the most psychometrically and content-relevant items from both subscales produced a unidimensional 9-item scale with adequate fit and reliability.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>Item difficulty estimates for the shortened WGCTA-II based on the Rasch model.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Item</th>
<th align="center" valign="top">Difficulty (b)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">WG_interpretation_51</td>
<td align="char" valign="middle" char=".">&#x2212;1.119</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_54</td>
<td align="char" valign="middle" char=".">&#x2212;2.403</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_56</td>
<td align="char" valign="middle" char=".">&#x2212;1.051</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_60</td>
<td align="char" valign="middle" char=".">&#x2212;1.813</td>
</tr>
<tr>
<td align="left" valign="middle">WG_interpretation_62</td>
<td align="char" valign="middle" char=".">0.407</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_68</td>
<td align="char" valign="middle" char=".">&#x2212;1.306</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_69</td>
<td align="char" valign="middle" char=".">&#x2212;1.506</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_71</td>
<td align="char" valign="middle" char=".">&#x2212;1.532</td>
</tr>
<tr>
<td align="left" valign="middle">WG_arguments_79</td>
<td align="char" valign="middle" char=".">&#x2212;0.631</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec12">
<label>3.4</label>
<title>Criterion validity of the shortened WGCTA version</title>
<p>Criterion validity was assessed by correlating the total scores of the shortened and original WGCTA with CRT performance. The nine-item version showed moderate correlations with CRT-N (<italic>r</italic>&#x202F;=&#x202F;0.381) and CRT-V (<italic>r</italic>&#x202F;=&#x202F;0.299), comparable to those of the original version (<italic>r</italic>&#x202F;=&#x202F;0.329 and <italic>r</italic>&#x202F;=&#x202F;0.298, respectively).</p>
<p>Differences between the correlations of the shortened and full WGCTA versions with external criteria were tested using Steiger&#x2019;s test (<xref ref-type="bibr" rid="ref64">Steiger, 1980</xref>), which accounts for the dependency between correlations sharing a common variable. Results indicated no statistically significant differences, suggesting that the shortened version preserves the criterion validity of the full test.</p>
</sec>
<sec id="sec13">
<label>3.5</label>
<title>Replication of the shortened WGCTA in an independent sample</title>
<p>The nine-item unidimensional model was also validated on an independent sample of university students using the Rasch model. The model demonstrated good fit to the data (&#x03C7;<sup>2</sup>(35)&#x202F;=&#x202F;40.768, <italic>p</italic>&#x202F;=&#x202F;0.231; RMSEA&#x202F;=&#x202F;0.035; SRMR&#x202F;=&#x202F;0.09) and high reliability (<italic>&#x03C9;</italic>&#x202F;=&#x202F;0.70). Infit and outfit values for all items fell within the acceptable range of &#x00B1;2, confirming the adequate fit of individual items. Furthermore, S-X<sup>2</sup> tests and item characteristic curves indicated an even distribution of the probability of correct responses across the latent trait range (&#x2212;2 to +2), supporting the suitability of the items for the shortened scale.</p>
<p>Criterion validity was also examined by correlating the shortened WGCTA with CRT-N (<italic>r</italic>&#x202F;=&#x202F;0.303, <italic>p</italic>&#x202F;&#x003C;&#x202F;0.001) and CRT-V (<italic>r</italic>&#x202F;=&#x202F;0.307, <italic>p</italic>&#x202F;&#x003C;&#x202F;0.001). The correlations were moderate and consistent with the results from the first data collection, further supporting the criterion validity of the nine-item version on an independent sample.</p>
</sec>
</sec>
<sec sec-type="discussion" id="sec14">
<label>4</label>
<title>Discussion</title>
<p>The present study examined the psychometric properties of the Slovak version of the WGCTA-II (form C) in a sample of university students and evaluated a shortened version that maintains psychometric rigor while offering greater practical feasibility. Specifically, the study assessed whether the original structure of the WGCTA is supported in a new cultural and linguistic context and sought to identify potential sources of measurement instability at both the subscale and item levels. By integrating empirical analyses and theoretical considerations, we aimed to provide a reliable and efficient tool for both research and educational applications.</p>
<p>Across multiple levels of analysis, the results consistently indicated limited support for the intended multidimensional structure, including poor model fit of several subscales, weak and unstable item&#x2013;factor relationships, and low internal consistency at the dimension level. From a measurement perspective, these findings also precluded meaningful testing of higher-order or hierarchical CFA models, as the prerequisite unidimensionality at the subscale level was not met. Although the total WGCTA score demonstrated acceptable reliability, this result should be interpreted with caution, as it appears to be driven primarily by test length rather than by coherent measurement of distinct components of critical thinking. In contrast, the convergence of low subscale reliability, inadequate factor loadings, and inconsistent dimensional fit raises substantial concerns regarding the construct validity of the original subscale configuration. The findings point to a broader pattern of psychometric weaknesses in the full Slovak version of the WGCTA-II, and together with its considerable time demands and complex administration, suggest that it is not optimal for research purposes. This conclusion aligns with previous findings (e.g., <xref ref-type="bibr" rid="ref5">Bernard et al., 2008</xref>; <xref ref-type="bibr" rid="ref36">Ku, 2009</xref>; <xref ref-type="bibr" rid="ref39">Liu et al., 2014</xref>; <xref ref-type="bibr" rid="ref68">Verburgh et al., 2013</xref>), and reflects the broader challenges associated with validating the factorial structure of instruments designed to measure complex constructs such as critical thinking, particularly when individual items are intentionally constructed to engage multiple, interrelated cognitive processes rather than discrete abilities (<xref ref-type="bibr" rid="ref38">Leach et al., 2020</xref>).</p>
<p>Current trends in critical thinking research point toward the development of shorter, psychometrically robust, and culturally adapted instruments (<xref ref-type="bibr" rid="ref44">Payan-Carreira et al., 2022</xref>; <xref ref-type="bibr" rid="ref52">Rivas et al., 2023</xref>; <xref ref-type="bibr" rid="ref18">Fabio et al., 2025</xref>; <xref ref-type="bibr" rid="ref9">Butler et al., 2012</xref>). Test shortening allows for more efficient assessment while maintaining validity and reducing the administrative burden on both participants and researchers. In line with this approach, we focused on developing a shortened version of the WGCTA that combines conceptually key dimensions and empirically validated items while preserving psychometric quality.</p>
<p>In selecting the most appropriate indicators of critical thinking, we applied a combination of theoretical and empirical criteria (<xref ref-type="bibr" rid="ref12">DeVellis, 2017</xref>; <xref ref-type="bibr" rid="ref34">Kline, 2016</xref>). Rather than attempting to preserve the original five-subscale structure at all costs, we adopted a parsimonious approach that prioritized dimensions demonstrating both empirical stability and conceptual centrality to critical thinking. The empirical criteria included factor loadings, model fit, and reliability indices, while the theoretical criteria were derived from the RED model of critical thinking and considered the content validity. From the construct perspective, we retained two dimensions&#x2014;Interpretation and Evaluation of Arguments&#x2014;which represent the core processes of critical thinking. The Interpretation dimension reflects the ability to assess the reliability and relevance of evidence and to avoid common cognitive biases such as reasoning fallacies, jumping to conclusions, confusing correlation with causation, or the indefinite pronoun fallacy (<xref ref-type="bibr" rid="ref62">Stanovich and West, 2008</xref>; <xref ref-type="bibr" rid="ref32">Kahneman, 2011</xref>). The Evaluation of Arguments dimension complements Interpretation by focusing on the assessment of the logical strength of claims and the identification of manipulative or insufficiently supported arguments. These dimensions showed comparatively higher internal consistency, more interpretable factor loadings, and more balanced item difficulty profiles than the remaining subscales.</p>
<p>The selection of these two dimensions is also consistent with broader theoretical conceptualizations of critical thinking (e.g., <xref ref-type="bibr" rid="ref19">Facione, 1990</xref>; <xref ref-type="bibr" rid="ref28">Halpern, 2014</xref>). The core skills identified by the widely recognized Delphi panel &#x2013; analysis, evaluation, and inference &#x2013; align with our dimensions, as they involve recognizing argument structure, assessing credibility, and evaluating evidence. Similarly, Reflective judgment theory (<xref ref-type="bibr" rid="ref33">King and Kitchener, 2004</xref>), the Dual Process Model (<xref ref-type="bibr" rid="ref63">Stanovich et al., 2016</xref>) and Kuhn&#x2019;s framework of argumentative reasoning (<xref ref-type="bibr" rid="ref37">Kuhn, 1991</xref>) emphasize analytic and evaluative reasoning as central components of critical thinking. Taken together, this body of evidence indicates that interpretation and evaluation of arguments constitute the core of critical thinking across theoretical models and represent fundamental skills for both academic and civic competence, internationally (<xref ref-type="bibr" rid="ref67">Van Gelder, 2012</xref>; <xref ref-type="bibr" rid="ref39">Liu et al., 2014</xref>; <xref ref-type="bibr" rid="ref43">Mueller et al., 2020</xref>; <xref ref-type="bibr" rid="ref3">American Psychological Association, 2013</xref>, <xref ref-type="bibr" rid="ref4">2016</xref>; <xref ref-type="bibr" rid="ref79">Yulian, 2021</xref>) as well as within the Slovak context (<xref ref-type="bibr" rid="ref50">Research Team, 2018</xref>; <xref ref-type="bibr" rid="ref35">Kotlebov&#x00E1; and Hankerov&#x00E1;, 2024</xref>). In today&#x2019;s information-rich environment, these two processes are critical for accurate reasoning and judgment, whereas other components, though supportive, do not form the core of critical thinking.</p>
<p>The only RED Model dimension not included in the final version was Recognizing Assumptions. Although this dimension &#x2013; the ability to distinguish facts from opinions and identify implicit beliefs &#x2013; represents a relevant but not central component of critical thinking, its psychometric performance was unsatisfactory. The items showed low internal consistency and limited contribution to the overall construct, failing to adequately differentiate between individuals with higher and lower levels of critical thinking. This pattern is consistent with theoretical perspectives on paradigmatic assumptions, which suggest that deeply held assumptions are often implicit and difficult to access through self-report or decontextualized questionnaire formats, thereby limiting the sensitivity of such items to individual differences (<xref ref-type="bibr" rid="ref8">Brookfield, 2015</xref>).</p>
<p>Moreover, item selection also considered content and contextual appropriateness, in addition to statistical indicators. Items requiring work with evidence, identification of logical fallacies, and evaluation of argumentative strength were prioritized, while less suitable items were excluded due to culturally specific contexts, ambiguous wording, or excessive linguistic complexity. A balanced representation of items with correct and incorrect answers was maintained to support content balance and measurement fairness. These results highlight the importance of culturally sensitive adaptation, as even originally validated items may function differently across linguistic and cultural contexts (<xref ref-type="bibr" rid="ref9">Butler et al., 2012</xref>; <xref ref-type="bibr" rid="ref18">Fabio et al., 2025</xref>). In addition, findings underscore that psychometric quality cannot be separated from cultural and linguistic appropriateness, highlighting the necessity of careful adaptation when applying standardized critical thinking measures across contexts.</p>
<p>The results of the IRT analysis indicated that selecting items that were both psychometrically and content-wise strong produced a unidimensional model with a reliable and valid structure, confirmed in an independent replication sample. Conceptually, interpretation and evaluation of arguments represent distinct but closely related processes of critical thinking. Empirically, however, their strong interrelation within the psychometrically refined item set resulted in a unidimensional measurement model in the shortened WGCTA. Importantly, the unidimensionality should not be interpreted as evidence that critical thinking itself is a single, undifferentiated ability, but rather as a property of the selected measurement model. This distinction is crucial, as unidimensionality at the measurement level does not preclude multidimensionality at the conceptual level and is consistent with theoretical and empirical perspectives that conceptualize critical thinking as a system of interrelated core processes (<xref ref-type="bibr" rid="ref5">Bernard et al., 2008</xref>; <xref ref-type="bibr" rid="ref14">Dwyer et al., 2014</xref>; <xref ref-type="bibr" rid="ref26">Grimm and Richter, 2024</xref>; <xref ref-type="bibr" rid="ref18">Fabio et al., 2025</xref>; <xref ref-type="bibr" rid="ref52">Rivas et al., 2023</xref>). Given the limited fit and instability of multi-factor solutions observed in the present data, the unidimensional model provided the most robust and interpretable representation.</p>
<p>The relationship between the shortened version and the Cognitive Reflection Test (CRT) provided additional evidence of criterion validity. A moderate relationship suggests that the two tests capture related, yet not identical, processes. While the CRT focuses on the inhibition of intuitive responses (System 1), the WGCTA reflects engagement of System 2 as well as broader analytical and argumentative skills. The ability to stop and start logical processing of information can be considered a prerequisite for CT. In this context, cognitive reflection may reflect a core cognitive mechanism underlying critical thinking, whereas the WGCTA captures the broader set of reasoning and evaluative skills, indicating a complementary, rather than redundant, relationship between the two measures. By examining this relationship across both the original and shortened WGCTA versions and replicating the findings in an independent sample, the present study further demonstrates the robustness, generalizability, and preservation of the theoretical and empirical properties of the shortened form (<xref ref-type="bibr" rid="ref24">Furr, 2021</xref>).</p>
<sec id="sec15">
<label>4.1</label>
<title>Limitation</title>
<p>The present study has several limitations. For the purposes of this research, we employed the version of the WGCTA that is currently the only standardized version available in our context, as more recent editions, although available internationally, have not yet been adapted or validated locally. Consequently, the results of this study are limited to this older version, which should be considered when interpreting the findings. Second, the shortened version of the test focuses on key, but not all, dimensions of critical thinking. Therefore, the results should be interpreted as a screening indicator of critical thinking, primarily suitable for research purposes, rather than as a comprehensive diagnostic assessment. Third, criterion validity was assessed only in relation to the CRT. Future studies should include additional indicators of criterion validity, such as other tests or performance-based tasks. It would also be valuable to examine the predictive validity of the test in relation to students&#x2019; cognitive and academic outcomes and to determine the extent to which it captures changes in critical thinking following specific courses or training interventions. Furthermore, it would be beneficial to investigate how the test performs across different academic disciplines, age groups, or in online settings, where critical thinking may manifest differently than in traditional testing situations.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="sec16">
<label>5</label>
<title>Conclusion</title>
<p>Despite these limitations, the validated and reliable 9-item version of the WGCTA provides a practical and culturally adapted tool for the rapid assessment of critical thinking in higher education. Its shortened format reduces administration time and cognitive load, making it suitable for large-scale research, course evaluation, and studies examining instructional interventions aimed at developing critical thinking. For educational institutions, it offers an accessible means of monitoring the development of a key 21st-century competency, while maintaining sufficient psychometric quality for research and applied assessment contexts.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec17">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="ethics-statement" id="sec18">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Ethics Committee of Constantine the Philosopher University in Nitra, Ethics Committee Approval Form number: UKF/370/2025/191013:024. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p>
<p>All items used in this study were sourced from the officially licensed Slovak edition of the Watson-Glaser Critical Thinking Appraisal &#x2013; WGCTA-II (Form C) (Watson and Glaser, 2000). The research did not modify the content or wording of the items; rather, it involved empirical evaluation and selection of existing, licensed items for research purposes. Therefore, the shortened version does not constitute a new test, but rather a research subset of items suitable for academic studies, with usage subject to obtaining a license from the copyright holder. The shortened version was created exclusively as a research adaptation for academic purposes and is not intended as a commercially distributed product.</p>
</sec>
<sec sec-type="author-contributions" id="sec19">
<title>Author contributions</title>
<p>G&#x0160;: Funding acquisition, Conceptualization, Writing &#x2013; review &#x0026; editing, Formal analysis, Writing &#x2013; original draft, Methodology. LR: Formal analysis, Writing &#x2013; review &#x0026; editing, Data curation, Software. JU: Writing &#x2013; review &#x0026; editing, Investigation, Data curation, Validation. EL: Writing &#x2013; review &#x0026; editing, Validation. TS: Validation, Writing &#x2013; review &#x0026; editing, Supervision.</p>
</sec>
<sec sec-type="COI-statement" id="sec20">
<title>Conflict of interest</title>
<p>The author(s) declared that this work was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec21">
<title>Generative AI statement</title>
<p>The author(s) declared that Generative AI was not used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="sec22">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec23">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fpsyg.2026.1764712/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fpsyg.2026.1764712/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Abrami</surname><given-names>P. C.</given-names></name> <name><surname>Bernard</surname><given-names>R. M.</given-names></name> <name><surname>Borokhovski</surname><given-names>E.</given-names></name> <name><surname>Wade</surname><given-names>A.</given-names></name> <name><surname>Surkes</surname><given-names>M. A.</given-names></name> <name><surname>Tamim</surname><given-names>R.</given-names></name> <etal/></person-group>. (<year>2008</year>). <article-title>Instructional interventions affecting critical thinking skills and dispositions: a stage 1 meta-analysis</article-title>. <source>Rev. Educ. Res.</source> <volume>78</volume>, <fpage>1102</fpage>&#x2013;<lpage>1134</lpage>. doi: <pub-id pub-id-type="doi">10.3102/0034654308326084</pub-id></mixed-citation></ref>
<ref id="ref2"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Afzal</surname><given-names>A.</given-names></name> <name><surname>Behlol</surname><given-names>P. G.</given-names></name> <name><surname>Sabir</surname><given-names>D.</given-names></name></person-group> (<year>2024</year>). <article-title>Testing effectiveness of critical thinking interventions in teaching English at secondary level</article-title>. <source>Pak. J. Psychol. Res.</source> <volume>39</volume>, <fpage>451</fpage>&#x2013;<lpage>467</lpage>. doi: <pub-id pub-id-type="doi">10.33824/PJPR.2024.39.3.25</pub-id></mixed-citation></ref>
<ref id="ref3"><mixed-citation publication-type="book"><person-group person-group-type="author"><collab id="coll1">American Psychological Association</collab></person-group> (<year>2013</year>). <source>APA guidelines for the undergraduate psychology major: Version 2.0</source>. <publisher-loc>Washington, DC</publisher-loc>: <publisher-name>American Psychological Association</publisher-name>.</mixed-citation></ref>
<ref id="ref4"><mixed-citation publication-type="book"><person-group person-group-type="author"><collab id="coll2">American Psychological Association</collab></person-group> (<year>2016</year>). <source>APA guidelines for the undergraduate psychology major: Version 2.0</source>. <publisher-loc>Washington, DC</publisher-loc>: <publisher-name>American Psychological Association</publisher-name>.</mixed-citation></ref>
<ref id="ref5"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bernard</surname><given-names>R.</given-names></name> <name><surname>Zhang</surname><given-names>D.</given-names></name> <name><surname>Abrami</surname><given-names>P.</given-names></name> <name><surname>Sicoly</surname><given-names>F.</given-names></name> <name><surname>Borokhovski</surname><given-names>E.</given-names></name> <name><surname>Surkes</surname><given-names>M.</given-names></name></person-group> (<year>2008</year>). <article-title>Exploring the structure of the Watson&#x2013;Glaser critical thinking appraisal: one scale or many subscales?</article-title> <source>Think. Skills Creat.</source> <volume>3</volume>, <fpage>15</fpage>&#x2013;<lpage>22</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.tsc.2007.11.001</pub-id></mixed-citation></ref>
<ref id="ref6"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Bond</surname><given-names>T.</given-names></name></person-group> (<year>2015</year>). <source>Applying the Rasch model: Fundamental measurement in the human sciences</source>. <publisher-loc>London</publisher-loc>: <publisher-name>Routledge</publisher-name>.</mixed-citation></ref>
<ref id="ref7"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bonnefon</surname><given-names>J. F.</given-names></name></person-group> (<year>2018</year>). <article-title>The pros and cons of identifying critical thinking with system 2 processing</article-title>. <source>Topoi</source> <volume>37</volume>, <fpage>113</fpage>&#x2013;<lpage>119</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11245-016-9375-2</pub-id></mixed-citation></ref>
<ref id="ref8"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brookfield</surname><given-names>S. D.</given-names></name></person-group> (<year>2015</year>). <article-title>Teaching students to think critically about social media</article-title>. <source>New Dir. Teach. Learn.</source> <volume>2015</volume>, <fpage>47</fpage>&#x2013;<lpage>56</lpage>. doi: <pub-id pub-id-type="doi">10.1002/tl.20162</pub-id></mixed-citation></ref>
<ref id="ref9"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Butler</surname><given-names>H. A.</given-names></name> <name><surname>Dwyer</surname><given-names>C. P.</given-names></name> <name><surname>Hogan</surname><given-names>M. J.</given-names></name> <name><surname>Franco</surname><given-names>A.</given-names></name> <name><surname>Rivas</surname><given-names>S. F.</given-names></name> <name><surname>Saiz</surname><given-names>C.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>The Halpern critical thinking assessment and real-world outcomes: cross-national applications</article-title>. <source>Think. Skills Creat.</source> <volume>7</volume>, <fpage>112</fpage>&#x2013;<lpage>121</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.tsc.2012.04.001</pub-id></mixed-citation></ref>
<ref id="ref10"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chalmers</surname><given-names>R. P.</given-names></name></person-group> (<year>2012</year>). <article-title>Mirt: a multidimensional item response theory package for the R environment</article-title>. <source>J. Stat. Softw.</source> <volume>48</volume>, <fpage>1</fpage>&#x2013;<lpage>29</lpage>. doi: <pub-id pub-id-type="doi">10.18637/jss.v048.i06</pub-id></mixed-citation></ref>
<ref id="ref11"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cortina</surname><given-names>J. M.</given-names></name></person-group> (<year>1993</year>). <article-title>What is coefficient alpha? An examination of theory and applications</article-title>. <source>J. Appl. Psychol.</source> <volume>78</volume>, <fpage>98</fpage>&#x2013;<lpage>104</lpage>. doi: <pub-id pub-id-type="doi">10.1037/0021-9010.78.1.98</pub-id></mixed-citation></ref>
<ref id="ref12"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>DeVellis</surname><given-names>R. F.</given-names></name></person-group> (<year>2017</year>). <source>Scale development: theory and applications</source>. <publisher-loc>Thousand Oaks, CA</publisher-loc>: <publisher-name>Sage</publisher-name>.</mixed-citation></ref>
<ref id="ref13"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Diedenhofen</surname><given-names>B.</given-names></name> <name><surname>Musch</surname><given-names>J.</given-names></name></person-group> (<year>2015</year>). <article-title>Cocor: a comprehensive solution for the statistical comparison of correlations</article-title>. <source>PLoS One</source> <volume>10</volume>:<fpage>e0121945</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0121945</pub-id>, <pub-id pub-id-type="pmid">25835001</pub-id></mixed-citation></ref>
<ref id="ref14"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dwyer</surname><given-names>C. P.</given-names></name> <name><surname>Hogan</surname><given-names>M. J.</given-names></name> <name><surname>Stewart</surname><given-names>I.</given-names></name></person-group> (<year>2014</year>). <article-title>An integrated critical thinking framework for the 21st century</article-title>. <source>Think. Skills Creat.</source> <volume>12</volume>, <fpage>43</fpage>&#x2013;<lpage>52</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.tsc.2013.12.004</pub-id></mixed-citation></ref>
<ref id="ref15"><mixed-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Ejiogu</surname><given-names>K. C.</given-names></name> <name><surname>Yang</surname><given-names>Z.</given-names></name> <name><surname>Trent</surname><given-names>J.</given-names></name> <name><surname>Rose</surname><given-names>M.</given-names></name></person-group> (<year>2006</year>). <article-title>Understanding the relationship between critical thinking and job performance</article-title>. <conf-name>Poster presented at the 21st annual conference of the Society for Industrial and Organizational Psychology</conf-name>, <publisher-name>SIOP</publisher-name>: <conf-loc>Dallas, TX</conf-loc></mixed-citation></ref>
<ref id="ref16"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ennis</surname><given-names>R.</given-names></name></person-group> (<year>2011</year>). <article-title>Critical thinking</article-title>. <source>Inquiry</source> <volume>26</volume>, <fpage>4</fpage>&#x2013;<lpage>18</lpage>. doi: <pub-id pub-id-type="doi">10.5840/inquiryctnews20112613</pub-id></mixed-citation></ref>
<ref id="ref17"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Ennis</surname><given-names>R. H.</given-names></name> <name><surname>Millman</surname><given-names>J.</given-names></name></person-group> (<year>1985</year>). <source>Cornell critical thinking test, level Z</source>. <publisher-loc>Pacific Grove, CA</publisher-loc>: <publisher-name>Midwest Publications</publisher-name>.</mixed-citation></ref>
<ref id="ref18"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fabio</surname><given-names>R. A.</given-names></name> <name><surname>Plebe</surname><given-names>A.</given-names></name> <name><surname>Ascone</surname><given-names>C.</given-names></name> <name><surname>Suriano</surname><given-names>R.</given-names></name></person-group> (<year>2025</year>). <article-title>Psychometric properties and validation of the critical reasoning assessment</article-title>. <source>Personal. Individ. Differ.</source> <volume>246</volume>:<fpage>113344</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.paid.2025.113344</pub-id></mixed-citation></ref>
<ref id="ref19"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Facione</surname><given-names>P. A.</given-names></name></person-group> (<year>1990</year>). <source>Critical thinking: A statement of expert consensus for purposes of educational assessment and instruction (the Delphi report)</source>. <publisher-loc>Millbrae, CA</publisher-loc>: <publisher-name>The California Academic Press</publisher-name>.</mixed-citation></ref>
<ref id="ref20"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Facione</surname><given-names>P. A.</given-names></name> <name><surname>Facione</surname><given-names>N. C.</given-names></name></person-group> (<year>2001</year>). <article-title>Analyzing explanations for seemingly irrational choices: linking argument analysis and cognitive science</article-title>. <source>Int. J. Appl. Philos.</source> <volume>15</volume>, <fpage>267</fpage>&#x2013;<lpage>286</lpage>. doi: <pub-id pub-id-type="doi">10.5840/ijap200115217</pub-id></mixed-citation></ref>
<ref id="ref21"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fawkes</surname><given-names>D.</given-names></name> <name><surname>Adajian</surname><given-names>T.</given-names></name> <name><surname>Flage</surname><given-names>D.</given-names></name> <name><surname>Hoeltzel</surname><given-names>S.</given-names></name> <name><surname>Knorpp</surname><given-names>B.</given-names></name> <name><surname>O'Meara</surname><given-names>B.</given-names></name> <etal/></person-group>. (<year>2003</year>). <article-title>Examining the exam: a critical look at the Watson-Glaser critical thinking appraisal exam</article-title>. <source>Inquiry</source> <volume>21</volume>, <fpage>31</fpage>&#x2013;<lpage>46</lpage>. doi: <pub-id pub-id-type="doi">10.5840/inquiryctnews200321316</pub-id></mixed-citation></ref>
<ref id="ref22"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Fox</surname><given-names>J.</given-names></name></person-group> (<year>2019</year>). Polycor: Polychoric and polyserial correlations. R package version 0.7-11. Available online at: <ext-link xlink:href="https://CRAN.R-project.org/package=polycor" ext-link-type="uri">https://CRAN.R-project.org/package=polycor</ext-link> (Accessed November 10, 2025).</mixed-citation></ref>
<ref id="ref23"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Frederick</surname><given-names>S.</given-names></name></person-group> (<year>2005</year>). <article-title>Cognitive reflection and decision making</article-title>. <source>J. Econ. Perspect.</source> <volume>19</volume>, <fpage>25</fpage>&#x2013;<lpage>42</lpage>. doi: <pub-id pub-id-type="doi">10.1257/089533005775196732</pub-id></mixed-citation></ref>
<ref id="ref24"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Furr</surname><given-names>R. M.</given-names></name></person-group> (<year>2021</year>). <source>Psychometrics: an introduction</source>. <publisher-loc>Thousand Oaks, CA</publisher-loc>: <publisher-name>Sage Publications</publisher-name>.</mixed-citation></ref>
<ref id="ref25"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gadzella</surname><given-names>B. M.</given-names></name> <name><surname>Hogan</surname><given-names>L.</given-names></name> <name><surname>Masten</surname><given-names>W.</given-names></name> <name><surname>Stacks</surname><given-names>J.</given-names></name> <name><surname>Stephens</surname><given-names>R.</given-names></name> <name><surname>Zascavage</surname><given-names>V.</given-names></name></person-group> (<year>2006</year>). <article-title>Reliability and validity of the watson&#x2013;glaser critical thinking appraisal-forms for different academic groups</article-title>. <source>J. Instr. Psychol.</source> <volume>33</volume>, <fpage>141</fpage>&#x2013;<lpage>143</lpage>.</mixed-citation></ref>
<ref id="ref26"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Grimm</surname><given-names>J.</given-names></name> <name><surname>Richter</surname><given-names>T.</given-names></name></person-group> (<year>2024</year>). <article-title>Rational thinking as a general cognitive ability: factorial structure, underlying cognitive processes, and relevance for university academic success</article-title>. <source>Learn. Individ. Differ.</source> <volume>111</volume>:<fpage>102428</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.lindif.2024.102428</pub-id></mixed-citation></ref>
<ref id="ref27"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Halpern</surname><given-names>D. F.</given-names></name></person-group> (<year>2010</year>). <source>Halpern critical thinking assessment</source>. <publisher-loc>Vienna</publisher-loc>: <publisher-name>Schuhfried</publisher-name>.</mixed-citation></ref>
<ref id="ref28"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Halpern</surname><given-names>D. F.</given-names></name></person-group> (<year>2014</year>). <source>Thought and knowledge: an introduction to critical thinking</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Psychology Press</publisher-name>.</mixed-citation></ref>
<ref id="ref29"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hassan</surname><given-names>K. E.</given-names></name> <name><surname>Madhum</surname><given-names>G.</given-names></name></person-group> (<year>2007</year>). <article-title>Validating the Watson&#x2013;Glaser critical thinking appraisal</article-title>. <source>High. Educ.</source> <volume>54</volume>, <fpage>361</fpage>&#x2013;<lpage>383</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10734-006-9002-z</pub-id></mixed-citation></ref>
<ref id="ref30"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname><given-names>L. T.</given-names></name> <name><surname>Bentler</surname><given-names>P. M.</given-names></name></person-group> (<year>1999</year>). <article-title>Cutoff criteria for fit indexes in covariance structure analysis: conventional criteria versus new alternatives</article-title>. <source>Struct. Equ. Model. Multidiscip. J.</source> <volume>6</volume>, <fpage>1</fpage>&#x2013;<lpage>55</lpage>. doi: <pub-id pub-id-type="doi">10.1080/10705519909540118</pub-id></mixed-citation></ref>
<ref id="ref31"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Jorgensen</surname><given-names>T. D.</given-names></name> <name><surname>Pornprasertmanit</surname><given-names>S.</given-names></name> <name><surname>Schoemann</surname><given-names>A. M.</given-names></name> <name><surname>Rosseel</surname><given-names>Y.</given-names></name></person-group> (<year>2022</year>). semTools: useful tools for structural equation modelling. Available online at: <ext-link xlink:href="https://CRAN.R-project.org/package=semTools" ext-link-type="uri">https://CRAN.R-project.org/package=semTools</ext-link> (Accessed November 10, 2025).</mixed-citation></ref>
<ref id="ref32"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Kahneman</surname><given-names>D.</given-names></name></person-group> (<year>2011</year>). <source>Thinking, fast and slow</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Farrar, Straus and Giroux</publisher-name>.</mixed-citation></ref>
<ref id="ref33"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>King</surname><given-names>P. M.</given-names></name> <name><surname>Kitchener</surname><given-names>K. S.</given-names></name></person-group> (<year>2004</year>). <article-title>Reflective judgment: theory and research on the development of epistemic assumptions through adulthood</article-title>. <source>Educ. Psychol.</source> <volume>39</volume>, <fpage>5</fpage>&#x2013;<lpage>18</lpage>. doi: <pub-id pub-id-type="doi">10.1207/s15326985ep3901_2</pub-id></mixed-citation></ref>
<ref id="ref34"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Kline</surname><given-names>R. B.</given-names></name></person-group> (<year>2016</year>). <source>Principles and practice of structural equation modeling</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Guilford Press</publisher-name>.</mixed-citation></ref>
<ref id="ref35"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kotlebov&#x00E1;</surname><given-names>I.</given-names></name> <name><surname>Hankerov&#x00E1;</surname><given-names>M.</given-names></name></person-group> (<year>2024</year>). <article-title>V&#x00FD;znam vyu&#x017E;&#x00ED;vania kritick&#x00E9;ho &#x010D;&#x00ED;tania a myslenia</article-title>. <source>Philologia</source> <volume>14</volume>, <fpage>69</fpage>&#x2013;<lpage>81</lpage>. doi: <pub-id pub-id-type="doi">10.18355/XL.2024.14.01.07</pub-id></mixed-citation></ref>
<ref id="ref36"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ku</surname><given-names>K. Y. L.</given-names></name></person-group> (<year>2009</year>). <article-title>Assessing students&#x2019; critical thinking performance: urging for measurements using multi-response format</article-title>. <source>Think. Skills Creat.</source> <volume>4</volume>, <fpage>70</fpage>&#x2013;<lpage>76</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.tsc.2009.02.001</pub-id></mixed-citation></ref>
<ref id="ref37"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Kuhn</surname><given-names>D.</given-names></name></person-group> (<year>1991</year>). <source>The skills of argument</source>. <publisher-loc>Cambridge, UK</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>.</mixed-citation></ref>
<ref id="ref38"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Leach</surname><given-names>S.</given-names></name> <name><surname>Immekus</surname><given-names>J. C.</given-names></name> <name><surname>Hand</surname><given-names>B.</given-names></name></person-group> (<year>2020</year>). <article-title>The factorial validity of the Cornell critical thinking tests: a multi-analytic approach</article-title>. <source>Think. Skills Creat.</source> <volume>37</volume>:<fpage>100676</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.tsc.2020.100676</pub-id></mixed-citation></ref>
<ref id="ref39"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>O. L.</given-names></name> <name><surname>Frankel</surname><given-names>L.</given-names></name> <name><surname>Roohr</surname><given-names>K.</given-names></name></person-group> (<year>2014</year>). <article-title>Assessing critical thinking in higher education: current state and directions for next-generation assessment</article-title>. <source>ETS Res. Rep. Ser.</source> <volume>2014</volume>, <fpage>1</fpage>&#x2013;<lpage>23</lpage>. doi: <pub-id pub-id-type="doi">10.1002/ets2.12009</pub-id></mixed-citation></ref>
<ref id="ref40"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Loo</surname><given-names>R.</given-names></name> <name><surname>Thorpe</surname><given-names>K.</given-names></name></person-group> (<year>1999</year>). <article-title>A psychometric investigation of scores on the Watson-Glaser critical thinking appraisal new form S</article-title>. <source>Educ. Psychol. Meas.</source> <volume>59</volume>, <fpage>995</fpage>&#x2013;<lpage>1003</lpage>. doi: <pub-id pub-id-type="doi">10.1177/00131649921970305</pub-id></mixed-citation></ref>
<ref id="ref41"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Minnameier</surname><given-names>G.</given-names></name> <name><surname>Hermkes</surname><given-names>R.</given-names></name></person-group> (<year>2020</year>). <article-title>Learning to fly through informational turbulence: critical thinking and the case of the minimum wage</article-title>. <source>Front. Educ.</source> <volume>5</volume>:<fpage>573020</fpage>. doi: <pub-id pub-id-type="doi">10.3389/feduc.2020.573020</pub-id></mixed-citation></ref>
<ref id="ref42"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Moore</surname><given-names>T. J.</given-names></name></person-group> (<year>2011</year>). <article-title>Critical thinking and disciplinary thinking: a continuing debate</article-title>. <source>High. Educ. Res. Dev.</source> <volume>30</volume>, <fpage>261</fpage>&#x2013;<lpage>274</lpage>. doi: <pub-id pub-id-type="doi">10.1080/07294360.2010.50132</pub-id></mixed-citation></ref>
<ref id="ref43"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mueller</surname><given-names>J.</given-names></name> <name><surname>Taylor</surname><given-names>H.</given-names></name> <name><surname>Brakke</surname><given-names>K.</given-names></name> <name><surname>Drysdale</surname><given-names>M.</given-names></name> <name><surname>Kelly</surname><given-names>K.</given-names></name> <name><surname>Levine</surname><given-names>G.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Assessment of scientific inquiry and critical thinking: measuring APA goal 2 student learning outcomes</article-title>. <source>Teach. Psychol.</source> <volume>47</volume>, <fpage>274</fpage>&#x2013;<lpage>284</lpage>. doi: <pub-id pub-id-type="doi">10.1177/0098628320945114</pub-id></mixed-citation></ref>
<ref id="ref44"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Payan-Carreira</surname><given-names>R.</given-names></name> <name><surname>Sacau-Fontenla</surname><given-names>A.</given-names></name> <name><surname>Rebelo</surname><given-names>H.</given-names></name> <name><surname>Sebasti&#x00E3;o</surname><given-names>L.</given-names></name> <name><surname>Pnevmatikos</surname><given-names>D.</given-names></name></person-group> (<year>2022</year>). <article-title>Development and validation of a critical thinking assessment scale short form</article-title>. <source>Educ. Sci.</source> <volume>12</volume>:<fpage>938</fpage>. doi: <pub-id pub-id-type="doi">10.3390/educsci12120938</pub-id></mixed-citation></ref>
<ref id="ref45"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Possin</surname><given-names>K.</given-names></name></person-group> (<year>2014</year>). <article-title>Critique of the Watson&#x2013;Glaser critical thinking appraisal test: the more things change, the more they stay the same</article-title>. <source>Inform. Logic</source> <volume>34</volume>, <fpage>65</fpage>&#x2013;<lpage>93</lpage>. doi: <pub-id pub-id-type="doi">10.22329/il.v34i4.4141</pub-id></mixed-citation></ref>
<ref id="ref47"><mixed-citation publication-type="book"><person-group person-group-type="author"><collab id="coll4">R Core Team</collab></person-group> (<year>2022</year>). <source>R: A language and environment for statistical computing</source>. <publisher-loc>Vienna, Austria</publisher-loc>: <publisher-name>R Foundation for Statistical Computing</publisher-name>.</mixed-citation></ref>
<ref id="ref48"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Raykov</surname><given-names>T.</given-names></name></person-group> (<year>1997</year>). <article-title>Scale reliability, Cronbach&#x2019;s coefficient alpha, and violations of essential tau-equivalence with fixed congeneric components</article-title>. <source>Multivar. Behav. Res.</source> <volume>32</volume>, <fpage>329</fpage>&#x2013;<lpage>353</lpage>. doi: <pub-id pub-id-type="doi">10.1207/s15327906mbr3204_2</pub-id>, <pub-id pub-id-type="pmid">26777071</pub-id></mixed-citation></ref>
<ref id="ref49"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ren</surname><given-names>X.</given-names></name> <name><surname>Tong</surname><given-names>Y.</given-names></name> <name><surname>Peng</surname><given-names>P.</given-names></name> <name><surname>Wang</surname><given-names>T.</given-names></name></person-group> (<year>2020</year>). <article-title>Critical thinking predicts academic performance beyond general cognitive ability: evidence from adults and children</article-title>. <source>Intelligence</source> <volume>82</volume>:<fpage>487</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.intell.2020.101487</pub-id></mixed-citation></ref>
<ref id="ref50"><mixed-citation publication-type="journal"><person-group person-group-type="author"><collab id="coll5">Research Team</collab></person-group> (<year>2018</year>). <article-title>Kritick&#x00E9; myslenie ako s&#x00FA;&#x010D;as&#x0165; kurikul&#x00E1;rnej reformy na Slovensku: Systematick&#x00FD; preh&#x013E;ad literat&#x00FA;ry</article-title>. <source>Pedagog. Orien.</source> <volume>28</volume>, <fpage>577</fpage>&#x2013;<lpage>599</lpage>. doi: <pub-id pub-id-type="doi">10.5817/PedOr2018-4-577</pub-id></mixed-citation></ref>
<ref id="ref51"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Revelle</surname><given-names>W.</given-names></name></person-group> (<year>2023</year>). <source>Psych: procedures for psychological, psychometric, and personality research</source>. <publisher-loc>Evanston, Illinois</publisher-loc>: <publisher-name>Northwestern University</publisher-name>.</mixed-citation></ref>
<ref id="ref52"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rivas</surname><given-names>S. F.</given-names></name> <name><surname>Carlos</surname><given-names>S.</given-names></name> <name><surname>Leandro</surname><given-names>S. A.</given-names></name></person-group> (<year>2023</year>). <article-title>The role of critical thinking in predicting and improving academic performance</article-title>. <source>Sustainability</source> <volume>15</volume>:<fpage>1527</fpage>. doi: <pub-id pub-id-type="doi">10.3390/su15021527</pub-id></mixed-citation></ref>
<ref id="ref53"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Robitzsch</surname><given-names>A.</given-names></name> <name><surname>Kiefer</surname><given-names>T.</given-names></name> <name><surname>Wu</surname><given-names>M.</given-names></name></person-group> (<year>2021</year>) TAM: test analysis modules for R. R package version 3. Available online at: <ext-link xlink:href="https://CRAN.R-project.org/package=TAM" ext-link-type="uri">https://CRAN.R-project.org/package=TAM</ext-link> (Accessed November 11, 2025).</mixed-citation></ref>
<ref id="ref54"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rosseel</surname><given-names>Y.</given-names></name></person-group> (<year>2012</year>). <article-title>Lavaan: an R package for structural equation modeling</article-title>. <source>J. Stat. Softw.</source> <volume>48</volume>, <fpage>1</fpage>&#x2013;<lpage>36</lpage>. doi: <pub-id pub-id-type="doi">10.18637/jss.v048.i02</pub-id></mixed-citation></ref>
<ref id="ref55"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rothinam</surname><given-names>N.</given-names></name> <name><surname>Vengrasalam</surname><given-names>R.</given-names></name> <name><surname>Naidu</surname><given-names>S.</given-names></name> <name><surname>Nachiappan</surname><given-names>S.</given-names></name> <name><surname>Jabamoney</surname><given-names>S.</given-names></name></person-group> (<year>2025</year>). <article-title>Systematic literature review on critical thinking in higher education</article-title>. <source>Edelweiss Appl. Sci. Technol.</source> <volume>9</volume>, <fpage>2046</fpage>&#x2013;<lpage>2063</lpage>. doi: <pub-id pub-id-type="doi">10.55214/25768484.v9i5.7377</pub-id></mixed-citation></ref>
<ref id="ref56"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rusmin</surname><given-names>L.</given-names></name> <name><surname>Misrahayu</surname><given-names>Y.</given-names></name> <name><surname>Pongpalilu</surname><given-names>F.</given-names></name> <name><surname>Radiansyah</surname><given-names>R.</given-names></name></person-group> (<year>2024</year>). <article-title>Critical thinking and problem-solving skills in the 21st century</article-title>. <source>J. Soc. Sci.</source> <volume>1</volume>, <fpage>144</fpage>&#x2013;<lpage>162</lpage>. doi: <pub-id pub-id-type="doi">10.59613/svhy3576</pub-id></mixed-citation></ref>
<ref id="ref57"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Rust</surname><given-names>J.</given-names></name></person-group> (<year>2002</year>). <source><italic>Watson Glaser critical thinking appraisal</italic> UK edition-manual</source>. <publisher-loc>London</publisher-loc>: <publisher-name>The Pschychological Corporation</publisher-name>.</mixed-citation></ref>
<ref id="ref59"><mixed-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>&#x0160;ebokov&#x00E1;</surname><given-names>G.</given-names></name> <name><surname>Uhl&#x00E1;rikov&#x00E1;</surname><given-names>J.</given-names></name> <name><surname>Giertlov&#x00E1;</surname><given-names>P.</given-names></name> <name><surname>G&#x00E1;blikov&#x00E1;</surname><given-names>T.</given-names></name></person-group> (<year>2025</year>). <article-title>Activating methods as a moderator of the relation between critical thinking and academic control</article-title>. <conf-name>EDULEARN25 conference proceedings: 17th international conference on education and new learning technologies</conf-name> <fpage>8681</fpage>&#x2013;<lpage>8686</lpage> <conf-loc>Spain</conf-loc>: <publisher-name>EDULEARN</publisher-name></mixed-citation></ref>
<ref id="ref60"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Simonovic</surname><given-names>B. C.</given-names></name> <name><surname>Vione</surname><given-names>K.</given-names></name> <name><surname>Fido</surname><given-names>D.</given-names></name> <name><surname>Stupple</surname><given-names>E.</given-names></name> <name><surname>Martin</surname><given-names>J.</given-names></name> <name><surname>Clarke</surname><given-names>R.</given-names></name></person-group> (<year>2022</year>). <article-title>The impact of attitudes, beliefs, and cognitive reflection on the development of critical thinking skills in online students</article-title>. <source>Online Learn.</source> <volume>26</volume>:<fpage>2725</fpage>. doi: <pub-id pub-id-type="doi">10.24059/olj.v26i2.2725</pub-id></mixed-citation></ref>
<ref id="ref61"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sirota</surname><given-names>M.</given-names></name> <name><surname>Dewberry</surname><given-names>C.</given-names></name> <name><surname>Juanchich</surname><given-names>M.</given-names></name> <name><surname>Valu&#x0161;</surname><given-names>L.</given-names></name> <name><surname>Marshall</surname><given-names>A. C.</given-names></name></person-group> (<year>2020</year>). <article-title>Measuring cognitive reflection without maths: development and validation of the verbal cognitive reflection test</article-title>. <source>J. Behav. Decis. Mak.</source> <volume>34</volume>, <fpage>322</fpage>&#x2013;<lpage>343</lpage>. doi: <pub-id pub-id-type="doi">10.1002/bdm.2213</pub-id></mixed-citation></ref>
<ref id="ref62"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stanovich</surname><given-names>K.</given-names></name> <name><surname>West</surname><given-names>R.</given-names></name></person-group> (<year>2008</year>). <article-title>On the failure of cognitive ability to predict myside and one-sided thinking biases</article-title>. <source>Think. Reason.</source> <volume>14</volume>, <fpage>129</fpage>&#x2013;<lpage>167</lpage>. doi: <pub-id pub-id-type="doi">10.1080/13546780701679764</pub-id></mixed-citation></ref>
<ref id="ref63"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Stanovich</surname><given-names>K. E.</given-names></name> <name><surname>West</surname><given-names>R. F.</given-names></name> <name><surname>Toplak</surname><given-names>M. E.</given-names></name></person-group> (<year>2016</year>). <source>The rationality quotient toward a test of rational thinking</source>. <publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>MIT Press</publisher-name>.</mixed-citation></ref>
<ref id="ref64"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Steiger</surname><given-names>J. H.</given-names></name></person-group> (<year>1980</year>). <article-title>Tests for comparing elements of a correlation matrix</article-title>. <source>Psychol. Bull.</source> <volume>87</volume>, <fpage>245</fpage>&#x2013;<lpage>251</lpage>. doi: <pub-id pub-id-type="doi">10.1037/0033-2909.87.2.245</pub-id></mixed-citation></ref>
<ref id="ref65"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stupple</surname><given-names>E. J. N.</given-names></name> <name><surname>Maratos</surname><given-names>F. A.</given-names></name> <name><surname>Elander</surname><given-names>J.</given-names></name> <name><surname>Hunt</surname><given-names>T. E.</given-names></name> <name><surname>Cheung</surname><given-names>K. Y. F.</given-names></name> <name><surname>Aubeeluck</surname><given-names>A. V.</given-names></name></person-group> (<year>2017</year>). <article-title>Development of the critical thinking toolkit (CriTT): a measure of student attitudes and beliefs about critical thinking</article-title>. <source>Think. Skills Creat.</source> <volume>23</volume>, <fpage>91</fpage>&#x2013;<lpage>100</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.tsc.2016.11.00</pub-id></mixed-citation></ref>
<ref id="ref66"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Suwita</surname><given-names>S.</given-names></name> <name><surname>Saputro</surname><given-names>S.</given-names></name> <name><surname>Sajidan</surname><given-names>S.</given-names></name> <name><surname>Sutarno</surname><given-names>S.</given-names></name></person-group> (<year>2024</year>). <article-title>Assessing lower-secondary school students&#x2019; critical thinking skills in photosynthesis: a Rasch model approach</article-title>. <source>J. Balt. Sci. Educ.</source> <volume>11</volume>, <fpage>1278</fpage>&#x2013;<lpage>1129</lpage>. doi: <pub-id pub-id-type="doi">10.33225/jbse/24.23.1278</pub-id></mixed-citation></ref>
<ref id="ref67"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Van Gelder</surname><given-names>T.</given-names></name></person-group> (<year>2012</year>). &#x201C;<article-title>Argument mapping as a learning tool</article-title>&#x201D; in <source>Critical thinking education and assessment: Can higher order thinking be tested?</source> eds. <person-group person-group-type="editor"><name><surname>Horvath</surname><given-names>C. P.</given-names></name> <name><surname>Forte</surname><given-names>J. M.</given-names></name></person-group> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Hampton Press</publisher-name>), <fpage>125</fpage>&#x2013;<lpage>146</lpage>.</mixed-citation></ref>
<ref id="ref68"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Verburgh</surname><given-names>A.</given-names></name> <name><surname>Fran&#x00E7;ois</surname><given-names>S.</given-names></name> <name><surname>Elen</surname><given-names>J.</given-names></name> <name><surname>Janssen</surname><given-names>R.</given-names></name></person-group> (<year>2013</year>). <article-title>The assessment of critical thinking critically assessed in higher education: a validation study of the CCTT and the HCTA</article-title>. <source>Educ. Res. Int.</source> <volume>2013</volume>:<fpage>198920</fpage>. doi: <pub-id pub-id-type="doi">10.1155/2013/198920</pub-id></mixed-citation></ref>
<ref id="ref69"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Voogt</surname><given-names>J.</given-names></name> <name><surname>Roblin</surname><given-names>N. P.</given-names></name></person-group> (<year>2012</year>). <article-title>A comparative analysis of international frameworks for 21st-century competences: implications for national curriculum policies</article-title>. <source>J. Curric. Stud.</source> <volume>44</volume>, <fpage>299</fpage>&#x2013;<lpage>321</lpage>. doi: <pub-id pub-id-type="doi">10.1080/00220272.2012.668938</pub-id></mixed-citation></ref>
<ref id="ref70"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wagner</surname><given-names>T. A.</given-names></name> <name><surname>Harvey</surname><given-names>R. J.</given-names></name></person-group> (<year>2006</year>). <article-title>Development of a new critical thinking test using item response theory</article-title>. <source>Psychol. Assess.</source> <volume>18</volume>, <fpage>100</fpage>&#x2013;<lpage>105</lpage>. doi: <pub-id pub-id-type="doi">10.1037/1040-3590.18.1.100</pub-id>, <pub-id pub-id-type="pmid">16594818</pub-id></mixed-citation></ref>
<ref id="ref71"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Watson</surname><given-names>G.</given-names></name> <name><surname>Glaser</surname><given-names>E. M.</given-names></name></person-group> (<year>1980</year>). <source><italic>Watson&#x2013;Glaser critical thinking appraisal</italic> (WGCTA)</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>The Psychological Corporation</publisher-name>.</mixed-citation></ref>
<ref id="ref72"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Watson</surname><given-names>G.</given-names></name> <name><surname>Glaser</surname><given-names>E. M.</given-names></name></person-group> (<year>1991</year>). <source>Watson&#x2013;Glaser critical thinking appraisal (WGCTA-II), form C</source>. <publisher-loc>San Antonio, TX</publisher-loc>: <publisher-name>Psychological Corporation</publisher-name>.</mixed-citation></ref>
<ref id="ref73"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Watson</surname><given-names>G.</given-names></name> <name><surname>Glaser</surname><given-names>E. M.</given-names></name></person-group> (<year>2000</year>). <source>Watson-Glaserov test kritick&#x00E9;ho myslenia (Formul&#x00E1;r C)</source>. <publisher-loc>Bratislava</publisher-loc>: <publisher-name>Psychodiagnostika, a.s</publisher-name>.</mixed-citation></ref>
<ref id="ref74"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Watson</surname><given-names>G. B.</given-names></name> <name><surname>Glaser</surname><given-names>E. M.</given-names></name></person-group> (<year>2018</year>). <source>Watson&#x2013;Glaser&#x2122; III Critical Thinking Appraisal: User&#x2019;s guide and technical manual</source>. <publisher-loc>Bloomington, MN</publisher-loc>: <publisher-name>Pearson Assessments</publisher-name>.</mixed-citation></ref>
<ref id="ref75"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Wayas</surname><given-names>K. J.</given-names></name> <name><surname>Alarcon</surname><given-names>J. L.</given-names></name> <name><surname>Sayson</surname><given-names>R.S.</given-names></name> <name><surname>Sacupayo</surname><given-names>G. L.</given-names></name> <name><surname>Wayas</surname><given-names>K.</given-names></name> <name><surname>Loyloy</surname><given-names>L.</given-names></name> <etal/></person-group>. (<year>2024</year>). Measuring the level of critical thinking ability of the students using Watson-Glaser appraisal. Available online at: <ext-link xlink:href="https://zenodo.org/records/11070889" ext-link-type="uri">https://zenodo.org/records/11070889</ext-link>.</mixed-citation></ref>
<ref id="ref76"><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Wickham</surname><given-names>H.</given-names></name></person-group> (<year>2016</year>). <source>ggplot2: Elegant graphics for data analysis</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Springer</publisher-name>.</mixed-citation></ref>
<ref id="ref77"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Wickham</surname><given-names>H.</given-names></name> <name><surname>Fran&#x00E7;ois</surname><given-names>R.</given-names></name> <name><surname>Henry</surname><given-names>L.</given-names></name> <name><surname>M&#x00FC;ller</surname><given-names>K.</given-names></name> <name><surname>Vaughan</surname><given-names>D.</given-names></name></person-group> (<year>2023a</year>). Dplyr: a grammar of data manipulation. Available online at: <ext-link xlink:href="https://CRAN.R-project.org/package=dplyr" ext-link-type="uri">https://CRAN.R-project.org/package=dplyr</ext-link> (Accessed November 11, 2025).</mixed-citation></ref>
<ref id="ref78"><mixed-citation publication-type="other"><person-group person-group-type="author"><name><surname>Wickham</surname><given-names>H.</given-names></name> <name><surname>Vaughan</surname><given-names>D.</given-names></name> <name><surname>Girlich</surname><given-names>M.</given-names></name></person-group> (<year>2023b</year>). Tidyr: tidy messy data. R package version 1.3.0. Available online at: <ext-link xlink:href="https://CRAN.R-project.org/package=tidyr" ext-link-type="uri">https://CRAN.R-project.org/package=tidyr</ext-link> (Accessed November 11, 2025).</mixed-citation></ref>
<ref id="ref79"><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yulian</surname><given-names>R.</given-names></name></person-group> (<year>2021</year>). <article-title>The flipped classroom: improving critical thinking for critical reading of EFL learners in higher education</article-title>. <source>Stud. Engl. Lang. Educ.</source> <volume>8</volume>, <fpage>508</fpage>&#x2013;<lpage>522</lpage>. doi: <pub-id pub-id-type="doi">10.24815/siele.v8i2.18366</pub-id></mixed-citation></ref>
</ref-list>
<fn-group>
<fn fn-type="custom" custom-type="edited-by" id="fn0001">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/13385/overview">Marco Scutari</ext-link>, Dalle Molle Institute for Artificial Intelligence Research, Switzerland</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by" id="fn0002">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3005990/overview">Okta Alpindo</ext-link>, Universitas Maritim Raja Ali Haji, Indonesia</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3317332/overview">Ivana Cimermanova</ext-link>, University of Pre&#x0161;ov, Slovakia</p>
</fn>
</fn-group>
</back>
</article>