<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Educ.</journal-id>
<journal-title>Frontiers in Education</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Educ.</abbrev-journal-title>
<issn pub-type="epub">2504-284X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/feduc.2025.1483092</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Education</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Context matters: adapting and validating the TEDS-instruct observation instrument assessing teaching quality for its use in Norwegian primary education</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Senden</surname> <given-names>Bas</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<xref ref-type="author-notes" rid="fn0003"><sup>&#x2020;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2820660/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Jentsch</surname> <given-names>Armin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2378122/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Teig</surname> <given-names>Nani</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/329590/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Nilsen</surname> <given-names>Trude</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/381057/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Fauskanger</surname> <given-names>Janne</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2909553/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Fr&#x00E5;g&#x00E5;t</surname> <given-names>Thomas</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Maugesten</surname> <given-names>Marianne</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Mork</surname> <given-names>Sonja Merethe</given-names></name>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Mosvold</surname> <given-names>Reidar</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Nortvedt</surname> <given-names>Guri A.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Olufsen</surname> <given-names>Magne</given-names></name>
<xref ref-type="aff" rid="aff8"><sup>8</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2903226/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Sj&#x00F8;berg</surname> <given-names>Mari</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff9"><sup>9</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Staberg</surname> <given-names>Ragnhild Lyngved</given-names></name>
<xref ref-type="aff" rid="aff10"><sup>10</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Selling</surname> <given-names>Alexander</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Stovner</surname> <given-names>Roar Bakken</given-names></name>
<xref ref-type="aff" rid="aff11"><sup>11</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2820946/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>&#x00D8;degaard</surname> <given-names>Marianne</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Teacher Education and School Research, University of Oslo</institution>, <addr-line>Oslo</addr-line>, <country>Norway</country></aff>
<aff id="aff2"><sup>2</sup><institution>Center for Research on Equality in Education, University of Oslo</institution>, <addr-line>Oslo</addr-line>, <country>Norway</country></aff>
<aff id="aff3"><sup>3</sup><institution>University of Stavanger</institution>, <addr-line>Stavanger</addr-line>, <country>Norway</country></aff>
<aff id="aff4"><sup>4</sup><institution>Norwegian Centre for Mathematics Education, Norwegian University of Science and Technology (NTNU)</institution>, <addr-line>Trondheim</addr-line>, <country>Norway</country></aff>
<aff id="aff5"><sup>5</sup><institution>Department of Mathematics, Natural Sciences, and Physical Education, University of Inland Norway</institution>, <addr-line>Hamar</addr-line>, <country>Norway</country></aff>
<aff id="aff6"><sup>6</sup><institution>&#x00D8;stfold University College</institution>, <addr-line>Halden</addr-line>, <country>Norway</country></aff>
<aff id="aff7"><sup>7</sup><institution>Norwegian Centre for Science Education, University of Oslo</institution>, <addr-line>Oslo</addr-line>, <country>Norway</country></aff>
<aff id="aff8"><sup>8</sup><institution>Department of Education, UiT The Arctic University of Norway</institution>, <addr-line>Troms&#x00F8;</addr-line>, <country>Norway</country></aff>
<aff id="aff9"><sup>9</sup><institution>Department of Science and Mathematics Education, University of South-Eastern Norway</institution>, <addr-line>Borre</addr-line>, <country>Norway</country></aff>
<aff id="aff10"><sup>10</sup><institution>Department of Teacher Education, Norwegian University of Science and Technology (NTNU)</institution>, <addr-line>Trondheim</addr-line>, <country>Norway</country></aff>
<aff id="aff11"><sup>11</sup><institution>Department of Primary and Secondary Teacher Education, OsloMet University</institution>, <addr-line>Oslo</addr-line>, <country>Norway</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0004">
<p>Edited by: Maria Cutumisu, McGill University, Canada</p>
</fn>
<fn fn-type="edited-by" id="fn0005">
<p>Reviewed by: Luis Alex Alzamora De Los Godos Urcia, Cesar Vallejo University, Peru</p>
<p>Kathrin Kohake, University of M&#x00FC;nster, Germany</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Bas Senden, <email>bassenden@gmail.com</email></corresp>
<fn fn-type="other" id="fn0003"><p><sup>&#x2020;</sup>ORCID: Bas Senden, <ext-link ext-link-type="uri" xlink:href="https://orcid.org/0000-0002-1955-2249">https://orcid.org/0000-0002-1955-2249</ext-link></p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>13</day>
<month>02</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>10</volume>
<elocation-id>1483092</elocation-id>
<history>
<date date-type="received">
<day>19</day>
<month>08</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>27</day>
<month>01</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Senden, Jentsch, Teig, Nilsen, Fauskanger, Fr&#x00E5;g&#x00E5;t, Maugesten, Mork, Mosvold, Nortvedt, Olufsen, Sj&#x00F8;berg, Staberg, Selling, Stovner and &#x00D8;degaard.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Senden, Jentsch, Teig, Nilsen, Fauskanger, Fr&#x00E5;g&#x00E5;t, Maugesten, Mork, Mosvold, Nortvedt, Olufsen, Sj&#x00F8;berg, Staberg, Selling, Stovner and &#x00D8;degaard</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract xml:lang="nb">
<p>The current investigation aims to adapt and validate the Teacher Education and Development Study-Instruct observation instrument for assessing teaching quality in new contexts: Norwegian Grade 6 mathematics and science lessons. More specifically, the article examines content validity and reliability in the new contexts using a multi-methods approach, involving the Delphi technique and generalizability theory. Findings suggest that while the core components of the instrument are relevant in the new contexts, specific adaptations are necessary to capture teaching quality in a more nuanced and meaningful way. Based on the findings, specific adaptions are made to the instrument. Finally, recommendations for developing and using the instrument in the new contexts are provided. The current investigation underscores the importance of contextual sensitivity in the assessment of teaching quality.</p>
</abstract>
<kwd-group>
<kwd>classroom observations</kwd>
<kwd>teaching quality</kwd>
<kwd>instructional quality</kwd>
<kwd>reliability</kwd>
<kwd>validity</kwd>
<kwd>Delphi technique</kwd>
<kwd>generalizability theory</kwd>
<kwd>Norwegian primary education</kwd>
</kwd-group>
<counts>
<fig-count count="3"/>
<table-count count="4"/>
<equation-count count="0"/>
<ref-count count="92"/>
<page-count count="13"/>
<word-count count="11512"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Assessment, Testing and Applied Measurement</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<title>Introduction</title>
<p>Standardized observation instruments are widely recognized and used as effective tools for assessing teaching quality (<xref ref-type="bibr" rid="ref5">Bell et al., 2019</xref>; <xref ref-type="bibr" rid="ref36">Janik et al., 2009</xref>; <xref ref-type="bibr" rid="ref71">Praetorius and Charalambous, 2018</xref>). Developing these instruments is a time-intensive and complex process (<xref ref-type="bibr" rid="ref71">Praetorius and Charalambous, 2018</xref>). Consequently, it is common practice to &#x201C;export&#x201D; observation instruments from one context &#x2014; such as national, subject, or grade-level context &#x2014; to another. For example, the Classroom Assessment Scoring System (CLASS), originally developed at the University of Virginia in the US (<xref ref-type="bibr" rid="ref70">Pianta et al., 2008</xref>), has been widely used in countries across the globe as diverse as Australia (<xref ref-type="bibr" rid="ref87">Thorpe et al., 2023</xref>), Chile (<xref ref-type="bibr" rid="ref51">Leyva et al., 2015</xref>), China (<xref ref-type="bibr" rid="ref35">Hu et al., 2016</xref>), Ecuador (<xref ref-type="bibr" rid="ref2">Araujo et al., 2016</xref>), Finland (<xref ref-type="bibr" rid="ref90">Virtanen et al., 2018</xref>), Germany (<xref ref-type="bibr" rid="ref8">Bihler et al., 2018</xref>), the Netherlands (<xref ref-type="bibr" rid="ref84">Slot et al., 2017</xref>), Norway (<xref ref-type="bibr" rid="ref91">Westerg&#x00E5;rd et al., 2019</xref>), Portugal (<xref ref-type="bibr" rid="ref14">Cadima et al., 2010</xref>), and Singapore (<xref ref-type="bibr" rid="ref65">Ng et al., 2021</xref>). Similarly, the Protocol for Language Arts Teaching Observation (PLATO) was developed at the University of Pennsylvania to assess English language arts instruction (<xref ref-type="bibr" rid="ref32">Grossman et al., 2013</xref>) and later adopted by the Linking Instruction and Student Achievement (LISA) study to assess mathematics teaching in Nordic classrooms (<xref ref-type="bibr" rid="ref45">Klette et al., 2017</xref>).</p>
<p>However, observation instruments generally conceptualize and operationalize a community&#x2019;s view of quality teaching and learning (<xref ref-type="bibr" rid="ref5">Bell et al., 2019</xref>). Given that &#x2018;quality&#x2019; is an inherently vague and ambiguous concept that requires value judgments (<xref ref-type="bibr" rid="ref92">Wittek and Kvernbekk, 2011</xref>; <xref ref-type="bibr" rid="ref7">Berliner, 2005</xref>), it is only natural that most observation instruments are context-specific and context-sensitive. As a result, applying observation instruments in contexts vastly different from those in which they were originally developed can be highly problematic (<xref ref-type="bibr" rid="ref54">Liu et al., 2019</xref>; <xref ref-type="bibr" rid="ref63">Muijs et al., 2018</xref>). For example, although transferring instruments across related subjects is often more feasible (<xref ref-type="bibr" rid="ref21">Cohen et al., 2018</xref>; <xref ref-type="bibr" rid="ref76">Praetorius et al., 2016</xref>), exporting them across national border can be problematic since educational systems might define key-components of high-quality teaching differently (<xref ref-type="bibr" rid="ref7">Berliner, 2005</xref>; <xref ref-type="bibr" rid="ref56">Luoto et al., 2022</xref>; <xref ref-type="bibr" rid="ref63">Muijs et al., 2018</xref>)<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref>. This issue is exacerbated when researchers use existing evidence of an instruments&#x2019; reliability and validity from one context to assert its applicability in another, neglecting that such evidence is not inherent to a specific instrument but rather to specific empirical studies using the instrument in their own unique context (<xref ref-type="bibr" rid="ref1">AERA, APA, NCME, 2014</xref>; <xref ref-type="bibr" rid="ref54">Liu et al., 2019</xref>; <xref ref-type="bibr" rid="ref71">Praetorius and Charalambous, 2018</xref>). Therefore, it is essential to reassess whether an observation instrument can be used in a meaningful and relevant way when applied in new contexts (<xref ref-type="bibr" rid="ref54">Liu et al., 2019</xref>; <xref ref-type="bibr" rid="ref56">Luoto et al., 2022</xref>).</p>
<p>The current study addresses this issue by examining the extent to which the Teacher Education and Development Study&#x2013;Instruct (TEDS-Instruct) standardized observation instrument &#x2014; developed in Germany (<xref ref-type="bibr" rid="ref79">Schlesinger and Jentsch, 2016</xref>; <xref ref-type="bibr" rid="ref80">Schlesinger et al., 2018</xref>; <xref ref-type="bibr" rid="ref41">Kaiser et al., 2017</xref>) &#x2014; can be used to assess teaching quality in new contexts: Norwegian Grade 6 mathematics and science lessons. To this end, the present investigation contains two separate but complementary empirical studies. The first study aims to obtain validity evidence based on the content of the instrument, which is at the core of validation and the beginning of any validation process (<xref ref-type="bibr" rid="ref1">AERA, APA, NCME, 2014</xref>; <xref ref-type="bibr" rid="ref54">Liu et al., 2019</xref>; <xref ref-type="bibr" rid="ref71">Praetorius and Charalambous, 2018</xref>). Taking the stance that teaching quality is context-specific and context-sensitive, we employ the Delphi technique to elicit the opinions of Norwegian mathematics and science education experts regarding the relevance and representativeness of TEDS-Instruct for its new contexts.</p>
<p>The second study aims to obtain evidence on the reliability of TEDS-Instruct within its new contexts by employing generalizability theory (GT) using a sample of Norwegian mathematics and science lessons rated by four trained raters using TEDS-Instruct. GT is a powerful way to examine reliability as it enables researchers to systematically distinguish between multiple sources of error (<xref ref-type="bibr" rid="ref11">Brennan, 1992</xref>; <xref ref-type="bibr" rid="ref71">Praetorius and Charalambous, 2018</xref>). This can be useful for understanding various sources of error and designing more efficient measurement procedures (<xref ref-type="bibr" rid="ref11">Brennan, 1992</xref>). The results from both studies will be used (1) make necessary adaptations to the instrument, (2) provide recommendations for its further development, and (3) provide recommendations for its use in the new contexts.</p>
<sec id="sec2">
<title>Teaching quality</title>
<p>In this study, we define teaching quality as those classroom interactions, both among students and between students and teachers, that provide (sustained) learning opportunities and align with contemporary educational norms and standards (<xref ref-type="bibr" rid="ref6">Berliner, 1987</xref>; <xref ref-type="bibr" rid="ref19">Charalambous et al., 2021</xref>; <xref ref-type="bibr" rid="ref29">Fenstermacher and Richardson, 2005</xref>; <xref ref-type="bibr" rid="ref72">Praetorius and Charalambous, 2023</xref>). Decades of research have established teaching quality as one of the most significant malleable factors in schools that influence student learning outcomes (<xref ref-type="bibr" rid="ref62">Muijs et al., 2014</xref>; <xref ref-type="bibr" rid="ref78">Scheerens et al., 2007</xref>; <xref ref-type="bibr" rid="ref81">Seidel and Shavelson, 2007</xref>). A wide range of theoretical frameworks, models, and (observation) instruments have been developed to understand and assess teaching quality (e.g., <xref ref-type="bibr" rid="ref23">Creemers and Kyriakides, 2008</xref>; <xref ref-type="bibr" rid="ref30">Ferguson and Danielson, 2015</xref>; <xref ref-type="bibr" rid="ref46">Klieme et al., 2009</xref>), each with distinct theoretical underpinning and development processes (<xref ref-type="bibr" rid="ref71">Praetorius and Charalambous, 2018</xref>). Consequently, they reflect a community&#x2019;s view on teaching and learning, resulting in variations in their coverage, structure, and terminology (<xref ref-type="bibr" rid="ref5">Bell et al., 2019</xref>; <xref ref-type="bibr" rid="ref10">Blikstad-Balas et al., 2021</xref>; <xref ref-type="bibr" rid="ref82">Senden et al., 2022</xref>).</p>
<p>In addition, there has been an ongoing debate about the extent to which the measurement of teaching quality should attend to subject-specific teaching practices, which are informed by the demands of teaching in a specific discipline, or to generic teaching practices that adhere to the demands of teaching across disciplines (<xref ref-type="bibr" rid="ref17">Charalambous and Kyriakides, 2017</xref>; <xref ref-type="bibr" rid="ref21">Cohen et al., 2018</xref>). Scholars have argued that including both sets of practices can provide a more complete picture of what happens in the classroom (<xref ref-type="bibr" rid="ref9">Blazar et al., 2017</xref>; <xref ref-type="bibr" rid="ref18">Charalambous and Praetorius, 2018</xref>). This might especially be important since subject-specific teaching practices have been shown to explain a substantial amount of variance in student learning outcomes (<xref ref-type="bibr" rid="ref4">Baumert et al., 2010</xref>; <xref ref-type="bibr" rid="ref17">Charalambous and Kyriakides, 2017</xref>; <xref ref-type="bibr" rid="ref81">Seidel and Shavelson, 2007</xref>). Including both set of practices can be done by simultaneously employing subject-specific and generic observation instruments, as done in the Measures of Effective Teaching (MET) project (<xref ref-type="bibr" rid="ref43">Kane and Cantrell, 2010</xref>). Another option is to employ observation instruments which include both sets of practices, so called hybrid observation instruments (<xref ref-type="bibr" rid="ref18">Charalambous and Praetorius, 2018</xref>; <xref ref-type="bibr" rid="ref82">Senden et al., 2022</xref>).The present study draws on such a hybrid instrument to conceptualize and operationalize teaching quality: the TEDS-Instruct observation instrument (<xref ref-type="bibr" rid="ref80">Schlesinger et al., 2018</xref>).</p>
</sec>
<sec id="sec3">
<title>The TEDS-instruct observation instrument</title>
<p>The observation instrument was developed as part of the German TEDS-Instruct study to examine student learning of mathematics in Grades 7&#x2013;10, independent of the topic discussed in class (<xref ref-type="bibr" rid="ref41">Kaiser et al., 2017</xref>). In response to repeated calls for bringing together generic and subject-specific teaching practices (<xref ref-type="bibr" rid="ref9">Blazar et al., 2017</xref>; <xref ref-type="bibr" rid="ref18">Charalambous and Praetorius, 2018</xref>), the developers of the TEDS-Instruct observation instrument extended the well-established generic framework of the Three Basic Dimensions of teaching quality (<xref ref-type="bibr" rid="ref47">Klieme et al., 2001</xref>; <xref ref-type="bibr" rid="ref46">Klieme et al., 2009</xref>; <xref ref-type="bibr" rid="ref73">Praetorius et al., 2018</xref>; <xref ref-type="bibr" rid="ref74">Praetorius et al., 2020</xref>) with mathematics-specific teaching practices (<xref ref-type="bibr" rid="ref79">Schlesinger and Jentsch, 2016</xref>; <xref ref-type="bibr" rid="ref80">Schlesinger et al., 2018</xref>). To achieve this, a mathematic-specific description of teaching quality was developed based, among others, on a systematic literature review of existing classroom observation instruments (<xref ref-type="bibr" rid="ref79">Schlesinger and Jentsch, 2016</xref>; <xref ref-type="bibr" rid="ref80">Schlesinger et al., 2018</xref>). The observation instrument was initially piloted in the TEDS-Instruct study, and later employed in the TEDS-Validate study (<xref ref-type="bibr" rid="ref41">Kaiser et al., 2017</xref>; <xref ref-type="bibr" rid="ref42">Kaiser and K&#x00F6;nig, 2020</xref>; <xref ref-type="bibr" rid="ref80">Schlesinger et al., 2018</xref>). The obtained data was additionally used to further develop the instrument, which led to the use of a refined four-dimensional conceptualization, including the Three Basic Dimensions &#x2014; <italic>classroom management</italic>, <italic>personal learning support</italic>, and <italic>cognitive activation &#x2014;</italic> and a fourth dimension specific to mathematics education: <italic>educational structuring</italic> (<xref ref-type="bibr" rid="ref37">Jentsch et al., 2020</xref>; <xref ref-type="bibr" rid="ref39">Jentsch et al., 2021b</xref>).</p><list list-type="order">
<list-item>
<p><italic>Classroom management</italic> refers to the strategies and techniques teachers use to organize and manage a complex classroom environment (<xref ref-type="bibr" rid="ref26">Doyle, 1985</xref>; <xref ref-type="bibr" rid="ref28">Emmer and Stough, 2001</xref>). In a well-managed classroom, undesirable behaviors are prevented from happening and desirable behaviors are identified and encouraged, thereby fostering a positive and respectful atmosphere while providing opportunities for instruction and learning (<xref ref-type="bibr" rid="ref26">Doyle, 1985</xref>; <xref ref-type="bibr" rid="ref28">Emmer and Stough, 2001</xref>). Effectively managed classrooms are characterized by the teacher setting clear and consistent rules and expectations for student behavior, providing stable routines and well-structured and planned lessons, and monitoring and redirecting student behavior (<xref ref-type="bibr" rid="ref46">Klieme et al., 2009</xref>; <xref ref-type="bibr" rid="ref49">Kounin, 1970</xref>). Effective classroom management can support students&#x2019; social&#x2013;emotional and academic learning, increase student and teacher retention, prevent burnout and stress symptoms of teachers, and avert serious aggression or behavioral problems among students (<xref ref-type="bibr" rid="ref13">Brouwers and Tomic, 2000</xref>; <xref ref-type="bibr" rid="ref67">Oliver et al., 2011</xref>; <xref ref-type="bibr" rid="ref77">Sabornie and Espelage, 2022</xref>; <xref ref-type="bibr" rid="ref81">Seidel and Shavelson, 2007</xref>).</p>
</list-item>
<list-item>
<p><italic>Personal learning support</italic> focuses on the teacher&#x2019;s capacity to deal with heterogeneity in students&#x2019; abilities and respond to comprehension difficulties of the individual student, and to the collective student body (<xref ref-type="bibr" rid="ref50">Kunter and Voss, 2013</xref>). As such, it includes aspects of instruction related to collaborative learning, inclusion, and differentiation. Personal learning support is assumed to increase the active participation of students, which can lead to more successful learning processes (<xref ref-type="bibr" rid="ref88">Turner et al., 1998</xref>)</p>
</list-item>
<list-item>
<p><italic>Cognitive activation</italic> refers to instructional strategies that facilitate opportunities for students to engage in higher-level cognitive thinking that promotes conceptual understanding (<xref ref-type="bibr" rid="ref46">Klieme et al., 2009</xref>; <xref ref-type="bibr" rid="ref53">Lipowsky et al., 2009</xref>). More specifically, it pertains to instructional situations in which learning is challenging, interactive, and co-constructive, and in which the teacher provides support for students&#x2019; individual construction of knowledge (<xref ref-type="bibr" rid="ref4">Baumert et al., 2010</xref>; <xref ref-type="bibr" rid="ref53">Lipowsky et al., 2009</xref>). Cognitive activation is assumed to increase students&#x2019; knowledge and understanding (<xref ref-type="bibr" rid="ref46">Klieme et al., 2009</xref>).</p>
</list-item>
<list-item>
<p><italic>Educational structuring</italic> includes subject-specific aspects of instruction that originally pertain to the demands of teaching mathematics. Educational structuring addresses the degree to which teachers adapt to students&#x2019; individual cognitive abilities and provide instructional support if needed (so-called scaffolding, <xref ref-type="bibr" rid="ref89">van de Pol et al., 2010</xref>). This adds to the previous quality dimensions by taking further aspects of teaching into account (e.g., structural clarity, explanations, consolidation, see also <xref ref-type="bibr" rid="ref27">Drollinger-Vetter, 2011</xref>). <xref ref-type="bibr" rid="ref44">Kleickmann et al. (2010)</xref> report positive associations between educational structuring (termed cognitive support in their study) and student achievement in science classrooms.</p>
</list-item>
</list>
<p>To accurately capture the four dimensions, each dimension is represented by several high-inferences sub-dimensions. These sub-dimensions are further broken down into specific indicators, which reflect typical observable behaviors associated with each item (see <xref ref-type="supplementary-material" rid="SM1">Appendix Table 1</xref>). It is important to note, that while observers use these indicators to guide their assessment, they actually rate the high-inference sub-dimensions themselves, not the indicators. Ratings are assigned using a four-point scale. Additionally, the instrument is accompanied by a comprehensive rating manual that provides further guidelines for scoring, as well as a detailed description of the high-inference sub-dimensions.</p>
</sec>
</sec>
<sec id="sec4">
<title>Study 1: content-validity</title>
<p>The current study aims to obtain validity evidence based on the content of TEDS-Instruct by eliciting the opinions of subject-matter experts using the Delphi technique. The study is guided by the following research questions (RQs):</p>
<p><italic>RQ1</italic>: To what extent do subject-matter experts agree that TEDS-Instruct can be used for a relevant assessment of teaching quality in Norwegian Grade 6 mathematics and science lessons?</p>
<p><italic>RQ2</italic>: To what extent do subject-matter experts agree that the sub-dimensions assessed through TEDS-Instruct are representative of the overarching dimensions of teaching quality in Norwegian Grade 6 mathematics and science lessons?</p>
<sec id="sec5">
<title>Materials and methods</title>
<sec id="sec6">
<title>Preliminary adaptations</title>
<p>The preliminary phase of this study involved the first four authors forming a focus group to assess the relevance of TEDS-Instruct for the new contexts: Norwegian Grade 6 mathematics and science lessons. The focus group met on five occasions, each lasting between 1.5&#x2013;2&#x202F;h. Initially, the group focused on identifying sub-dimensions and indicators requiring adjustment. It was concluded that the sub-dimensions and indicators of the three basic dimensions &#x2014; classroom management, personal learning support, and cognitive activation &#x2014; required minimal adaptions to be applicable for Grade 6 mathematics and science lessons in Norway. However, the fourth dimension, educational structuring, which was tailored specifically to mathematics, was identified as needing several adaptations to be applicable to science lessons.</p>
<p>Subsequent adaptations to the instrument were based on dialogue and discussion, supplemented by examples from other observation instruments. Adaptations were kept to a minimum and only when there was sufficient agreement among the group. This process led to the first version of the instrument, which was further tested during study 1 and study 2 (see <xref ref-type="fig" rid="fig1">Figure 1</xref>). A complete overview of the preliminary adaptions made to the instrument is available in the <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>. The version of the instrument employed in Study 1 and Study 2 is found in the <xref ref-type="supplementary-material" rid="SM1">Appendix Table 1</xref>.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>An overview of the design of the current investigation.</p>
</caption>
<graphic xlink:href="feduc-10-1483092-g001.tif"/>
</fig>
</sec>
<sec id="sec7">
<title>The Delphi technique and process</title>
<p>The Delphi technique can be broadly defined as &#x201C;a method for structuring a group communication process so that the process is effective in allowing a group of individuals, as a whole, to deal with a complex problem&#x201D; (<xref ref-type="bibr" rid="ref52">Linstone and Turoff, 1975</xref>, p. 3). The Delphi technique has been successfully employed in educational research (<xref ref-type="bibr" rid="ref31">Green, 2014</xref>; <xref ref-type="bibr" rid="ref33">Helmer, 1966</xref>), including for validation purposes (e.g., <xref ref-type="bibr" rid="ref60">Mengual-Andr&#x00E9;s et al., 2016</xref>; <xref ref-type="bibr" rid="ref85">Smith and Simpson, 1995</xref>). In the current study, we did not adopt the traditional approach of generating items. Instead, we ask participants to evaluate a previously developed instrument. This variation of the Delphi technique is also known as the &#x201C;reactive Delphi&#x201D; (<xref ref-type="bibr" rid="ref59">McKenna, 1994</xref>).</p>
<p>It was agreed upon that we would recruit Norwegian subject-matter experts to evaluate the TEDS-Instruct observation instrument. To this end, we employed purposeful sampling to select Norwegian subject-matter experts (<xref ref-type="bibr" rid="ref69">Palinkas et al., 2015</xref>). Our criteria required significant experience in teaching, involvement in teacher education, and/or research expertise in mathematics or science education. Furthermore, we aimed to include experts from various public educational institutions (1) ensure diverse perspectives and maintain anonymity, and (2) minimize the possibility that opinions would be influenced through participant interactions. Finally, we targeted a balanced representation of mathematics and science education experts.</p>
<p>Based on these criteria, we identified 16 experts and invited them by mail to participate and co-author the Delphi study. Of these, 12 experts (75%) confirmed their participation, representing eight governmental institutions. The participants included six experts specialized in mathematics education &#x2014; four full professors, one associate professor, and one Ph.D. candidate &#x2014; and six in science education, comprising two full professors and four associate professors.</p>
<p>Next, we started the iterative two-round Delphi process (see, e.g., <xref ref-type="bibr" rid="ref25">Dalkey and Helmer, 1963</xref>; <xref ref-type="bibr" rid="ref52">Linstone and Turoff, 1975</xref>; <xref ref-type="bibr" rid="ref59">McKenna, 1994</xref>). During this stage, we aimed to obtain data about the <italic>relevance</italic> and <italic>representativeness</italic> of TEDS-Instruct by eliciting the opinions of subject-matter experts. Instead of meeting for face-to-face discussions, participants were provided with extensive online questionnaires to complete individually, ensuring participant anonymity and confidentiality of their responses. These measures were taken to reduce the impact of social-psychological influences, such as reluctance to express divergent opinions, the unwillingness to abandon publicly expressed opinions, or following what seems to be the majority&#x2019;s opinion (<xref ref-type="bibr" rid="ref33">Helmer, 1966</xref>; <xref ref-type="bibr" rid="ref34">Ho and McLeod, 2008</xref>).</p>
<sec id="sec8">
<title>First questionnaire round</title>
<p>In the first questionnaire round, participants received detailed information about TEDS-Instruct and instructions on how to complete the questionnaire. They were then asked to express their opinions on two validity-related topics: (1) <italic>the relevance</italic> and (2) <italic>the representativeness</italic> of the sub-dimensions assessed through the instrument.</p>
</sec>
<sec id="sec9">
<title>Relevance</title>
<p>Mathematics experts were asked to rate the extent to which they believe the sub-dimensions of TEDS-Instruct are relevant for Norwegian Grade 6 mathematics lessons. Similarly, science experts assessed their relevance for science lessons (e.g., &#x201C;<italic>Rate the extent to which you believe time-on-task is relevant to assess in Norwegian Grade 6 science lessons&#x201D;</italic>). The experts were provided with a brief description of each sub-dimension and typical examples of observable behaviors (indicators). They then answered on a four-point Likert scale with strongly irrelevant (coded 1), irrelevant (coded 2), relevant (coded 3), and strongly relevant (coded 4). If experts rated a sub-dimension as strongly irrelevant or irrelevant, they were requested to provide a reason/explanation for their opinion.</p>
</sec>
<sec id="sec10">
<title>Representativeness</title>
<p>All experts were asked to assess how well the sub-dimensions represent the overarching dimensions (e.g., &#x201C;<italic>Rate the extent to which you believe the sub-dimensions adequately represent the dimension of classroom management&#x201D;</italic>). They responded on a four-point Likert scale with <italic>not at all</italic> (coded 1), <italic>not very</italic> (coded 2), <italic>somewhat</italic> (coded 3), and <italic>to a large extent</italic> (coded 4). While providing explanations for their ratings was optional, experts were encouraged to elaborate on their responses.</p>
</sec>
<sec id="sec11">
<title>Iterative Delphi process</title>
<p>The second questionnaire round was developed based on the evaluation and analysis of results obtained from the first round (see <xref ref-type="fig" rid="fig1">Figure 1</xref>). Initially, we assessed the extent of agreement among experts. For this purpose, consensus criteria were established (see <xref ref-type="table" rid="tab1">Table 1</xref>) based on a systematic review of definitions of consensus in Delphi studies (<xref ref-type="bibr" rid="ref1008">Diamond et al., 2014</xref>). The criteria for <italic>positive agreement</italic> were set to a median score&#x202F;&#x2265;&#x202F;<italic>3,</italic> and at least 75% of responses being either 3 or 4. For <italic>negative agreement</italic>, the criteria were a median&#x202F;&#x2264;&#x202F;2 and less than 25% of the responses being either 3 or 4<italic>. Disagreement</italic> was defined as between 25 and 75% of the responses being either 3 or 4.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Consensus criteria to assess the extent of agreement among experts.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Type of consensus</th>
<th align="left" valign="top">Criteria</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Positive agreement</td>
<td align="left" valign="top">Mdn&#x202F;&#x2265;&#x202F;3, frequency [3&#x2013;4]&#x202F;&#x2265;&#x202F;75%</td>
</tr>
<tr>
<td align="left" valign="top">Negative agreement</td>
<td align="left" valign="top">Mdn&#x202F;&#x2264;&#x202F;2, frequency [3&#x2013;4]&#x202F;&#x2264;&#x202F;25%</td>
</tr>
<tr>
<td align="left" valign="top">Disagreement</td>
<td align="left" valign="top">Frequency [3&#x2013;4] 25 to 75%</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Mdn, Median.</p>
</table-wrap-foot>
</table-wrap>
<p>The following approach was agreed upon: If the experts demonstrated positive agreement, the relevant question would not be followed up on in the second questionnaire round. In cases where the experts showed disagreement, the reason or explanation provided by them in their open-ended responses would be summarized and provided back to them in the second round, offering an opportunity to revise their opinions. Additionally, if there were negative agreement on the relevance of a sub-dimension, it would be considered for removal from the instrument. Similarly, negative agreement on the representativeness would prompt us to consider revising the structure of the dimensions, based on the expert&#x2019;s feedback. The revised structure would then be provided back to the experts in the second questionnaire round to assess if the changes enhanced representativeness (see <xref ref-type="fig" rid="fig2">Figure 2</xref>).</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>The iterative Delphi process.</p>
</caption>
<graphic xlink:href="feduc-10-1483092-g002.tif"/>
</fig>
<p>Finally, responses to the open-ended questions would be summarized and analyzed to identify any problems or recurring themes that would need to be addressed in the second questionnaire round.</p>
</sec>
<sec id="sec12">
<title>Second questionnaire round</title>
<p>Based on the evaluation of the results from the first questionnaire, the second round did not include questions on the <italic>representativeness</italic> or <italic>relevance</italic> of the sub-dimension, due to overall positive agreement. Instead, based on insights gained from the analyses of the open-ended responses, the experts were asked to assess the relevance of the indicators (e.g., &#x201C;<italic>Rate the extent to which you believe these indicators are relevant for assessing time on task in Norwegian Grade 6 science lessons&#x201D;</italic>). Expert agreement was again evaluated using the previously stated consensus criteria. Additionally, an open-ended question was included to solicit recommendations for potentially more suitable indicators (e.g., <italic>&#x201C;Are there any alternative indicators that you believe would be more suitable for assessing time on task in Norwegian Grade 6 mathematics lessons?&#x201D;</italic>).</p>
</sec>
</sec>
</sec>
<sec id="sec13">
<title>Results</title>
<p>All 12 experts who initially agreed to participate responded to both questionnaires. Of these, seven experts indicated that they had previously used observation instruments to assess teaching quality.</p>
<sec id="sec14">
<title><italic>RQ1</italic>: relevance of the instrument</title>
<sec id="sec15">
<title>Results round 1</title>
<p>Analyses of the responses from the first questionnaire round revealed a high degree of consensus among experts regarding the relevance of the sub-dimensions assessed through TEDS-Instruct for Norwegian Grade 6 mathematics and science lessons (see <xref ref-type="supplementary-material" rid="SM1">Appendix Table 1</xref>). Based on the consensus criteria, both mathematics and science experts indicated positive agreement for all 21 sub-dimensions. In other words, all sub-dimensions were relevant to strongly relevant for assessing teaching quality in Norwegian Grade 6 mathematics and science lessons. Consequently, we concluded that it was unnecessary to continue enquiring about the relevance of the sub-dimensions in the second questionnaire round.</p>
<p>However, subsequent analysis of the open-ended responses revealed two main themes. Firstly, while there was broad agreement on the relevance of the sub-dimensions, several experts were critical about the relevance of the indicators, which were presented as typical examples of observable behaviors. Secondly, experts offered recommendations for clarifying and developing the instrument, often with a specific focus on refining the indicators. Based on these insights, it was determined that the second questionnaire round should focus on exploring the relevance of the indicators and soliciting recommendations for their improvement.</p>
</sec>
<sec id="sec16">
<title>Results round 2</title>
<p>Results from the second questionnaire showed a large amount of positive agreement regarding the relevance of the indicators. In other words, the majority of indicators were considered relevant to strongly relevant to assess the sub-dimensions, with similar findings in both mathematics and science (see <xref ref-type="supplementary-material" rid="SM1">Appendix Table 1</xref>). However, there was also disagreement regarding the relevance of nine indicators in mathematics and four in science. Notably, following the consensus criteria, no indicators reached negative agreement, indicating experts did not agree on any indicators being irrelevant or strongly irrelevant. Further analysis of the open-ended responses revealed numerous suggestions from experts on how to improve the indicators to better suit the Norwegian Grade 6 mathematics and science context. These suggestions were later analyzed by the focus group to adapt the instrument and provide recommendations for its future development and use.</p>
</sec>
</sec>
<sec id="sec17">
<title><italic>RQ2</italic>: representativeness</title>
<sec id="sec18">
<title>Results round 1</title>
<p>Analysis of the responses from the first questionnaire round (see <xref ref-type="table" rid="tab2">Table 2</xref>) revealed that experts in mathematics education agreed that, across all four dimensions, the sub-dimensions were representative of the overarching dimensions. Similar results were found when analyzing the responses of science experts. However, the median reveals that science experts agreed to a lesser extent than mathematics experts regarding personal learning, support, cognitive activation, and educational structuring. A median score of 3 indicates that these dimensions are &#x201C;somewhat&#x201D; represented by their respective sub-dimensions.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Expert opinions on the extent to which the sub-dimensions of the TEDS-Instruct observation instrument adequately represent the overarching dimension.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Dimension</th>
<th align="center" valign="top" colspan="4">Math</th>
<th align="center" valign="top" colspan="4">Science</th>
</tr>
<tr>
<th/>
<th align="center" valign="top">M (SD)</th>
<th align="center" valign="top">Mdn</th>
<th align="center" valign="top">[3&#x2013;4]</th>
<th align="center" valign="top">Cons.</th>
<th align="center" valign="top">M (SD)</th>
<th align="center" valign="top">Mdn</th>
<th align="center" valign="top">[3&#x2013;4]</th>
<th align="center" valign="top">Cons.</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Classroom management</td>
<td align="center" valign="top">3.8 (0.41)</td>
<td align="center" valign="top">4</td>
<td align="center" valign="top">100%</td>
<td align="center" valign="top">PA</td>
<td align="center" valign="top">3.8 (0.41)</td>
<td align="center" valign="top">4</td>
<td align="center" valign="top">100%</td>
<td align="center" valign="top">PA</td>
</tr>
<tr>
<td align="left" valign="top">Personal learning support</td>
<td align="center" valign="top">3.5 (0.55)</td>
<td align="center" valign="top">3.5</td>
<td align="center" valign="top">100%</td>
<td align="center" valign="top">PA</td>
<td align="center" valign="top">2.8 (0.41)</td>
<td align="center" valign="top">3</td>
<td align="center" valign="top">83%</td>
<td align="center" valign="top">PA</td>
</tr>
<tr>
<td align="left" valign="top">Cognitive activation</td>
<td align="center" valign="top">3.7 (0.52)</td>
<td align="center" valign="top">4</td>
<td align="center" valign="top">100%</td>
<td align="center" valign="top">PA</td>
<td align="center" valign="top">3 (0.00)</td>
<td align="center" valign="top">3</td>
<td align="center" valign="top">100%</td>
<td align="center" valign="top">PA</td>
</tr>
<tr>
<td align="left" valign="top">Educational structuring</td>
<td align="center" valign="top">3.8 (0.41)</td>
<td align="center" valign="top">4</td>
<td align="center" valign="top">100%</td>
<td align="center" valign="top">PA</td>
<td align="center" valign="top">2.8 (0.41)</td>
<td align="center" valign="top">3</td>
<td align="center" valign="top">100%</td>
<td align="center" valign="top">PA</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>N</italic>&#x202F;=&#x202F;12 (<italic>n</italic>&#x202F;=&#x202F;6 for each subject). M, Mean; SD, Standard Deviation; Mdn, Median, [3&#x2013;4]&#x202F;=&#x202F;the percentage of experts who rated the sub-dimensions or indicators as either 3 &#x201C;somewhat&#x201D; or 4 &#x201C;to a large extent.&#x201D; Cons. = consensus according to predetermined criteria. Positive Agreement (PA): Median&#x202F;&#x2265;&#x202F;3, [3&#x2013;4]&#x202F;&#x2265;&#x202F;75%. Negative Agreement (NA): Median&#x202F;&#x2264;&#x202F;2, frequency [3&#x2013;4]&#x202F;&#x2264;&#x202F;25%. Disagreement (D): [3&#x2013;4]&#x202F;&#x2264;&#x202F;75% and&#x202F;&#x2265;&#x202F;25%. Responses to the scale range from 1 to 4, with 1 being &#x201C;not at all&#x201D; and 4 being &#x201C;to a large extent&#x201D;.</p>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
</sec>
<sec id="sec19">
<title>Conclusion and discussion of study 1</title>
<p>According to previously established consensus criteria, experts in this study largely agreed that the sub-dimensions and indicators provided by the TEDS-Instruct observation instrument appear relevant to strongly relevant for assessing teaching quality in Norwegian Grade 6 mathematics and science lessons. However, due to the small sample size, these findings should be interpreted as preliminary, and further validation with a larger expert group, and including teachers, could clarify whether these findings hold more broadly.</p>
<p>One possible hypothesis is, that this finding might be a result of validating the content of the instrument in contexts that are relatively similar, such as Germany and Norway or across mathematics and science. In line with the notion that the conceptualization of teaching quality is shaped by societal and cultural values (<xref ref-type="bibr" rid="ref55">Luoto, 2020</xref>; <xref ref-type="bibr" rid="ref68">Pacheco, 2009</xref>), we might expect less agreement in vastly different contexts. Nevertheless, we argue that the instrument could benefit from further development based on the recommendations provided by the experts through answering the open-ended questions. Incorporating these recommendations could improve the instrument&#x2019;s relevance for assessing teaching quality in the new contexts. To this end, final adaptations and recommendations are provided at the end of this article.</p>
<p>However, there is also a drawback when adapting an existing instrument. While adapting an instrument can improve the relevance of the assessment, it also reduces the potential for cross-contextual and international comparisons. In the field of international comparative research, <xref ref-type="bibr" rid="ref20">Clarke et al. (2012)</xref> refer to this trade-off as the &#x2018;validity-comparability compromise&#x2019;. They argue that &#x201C;pursuing commensurability by imposing general classificatory frameworks can misrepresent valued performances, school knowledge and classroom practice as these are conceived by each community and sacrifice validity in the interest of comparability&#x201D; (<xref ref-type="bibr" rid="ref20">Clarke et al., 2012</xref>, p. 171). While initially discussed as a theoretical concern in cross-cultural research, we argue that this compromise also applies to cross-contextual comparisons, such as across grade levels and subjects. Therefore, the current approach of adapting an instrument to new contexts might lead to a more relevant assessment of teaching quality, but at the expense of comparability. A possible solution would be to develop a set of &#x2018;core practices&#x2019; that are shown to be stable across contexts alongside flexible, context-specific ones.</p>
<p>Moreover, an interesting finding of the current study was that experts predominantly discussed and referred to the indicators when asked about the relevance of the instrument in the new contexts. In responding to the open-ended questions, experts provided a substantial number of recommendations for modifying the indicators. This might be due to the indicators being more concrete and easier to suggest modifications for, or it could be that the indicators are more sensitive to context. Regardless, this provides an indication that while the dimensions and sub-dimensions are applicable to the new contexts, refining the indicators might be beneficial to better reflect the typical observable behaviors of the overarching sub-dimensions in the new contexts.</p>
<p>Finally, experts agreed that the sub-dimensions are representative of the overarching dimensions in both mathematics and science. However, the results also indicate that for personal learning support, cognitive activation, and educational structuring, the sub-dimensions are less representative of the overarching dimension in science than mathematics. A possible cause is the instrument initially being developed for mathematics lessons and thus more geared towards the subject of mathematics. In this case, the instrument might benefit from further development aimed at increasing the representativeness of the sub-dimensions to assess teaching quality in science lessons. This seems especially so for the dimension of educational structuring.</p>
<p>In conclusion, while these findings provide valuable insights, future studies with a larger and more diverse expert sample are needed to confirm and expand upon these results, further informing instrument adaptation for cross-contextual applications.</p>
</sec>
</sec>
<sec id="sec20">
<title>Study 2: generalizability</title>
<p>Together, Study 1 and 2 provide evidence of validity and reliability in relation to using the TEDS-Instruct observation instrument in the context of Norwegian primary school (Grade 6) within science and mathematics. However, while Study 1 evaluated the validity of the <italic>content</italic> of TEDS-Instruct, the current study utilizes this instrument to score video observations to evaluate its reliability. The goal of the present study is to investigate the reliability of the scores on the four dimensions of teaching quality using TEDS-Instruct. More specifically, we employ generalizability theory (GT) to investigate whether the scores adequately reflect variation across lessons and classrooms, whether the rater bias is high, whether the reliability is sufficient, and how all this differs between mathematics and science. We address this by asking the following research questions (RQs):</p>
<p><italic>RQ1</italic>: What is the psychometric quality of the scoring of the four dimensions of teaching quality in terms of:</p><list list-type="alpha-lower">
<list-item>
<p>the share of variance attributed by differences across classrooms, lessons, segments, and raters?</p>
</list-item>
<list-item>
<p>relative (without rater bias) and absolute (with rater bias) reliability and standard errors of measurement?</p>
</list-item>
</list>
<p><italic>RQ2</italic>: How do these variances, relative and absolute reliabilities and standard errors of measurement differ between mathematics and science?</p>
<sec id="sec21">
<title>Materials and methods</title>
<sec id="sec22">
<title>Generalizability theory</title>
<p>An important consideration when assessing teaching quality through classroom observations is how to ensure high score reliability and valid conclusions while allocating limited resources. To examine this, we employ generalizability theory (GT; <xref ref-type="bibr" rid="ref24">Cronbach et al., 1972</xref>; <xref ref-type="bibr" rid="ref12">Brennan, 2001</xref>) to explore to what extent scores obtained with our instrument reflect meaningful variation across classrooms and lessons and sufficient reliability. Based on these findings we provide recommendations for future use of the instrument.</p>
<p>GT was developed specifically for complex measurement situations with many potential sources of variation, such as classrooms, lessons, or raters. This is often encountered in research employing classroom observations, and GT has successfully been used to analyze data for these purposes (e.g., <xref ref-type="bibr" rid="ref16">Casabianca et al., 2013</xref>; <xref ref-type="bibr" rid="ref58">Mashburn et al., 2014</xref>). GT provides a fine-grained picture of reliability, which is done in two steps. In step one, the different sources of variation (such as rater bias or variation across classrooms) are investigated. For instruments used in classroom observations, it is useful to estimate the variance across classrooms, lessons, and raters. In studies examining teaching quality, the variance across classrooms would reflect the degree to which teaching quality differs from one classroom to the next. Large variation could reflect differences in teachers&#x2019; competence, differences in the classroom composition (some classrooms may be more difficult to teach than others), or a combination thereof. Variation across lessons within the same classrooms would reflect that teaching quality changes over time. Large variations across raters would reflect high rater bias and could indicate that more training or a higher number of raters is needed.</p>
<p>Step two in GT is based on the amount of variance identified from the different sources. If, for instance, the variance between raters is found to be large, one could decide to sample additional raters to control for measurement error in a follow-up study. Given the purpose of a study, one can estimate the overall reliability coefficient and standard errors of measurement with and without rater bias (i.e., some raters being stricter than others, such that they systematically assign lower scores throughout the scoring process). Rater bias can be problematic if scores are used for criterion-referenced decisions. However, it may be ignored in situations where only the rank ordering of lessons or classrooms is of interest (e.g., correlational studies). In our study, the relative standard error and reliability coefficient reflect estimates that ignore rater bias, while the absolute error includes rater bias. If there are large differences between raters, the absolute error would thus be much higher than the relative.<xref ref-type="fn" rid="fn0002"><sup>2</sup></xref></p>
</sec>
<sec id="sec23">
<title>Videotaped lessons</title>
<p>The data analyzed in this study are videotaped lessons collected for the Teachers&#x2019; Effect on Student Outcome (TESO) project. Data was obtained from nine schools and 15 classrooms from the Oslo metropolitan area in Norway in the autumn of 2019 and spring of 2020. The 15 classrooms that were sampled, also participated in the large-scale assessment Trends in Mathematics and Science Study (TIMSS) in 2019. In each classroom, 1&#x2013;6 mathematics and/or science lessons were videotaped over the course of several months. The length of the lessons varied between 24 and 106&#x202F;min, and lessons were cut into, on average, 20-min segments for analysis (<xref ref-type="bibr" rid="ref80">Schlesinger et al., 2018</xref>), as recommended in several studies (e.g., <xref ref-type="bibr" rid="ref58">Mashburn et al., 2014</xref>). The sample size per subject can be found in <xref ref-type="table" rid="tab3">Table 3</xref>. Nine teachers taught both mathematics and science.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Sample sizes by subject.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th/>
<th align="center" valign="top">Schools</th>
<th align="center" valign="top">Teachers</th>
<th align="center" valign="top">Classrooms</th>
<th align="center" valign="top">Lessons</th>
<th align="center" valign="top">Segments</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Mathematics</td>
<td align="center" valign="top"><italic>N</italic> =&#x202F;9</td>
<td align="center" valign="top"><italic>N</italic> =&#x202F;15</td>
<td align="center" valign="top"><italic>N</italic> =&#x202F;15</td>
<td align="center" valign="top"><italic>N</italic> =&#x202F;24</td>
<td align="center" valign="top"><italic>N</italic> =&#x202F;66</td>
</tr>
<tr>
<td align="left" valign="top">Science</td>
<td align="center" valign="top"><italic>N</italic> =&#x202F;8</td>
<td align="center" valign="top"><italic>N</italic> =&#x202F;13</td>
<td align="center" valign="top"><italic>N</italic> =&#x202F;13</td>
<td align="center" valign="top"><italic>N</italic> =&#x202F;21</td>
<td align="center" valign="top"><italic>N</italic> =&#x202F;55</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec24">
<title>Measures and procedures</title>
<p>Rating the videos took place during three weeks in the summer of 2023. We applied the TEDS-Instruct instrument comprising 21 items that were scored on a four-point Likert-type rating scale (ranging from 1 through 4). In the first week, raters were trained extensively by studying the rating manual, conducting video observations, and discussing the results with master raters. However, no benchmarks were applied. During the second week, the raters double-scored the first two segments of each lesson (a total of 30 segments in math, and 26 in science). After double-scoring a lesson, the raters discussed their ratings among each other, but they did not have to agree on a score. In the third week, the ratings used in the current study were obtained by having each individual lesson scored by a single rater. A total of four raters scored the videos. All raters were student teachers in their third year or later within different science, technology, engineering, and mathematics (STEM) programs. Scores were assigned using Interact software (<xref ref-type="bibr" rid="ref57">Mangold, 2023</xref>).</p>
</sec>
<sec id="sec25">
<title>Statistical analysis</title>
<p>In a first step, we calculated the mean scores of each dimension (i.e., classroom management, personal learning support, cognitive activation, educational structuring). Second, we estimated descriptives and correlations on the dimension level (see <xref ref-type="supplementary-material" rid="SM1">Appendix Table 2</xref>). We then applied GT (<xref ref-type="bibr" rid="ref24">Cronbach et al., 1972</xref>; <xref ref-type="bibr" rid="ref12">Brennan, 2001</xref>) to estimate measurement error and reliability in our study. GT makes use of the linear mixed model to estimate variance components for each measurement facet of interest (<xref ref-type="bibr" rid="ref12">Brennan, 2001</xref>). We estimated variance components for classrooms, lessons (i.e., the objects of measurement), lesson segments, and raters using the REML estimator from the free R package lme4 (<xref ref-type="bibr" rid="ref3">Bates et al., 2015</xref>). Separate GT analyses were performed for the two subjects and all teaching quality dimensions. They were compared across subject domains regarding absolute and relative error of measurement as well as reliability. The reliability coefficients were calculated correspondingly by taking either true variance over the sum of true variance and <italic>relative</italic> error variance, or true variance over the sum of true variance and <italic>absolute</italic> error variance. Reliability coefficients were interpreted similar as classical reliability coefficients: &#x2264; 0.5: low reliability, 0.5&#x2013;0.70: moderate reliability, 0.70&#x2013;0.9: good reliability, &#x2265; 0.90: excellent reliability (<xref ref-type="bibr" rid="ref22">Cortina, 1993</xref>; <xref ref-type="bibr" rid="ref48">Koo and Li, 2016</xref>).</p>
</sec>
</sec>
<sec id="sec26">
<title>Results</title>
<p>In the following, the results are presented in the order of the research questions.</p>
<sec id="sec27">
<title><italic>RQ1</italic>(a) variance</title>
<p><xref ref-type="fig" rid="fig3">Figure 3</xref> presents the results of a variance decomposition for mathematics and science lessons. In mathematics classrooms, only small portions of variance are due to differences between classrooms. The share of total variance across classrooms is below 7 %, except for ES (which is about 33%). This suggests that teaching quality regarding classroom management, personal learning support, and cognitive activation is similar across the observed mathematics classrooms. However, lessons within classrooms in mathematics contribute largely to the total variability. Except for cognitive activation, the variances between lessons are between 24 and 33 percent. Further sources of variance include segments and raters. We see that variability in scores attributed to segments was only substantial for classroom management (23%) and cognitive activation (10%), which suggests that personal learning support and educational structuring scores do not vary much during a lesson. Finally, we observe a large rater (main) effect for cognitive activation, which suggests that raters systematically vary in their strictness when scoring this dimension.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Variance components (in percentage) for the four dimensions of teaching quality in mathematics and science. CM, classroom management; PLS, personal learning support; CA, cognitive activation; ES, educational structuring.</p>
</caption>
<graphic xlink:href="feduc-10-1483092-g003.tif"/>
</fig>
<p>For science lessons and classrooms, the results point in a different direction. The share of total variance that is due to variation across classrooms is substantial for all dimensions (up to 57% for classroom management), but no variation between lessons within classrooms was observed. This suggests that raters assign similar scores to a classroom independent of the specific lessons that they scored. In other words, this means that scores are stable across lessons within a science classroom. What is more, little variation was found for segments within lessons (except personal learning support, approx. 18%). This indicates that teaching quality is stable during a lesson. With regards to rater bias, only a small portion of variance was due to differences between raters in terms of classroom management and personal learning support, but the share was large for the other two dimensions. This might show that raters had more difficulties in assigning scores for cognitive activation and educational structuring.</p>
</sec>
<sec id="sec28">
<title><italic>RQ1</italic>(b) standard errors and reliability</title>
<p><xref ref-type="table" rid="tab4">Table 4</xref> provides combined relative and absolute measurement error and reliabilities coefficients. The relative measures do not include rater bias, while the absolute does. We see that similar relative and absolute estimates are found for cases in which the portion of variance attributed to raters is low. For mathematics, all reliability coefficients can be considered sufficient except for cognitive activation. Which had low reliability.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>Standard errors and reliability.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th rowspan="2"/>
<th align="center" valign="top" colspan="4">Mathematics</th>
<th align="center" valign="top" colspan="4">Science</th>
</tr>
<tr>
<th align="right" valign="top">CM</th>
<th align="right" valign="top">PLS</th>
<th align="right" valign="top">CA</th>
<th align="right" valign="top">ES</th>
<th align="right" valign="top">CM</th>
<th align="right" valign="top">PLS</th>
<th align="right" valign="top">CA</th>
<th align="right" valign="top">ES</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top" colspan="9">Standard error</td>
</tr>
<tr>
<td align="left" valign="top">Relative</td>
<td align="center" valign="top">0.43</td>
<td align="center" valign="top">0.38</td>
<td align="center" valign="top">0.92</td>
<td align="center" valign="top">0.50</td>
<td align="center" valign="top">0.13</td>
<td align="center" valign="top">0.64</td>
<td align="center" valign="top">0.69</td>
<td align="center" valign="top">0.73</td>
</tr>
<tr>
<td align="left" valign="top">Absolute</td>
<td align="center" valign="top">0.43</td>
<td align="center" valign="top">0.38</td>
<td align="center" valign="top">1.24</td>
<td align="center" valign="top">0.63</td>
<td align="center" valign="top">0.18</td>
<td align="center" valign="top">0.70</td>
<td align="center" valign="top">1.40</td>
<td align="center" valign="top">1.16</td>
</tr>
<tr>
<td align="left" valign="top" colspan="9">Reliability</td>
</tr>
<tr>
<td align="left" valign="top">Relative</td>
<td align="center" valign="top">0.64</td>
<td align="center" valign="top">0.80</td>
<td align="center" valign="top">0.49</td>
<td align="center" valign="top">0.94</td>
<td align="center" valign="top">0.96</td>
<td align="center" valign="top">0.84</td>
<td align="center" valign="top">0.91</td>
<td align="center" valign="top">0.83</td>
</tr>
<tr>
<td align="left" valign="top">Absolute</td>
<td align="center" valign="top">0.64</td>
<td align="center" valign="top">0.80</td>
<td align="center" valign="top">0.35</td>
<td align="center" valign="top">0.91</td>
<td align="center" valign="top">0.94</td>
<td align="center" valign="top">0.81</td>
<td align="center" valign="top">0.71</td>
<td align="center" valign="top">0.65</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>CM, classroom management; PLS, personal learning support; CA, cognitive activation; ES, educational structuring.</p>
</table-wrap-foot>
</table-wrap>
<p>For science, all dimensions reached good or even excellent reliabilities with regard to the relative estimates. The absolute estimates are sufficient but substantially lower and could be improved by, for example, having multiple raters rate the same lesson. This is especially the case for cognitive activation in both subjects, but also for educational structuring in science.</p>
</sec>
<sec id="sec29">
<title><italic>RQ2</italic>: comparisons between mathematics and science</title>
<p>Comparing the overall reliability between mathematics and science encompasses taking all the evidence from the variance components, standard errors, and reliability coefficients into account. The variance attributed to raters in science was highest for educational structuring and cognitive activation which resulted in lower absolute reliabilities. Cognitive activation stands out as having a high amount of variance attributed to raters in both subjects. However, while the absolute reliability for this dimension was acceptable for science (0.71), it was very low for mathematics (0.35), whereas educational structuring had high absolute reliability in mathematics.</p>
<p>Classroom management differed substantially between the two subjects. In mathematics, most of the variance in the scores were due to variance over lessons and segments, while most of the variance in scores in science were due to differences between classrooms. In spite of no rater bias in mathematics, the reliabilities are smaller, and standard errors larger, than in science for this dimension of teaching quality. The similarities and differences pointed out, will be discussed in light of the Norwegian context and previous research in the next section.</p>
</sec>
</sec>
<sec id="sec30">
<title>Conclusion and discussion of study 2</title>
<p>One of the main findings in study 2, is that a very small share of variance in mathematics could be attributed to variation between classrooms. Rather, the larger portion of variance was found between lessons. This finding is different from what was found in German studies (e.g., <xref ref-type="bibr" rid="ref38">Jentsch et al., 2021a</xref>) and also different from the findings in science found in the current study. This could imply that differences between mathematics teachers&#x2019; competence is smaller than that between science teachers. Indeed, findings from the larger, representative sample of TIMSS 2019, show that mathematics teachers have better qualifications, more specialization, and participated three times more often in professional development activities (<xref ref-type="bibr" rid="ref64">Mullis et al., 2020</xref>). Moreover, these qualifications varied more between science teachers than mathematics teachers. Even though our sample is a sub-sample of TIMSS 2019, this could be a plausible explanation, and could also be observed <italic>in situ</italic> by those filming the lessons. In fact, the ratings of the mean overall teaching quality in mathematics for our sample was higher than those in science (see <xref ref-type="supplementary-material" rid="SM1">Appendix Table 2</xref>).</p>
<p>However, why the ratings mostly varied across lessons and segments in mathematics, in contrast to small variations across lessons and segments in science, is a more complicated question. It could be that high quality teaching is more sensitive to the composition of the classroom. In other words, it could be more difficult to maintain quality teaching over time and during a lesson in classrooms with many low SES students and students who do not speak Norwegian. Classroom management was the dimension that varied the most across lessons and segments in mathematics, and this dimension would probably also be most sensitive to the classroom composition.</p>
<p>It could, however, also be that mathematics lessons in the sample vary a lot in quality, and that more lessons are needed to capture more robust scores. In general, if scores mostly vary (much) between lessons within a classroom, this suggests that they are useful for giving feedback to teachers, for instance, but less so for long-term decisions, such as predicting student learning outcomes. This is because variance between classes or teachers most often is utilized when analyzing the effect of teaching quality on students&#x2019; learning outcomes. For science, a large proportion of the variance could be attributed to classrooms, whereas there was no variance that could be attributed to lessons. These findings suggest that the science scores could be useful for long-term decisions, but not necessarily to give feedback to teachers regarding a single lesson.</p>
<p>Finally, we experienced higher amounts of rater bias in science than in mathematics, in the sense that some raters were stricter than others. The rater bias was high in educational structuring in science, and in cognitive activation in both subjects. There were further relatively large portions of residual variance. Although this does not necessarily affect score reliability, more research is needed (1) find ways to train raters most efficiently, and (2) look at other variables or measurement facets that might affect the results, particularly regarding cognitive activation in mathematics classrooms. Overall, the reliability of the instrument is sufficient for scoring in the context of Norwegian 6<sup>th</sup> grade classrooms in science and mathematics, although cognitive activation in mathematics requires further work.</p>
</sec>
</sec>
<sec id="sec31">
<title>Final adaptations and recommendations</title>
<sec id="sec32">
<title>Adaptions made to the instrument</title>
<p>In the present investigation, both studies provided information on how to adapt and use the instrument in new contexts. In study 1, subject-matter experts provided recommendations for adapting the instrument further to be more relevant in its new contexts. Developing the instrument based on these recommendations could potentially improve the extent to which the instrument assesses teaching quality in a meaningful and relevant way in the new contexts. In addition, study 2 examined the functioning of the instrument in the new contexts. Recommendations based on these findings could potentially inform future applications of the instrument in research and practice.</p>
<p>To deal with this information, the first four authors of this paper came together again as a focus group to discuss the recommendations. The focus group met on six occasions for 1.5&#x2013;2. hours. Based on information received through the open-ended questions in the Delphi study and information from the generalizability analysis, the focus group made adaptations to the instruments and its manual. In addition, the focus group concluded on recommendations for using and further developing TEDS-Instruct in Norwegian Grade 6 mathematics and science lessons. A full overview of the adaptations made to the instrument and the updated version of the instrument can be found in the <xref ref-type="supplementary-material" rid="SM1">Supplementary materials</xref>. Future research will have to test the updated instrument and possibly develop it further. To this end, collecting additional validity evidence will be necessary to form a more complete validity argument.</p>
</sec>
<sec id="sec33">
<title>Recommendations for further development</title>
<p>The current instrument assesses personal learning support, which is considered a component of a supportive classroom climate. Teaching practices related to emotional or social support are not explicitly included in the assessment of personal learning support. However, these forms of support are considered a key element in Norwegian classrooms. For further development, we would therefore recommend explicitly including such teaching practices in the instrument.</p>
<p>Moreover, the instrument was originally developed to assess teaching quality in mathematics lessons. Even though the current study has made several adaptions in order to use the instrument in science lessons, results from the Delphi study hinted at the need to include a more thorough assessment of subject-specific teaching practices in science. The generalizability study further confirmed these findings by providing evidence that educational structuring in science, in contrast to mathematics, was difficult to rate consistently across raters. Developing the instrument to be more directed towards subject-specific science teaching &#x2014; for example, by including measures assessing the &#x201C;nature of science&#x201D; or &#x201C;inquiry&#x201D; &#x2014; could provide a more meaningful and relevant assessment of teaching quality in science lessons and more reliable scores.</p>
</sec>
<sec id="sec34">
<title>Recommendations for using the instrument</title>
<p>Since subject-specific teaching practices in science might still be underrepresented in the current instrument, it could be recommended to use the instrument together with an instrument developed for measuring subject-specific science practices (e.g., inquiry) such as ISIOP (<xref ref-type="bibr" rid="ref61">Minner and DeLisi, 2012</xref>). Moreover, findings from the generalizability study indicate that it might be necessary to rate more lessons in mathematics than in science to obtain reliable scores. However, these findings should be confirmed by subsequent studies. In line with <xref ref-type="bibr" rid="ref75">Praetorius et al. (2014)</xref>, our results confirm that having multiple raters score cognitive activation is necessary to obtain reliable scores. Additionally, more focus on rating cognitive activation during the training might be beneficial. For example, by employing additional quality criteria, such as comparing scores in cognitive activation to a master rater score.</p>
</sec>
</sec>
<sec id="sec35">
<title>Strengths and limitations</title>
<p>A major strength of the current investigation is that it considers different sources of evidence to examine the validity and reliability of TEDS-Instruct in the new contexts.</p>
<p>The Delphi panel in study 1 consisted of a relatively small sample of experts, comprising six mathematics and six science education experts. The limited sample size may have contributed to the large amount of agreement among experts due to accidentally sampling experts with similar value systems. On the other hand, the panel represented a range of public educational institutions from across the country, ensuring a variety of perspectives. Consequently, future research taking a similar approach could increase the sample while aiming to keep a diverse panel. Additionally, the Delphi panel consisted solely of scholars, and future research could benefit from including teachers&#x2019; as experts. However, this might also lead to more disagreement as teachers could lack the theoretical knowledge on teaching quality. Despite these potential challenges, the Delphi technique can be a powerful approach to simultaneously validate and adapt an observation instrument to new contexts. In doing so, the adapted instrument better represents the communities view on teaching and learning.</p>
<p>In study 2, a meaningful inter-rater reliability coefficient could not be calculated or reported. This was due to the raters being given the opportunity to discuss their scores with each other during the double scoring process in week 2, which compromised the independence of their ratings. Moreover, the small sample utilized in the generalizability study limited the statistical power. These constraints limited our ability to estimate models with more variables or interaction terms which likely would have provided more information. For example, including the time of the day the lessons were recorded as an extra variable in the GT analysis could prove insightful. Additionally, while the study focused on the four overarching dimensions of teaching quality, it is expected that variation differs between sub-dimensions as well. Furthermore, the lesson content is not standardized, and the sample of classrooms is expected to vary in several regards, e.g., background characteristics of the students, teacher qualifications. These contextual factors are not taken into account in the generalizability analysis but can have implications for the reliability estimates (<xref ref-type="bibr" rid="ref1009">White et al., 2022</xref>). Despite these limitations, we believe the results of the generalizability study to be valuable for gauging reliability and providing recommendations for the future use of the instrument in its new context. Findings from the study can increase the reliability of ratings in future studies, while decreasing costs associated with using observation instruments.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec36">
<title>Data availability statement</title>
<p>The datasets presented in this article are not readily available because they contain confidential and sensitive information, which cannot be shared due to privacy and ethical restrictions. Requests to access the datasets should be directed to Bas Senden at <email>bassenden@gmail.com</email>.</p>
</sec>
<sec sec-type="author-contributions" id="sec37">
<title>Author contributions</title>
<p>BS: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Project administration, Validation, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. AJ: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Validation, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. NT: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Project administration, Supervision, Validation, Writing &#x2013; review &#x0026; editing. TN: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Supervision, Validation, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. JF: Writing &#x2013; review &#x0026; editing. TF: Writing &#x2013; review &#x0026; editing. MM: Writing &#x2013; review &#x0026; editing. SM: Writing &#x2013; review &#x0026; editing. RM: Writing &#x2013; review &#x0026; editing. GN: Writing &#x2013; review &#x0026; editing. MO: Writing &#x2013; review &#x0026; editing. MS: Writing &#x2013; review &#x0026; editing. RLS: Writing &#x2013; review &#x0026; editing. AS: Writing &#x2013; review &#x0026; editing. RBS: Writing &#x2013; review &#x0026; editing. M&#x00D8;: Writing &#x2013; review &#x0026; editing.</p>
</sec>
<sec sec-type="funding-information" id="sec38">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work was supported by the FINNUT Researcher Project awarded by the Research Council of Norway under Grant number 283587 and partially supported by the Research Council of Norway through its Centers of Excellence scheme under Grant number 331640.</p>
</sec>
<ack>
<p>We would like to express our utmost gratitude to Professor Gabriele Kaiser and Professor Johannes K&#x00F6;nig for their foundational work in developing the TEDS-Instruct observation instrument and their generosity in granting us access to the instrument. We would also like to thank the four raters&#x2014;Helene, Neha, Sandra, and Jonathan&#x2014;for their enthusiasm and devotion in rating the videos. Finally, our thanks goes to Andreas Petersen for testing, and providing feedback on, the initial Delphi questionnaires.</p>
</ack>
<sec sec-type="COI-statement" id="sec39">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="sec40">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec41">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/feduc.2025.1483092/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/feduc.2025.1483092/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_1.DOCX" id="SM2" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn id="fn0001"><p><sup>1</sup>A notable exception is the International System for Teacher Observation and Feedback (ISFOT), which was developed to ensure cross-cultural relevance (<xref ref-type="bibr" rid="ref63">Muijs et al., 2018</xref>; <xref ref-type="bibr" rid="ref86">Teddlie et al., 2006</xref>).</p></fn>
<fn id="fn0002"><p><sup>2</sup>In more precise and statistical terms, the <italic>relative</italic> error (<xref ref-type="bibr" rid="ref83">Shavelson and Webb, 1991</xref>) reflects the extent to which rank ordering of lessons is distorted (i.e., their relative standing). The <italic>absolute</italic> error is estimated if scores are compared to a certain cutoff value.</p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="book"><person-group person-group-type="author"><collab id="coll1">AERA, APA, NCME</collab></person-group> (<year>2014</year>). <source>Standards for educational and psychological testing</source>. <publisher-loc>Washington, DC</publisher-loc>: <publisher-name>American Educational Research Association</publisher-name>.</citation></ref>
<ref id="ref2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Araujo</surname> <given-names>M. C.</given-names></name> <name><surname>Carneiro</surname> <given-names>P.</given-names></name> <name><surname>Cruz-Aguayo</surname> <given-names>Y.</given-names></name> <name><surname>Schady</surname> <given-names>N.</given-names></name></person-group> (<year>2016</year>). <article-title>Teacher quality and learning outcomes in kindergarten</article-title>. <source>Q. J. Econ.</source> <volume>131</volume>, <fpage>1415</fpage>&#x2013;<lpage>1453</lpage>. doi: <pub-id pub-id-type="doi">10.1093/qje/qjw016</pub-id></citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bates</surname> <given-names>D.</given-names></name> <name><surname>M&#x00E4;chler</surname> <given-names>M.</given-names></name> <name><surname>Bolker</surname> <given-names>B.</given-names></name> <name><surname>Walker</surname> <given-names>S.</given-names></name></person-group> (<year>2015</year>). <article-title>Fitting linear mixed-effects models using lme4</article-title>. <source>J. Stat. Softw.</source> <volume>67</volume>, <fpage>1</fpage>&#x2013;<lpage>48</lpage>. doi: <pub-id pub-id-type="doi">10.18637/jss.v067.i01</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Baumert</surname> <given-names>J.</given-names></name> <name><surname>Kunter</surname> <given-names>M.</given-names></name> <name><surname>Blum</surname> <given-names>W.</given-names></name> <name><surname>Brunner</surname> <given-names>M.</given-names></name> <name><surname>Voss</surname> <given-names>T.</given-names></name> <name><surname>Jordan</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2010</year>). <article-title>Teachers' mathematical knowledge, cognitive activation in the classroom, and student progress</article-title>. <source>Am. Educ. Res. J.</source> <volume>47</volume>, <fpage>133</fpage>&#x2013;<lpage>180</lpage>. doi: <pub-id pub-id-type="doi">10.3102/0002831209345157</pub-id></citation></ref>
<ref id="ref5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bell</surname> <given-names>C. A.</given-names></name> <name><surname>Dobbelaer</surname> <given-names>M. J.</given-names></name> <name><surname>Klette</surname> <given-names>K.</given-names></name> <name><surname>Visscher</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>Qualities of classroom observation systems</article-title>. <source>Sch. Eff. Sch. Improv.</source> <volume>30</volume>, <fpage>3</fpage>&#x2013;<lpage>29</lpage>. doi: <pub-id pub-id-type="doi">10.1080/09243453.2018.1539014</pub-id></citation></ref>
<ref id="ref6"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Berliner</surname> <given-names>D. C.</given-names></name></person-group> (<year>1987</year>). &#x201C;<article-title>Simple views of effective teaching and a simple theory of classroom instruction</article-title>&#x201D; in <source>Talks to teachers</source>. eds. <person-group person-group-type="editor"><name><surname>Berliner</surname> <given-names>D. C.</given-names></name> <name><surname>Rosenshine</surname> <given-names>B.</given-names></name></person-group> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Random House</publisher-name>), <fpage>93</fpage>&#x2013;<lpage>110</lpage>.</citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Berliner</surname> <given-names>D. C.</given-names></name></person-group> (<year>2005</year>). <article-title>The near impossibility of testing for teacher quality</article-title>. <source>J. Teach. Educ.</source> <volume>56</volume>, <fpage>205</fpage>&#x2013;<lpage>213</lpage>. doi: <pub-id pub-id-type="doi">10.1177/0022487105275904</pub-id></citation></ref>
<ref id="ref8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bihler</surname> <given-names>L.-M.</given-names></name> <name><surname>Agache</surname> <given-names>A.</given-names></name> <name><surname>Kohl</surname> <given-names>K.</given-names></name> <name><surname>Willard</surname> <given-names>J. A.</given-names></name> <name><surname>Leyendecker</surname> <given-names>B.</given-names></name></person-group> (<year>2018</year>). <article-title>Factor analysis of the classroom assessment scoring system replicates the three domain structure and reveals no support for the bifactor model in German preschools</article-title>. <source>Front. Psychol.</source> <volume>9</volume>:<fpage>1232</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpsyg.2018.01232</pub-id>, PMID: <pub-id pub-id-type="pmid">30072940</pub-id></citation></ref>
<ref id="ref9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Blazar</surname> <given-names>D.</given-names></name> <name><surname>Braslow</surname> <given-names>D.</given-names></name> <name><surname>Charalambous</surname> <given-names>C.</given-names></name> <name><surname>Hill</surname> <given-names>H.</given-names></name></person-group> (<year>2017</year>). <article-title>Attending to general and mathematics-specific dimensions of teaching: exploring factors across two observation instruments</article-title>. <source>Educ. Assess.</source> <volume>22</volume>, <fpage>71</fpage>&#x2013;<lpage>94</lpage>. doi: <pub-id pub-id-type="doi">10.1080/10627197.2017.1309274</pub-id></citation></ref>
<ref id="ref10"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Blikstad-Balas</surname> <given-names>M.</given-names></name> <name><surname>Tengberg</surname> <given-names>M.</given-names></name> <name><surname>Klette</surname> <given-names>K.</given-names></name></person-group> (<year>2021</year>). &#x201C;<article-title>Why &#x2013; and how &#x2013; should we measure instructional quality?</article-title>&#x201D; in <source>Ways of analyzing teaching quality</source>. eds. <person-group person-group-type="editor"><name><surname>Blikstad-Balas</surname> <given-names>M.</given-names></name> <name><surname>Klette</surname> <given-names>K.</given-names></name> <name><surname>Tengberg</surname> <given-names>M.</given-names></name></person-group> (<publisher-loc>Oslo</publisher-loc>: <publisher-name>Scandinavian University Press</publisher-name>), <fpage>9</fpage>&#x2013;<lpage>20</lpage>.</citation></ref>
<ref id="ref11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brennan</surname> <given-names>R. L.</given-names></name></person-group> (<year>1992</year>). <article-title>Generalizability theory</article-title>. <source>Educ. Meas. Issues Pract.</source> <volume>11</volume>, <fpage>27</fpage>&#x2013;<lpage>34</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1745-3992.1992.tb00260.x</pub-id></citation></ref>
<ref id="ref12"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Brennan</surname> <given-names>R. L.</given-names></name></person-group> (<year>2001</year>). <source>Generalizability theory</source>. <publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer</publisher-name>.</citation></ref>
<ref id="ref13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brouwers</surname> <given-names>A.</given-names></name> <name><surname>Tomic</surname> <given-names>W.</given-names></name></person-group> (<year>2000</year>). <article-title>A longitudinal study of teacher burnout and perceived self-efficacy in classroom management</article-title>. <source>Teach. Teach. Educ.</source> <volume>16</volume>, <fpage>239</fpage>&#x2013;<lpage>253</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0742-051X(99)00057-8</pub-id></citation></ref>
<ref id="ref14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cadima</surname> <given-names>J.</given-names></name> <name><surname>Leal</surname> <given-names>T.</given-names></name> <name><surname>Burchinal</surname> <given-names>M.</given-names></name></person-group> (<year>2010</year>). <article-title>The quality of teacher&#x2013;student interactions: associations with first graders' academic and behavioral outcomes</article-title>. <source>J. Sch. Psychol.</source> <volume>48</volume>, <fpage>457</fpage>&#x2013;<lpage>482</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jsp.2010.09.001</pub-id>, PMID: <pub-id pub-id-type="pmid">21094394</pub-id></citation></ref>
<ref id="ref16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Casabianca</surname> <given-names>J. M.</given-names></name> <name><surname>McCaffrey</surname> <given-names>D. F.</given-names></name> <name><surname>Gitomer</surname> <given-names>D.</given-names></name> <name><surname>Bell</surname> <given-names>C.</given-names></name> <name><surname>Hamre</surname> <given-names>B. K.</given-names></name> <name><surname>Pianta</surname> <given-names>R. C.</given-names></name></person-group> (<year>2013</year>). <article-title>Effect of observation mode on measures of secondary mathematics teaching</article-title>. <source>Educ. Psychol. Meas.</source> <volume>73</volume>, <fpage>757</fpage>&#x2013;<lpage>783</lpage>. doi: <pub-id pub-id-type="doi">10.1177/0013164413486987</pub-id></citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Charalambous</surname> <given-names>C. Y.</given-names></name> <name><surname>Kyriakides</surname> <given-names>E.</given-names></name></person-group> (<year>2017</year>). <article-title>Working at the nexus of generic and content-specific teaching practices: an exploratory study based on TIMSS secondary analyses</article-title>. <source>Elem. Sch. J.</source> <volume>117</volume>, <fpage>423</fpage>&#x2013;<lpage>454</lpage>. doi: <pub-id pub-id-type="doi">10.1086/690221</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Charalambous</surname> <given-names>C. Y.</given-names></name> <name><surname>Praetorius</surname> <given-names>A.-K.</given-names></name></person-group> (<year>2018</year>). <article-title>Studying mathematics instruction through different lenses: setting the ground for understanding instructional quality more comprehensively</article-title>. <source>ZDM</source> <volume>50</volume>, <fpage>355</fpage>&#x2013;<lpage>366</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11858-018-0914-8</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Charalambous</surname> <given-names>C. Y.</given-names></name> <name><surname>Praetorius</surname> <given-names>A.-K.</given-names></name> <name><surname>Sammons</surname> <given-names>P.</given-names></name> <name><surname>Walkowiak</surname> <given-names>T.</given-names></name> <name><surname>Jentsch</surname> <given-names>A.</given-names></name> <name><surname>Kyriakides</surname> <given-names>L.</given-names></name></person-group> (<year>2021</year>). <article-title>Working more collaboratively to better understand teaching and its quality: challenges faced and possible solutions</article-title>. <source>Stud. Educ. Eval.</source> <volume>71</volume>:<fpage>101092</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.stueduc.2021.101092</pub-id>, PMID: <pub-id pub-id-type="pmid">39863507</pub-id></citation></ref>
<ref id="ref20"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Clarke</surname> <given-names>D.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name> <name><surname>Xu</surname> <given-names>L.</given-names></name> <name><surname>Aizikovitsh-Udi</surname> <given-names>E.</given-names></name> <name><surname>Cao</surname> <given-names>Y.</given-names></name></person-group> (<year>2012</year>). &#x201C;<article-title>International comparisons of mathematics classrooms and curricula: the validity-comparability compromise</article-title>&#x201D; in <source>PME 36: Opportunities to learn in mathematics education: Proceedings of the 36th conference of the International Group for the Psychology of mathematics education 2012</source> (<publisher-loc>Taipei, Taiwan</publisher-loc>: <publisher-name>National Taiwan Normal University</publisher-name>), <fpage>171</fpage>&#x2013;<lpage>178</lpage>.</citation></ref>
<ref id="ref21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cohen</surname> <given-names>J.</given-names></name> <name><surname>Ruzek</surname> <given-names>E.</given-names></name> <name><surname>Sandilos</surname> <given-names>L.</given-names></name></person-group> (<year>2018</year>). <article-title>Does teaching quality cross subjects? Exploring consistency in elementary teacher practice across subjects</article-title>. <source>AERA Open</source> <volume>4</volume>, <fpage>1</fpage>&#x2013;<lpage>16</lpage>. doi: <pub-id pub-id-type="doi">10.1177/2332858418794492</pub-id></citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cortina</surname> <given-names>J. M.</given-names></name></person-group> (<year>1993</year>). <article-title>What is coefficient alpha? An examination of theory and applications</article-title>. <source>J. Appl. Psychol.</source> <volume>78</volume>, <fpage>98</fpage>&#x2013;<lpage>104</lpage>. doi: <pub-id pub-id-type="doi">10.1037/0021-9010.78.1.98</pub-id></citation></ref>
<ref id="ref23"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Creemers</surname> <given-names>B. P. M.</given-names></name> <name><surname>Kyriakides</surname> <given-names>L.</given-names></name></person-group> (<year>2008</year>). <source>The dynamics of educational effectiveness: A contribution to policy, practice and theory in contemporary schools</source>. <publisher-loc>London</publisher-loc>: <publisher-name>Routledge</publisher-name>.</citation></ref>
<ref id="ref24"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Cronbach</surname> <given-names>L. J.</given-names></name> <name><surname>Glaser</surname> <given-names>G. C.</given-names></name> <name><surname>Nanda</surname> <given-names>H.</given-names></name> <name><surname>Rajaratnam</surname> <given-names>N.</given-names></name></person-group> (<year>1972</year>). <source>The dependability of behavioral measurements: Theory of generalizability for scores and profiles</source>. <publisher-loc>Hoboken, NJ</publisher-loc>: <publisher-name>John Wiley</publisher-name>.</citation></ref>
<ref id="ref25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dalkey</surname> <given-names>N.</given-names></name> <name><surname>Helmer</surname> <given-names>O.</given-names></name></person-group> (<year>1963</year>). <article-title>An experimental application of the Delphi method to the use of experts</article-title>. <source>Manag. Sci.</source> <volume>9</volume>, <fpage>458</fpage>&#x2013;<lpage>467</lpage>. doi: <pub-id pub-id-type="doi">10.1287/mnsc.9.3.458</pub-id>, PMID: <pub-id pub-id-type="pmid">19642375</pub-id></citation></ref>
<ref id="ref1008"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Diamond</surname> <given-names>I. R.</given-names></name> <name><surname>Grant</surname> <given-names>C.</given-names></name> <name><surname>Feldman</surname> <given-names>M.</given-names></name> <name><surname>Pencharz</surname> <given-names>P. B.</given-names></name> <name><surname>Ling</surname> <given-names>S. C.</given-names></name> <name><surname>Moore</surname> <given-names>A. M.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Defining consensus: A systematic review recommends methodologic criteria for reporting of Delphi studies</article-title>. <source>J. Clinic. Epidemiol.</source> <volume>67</volume>, <fpage>401</fpage>&#x2013;<lpage>409</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jclinepi.2013.12.002</pub-id>, PMID: <pub-id pub-id-type="pmid">25626642</pub-id></citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Doyle</surname> <given-names>W.</given-names></name></person-group> (<year>1985</year>). <article-title>Recent research on classroom management: implications for teacher education</article-title>. <source>J. Teach. Educ.</source> <volume>36</volume>, <fpage>31</fpage>&#x2013;<lpage>35</lpage>. doi: <pub-id pub-id-type="doi">10.1177/002248718503600307</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Drollinger-Vetter</surname> <given-names>B.</given-names></name></person-group> (<year>2011</year>). <source>Verstehenselemente und strukturelle klarheit: Fachdidaktische qualit&#x00E4;t der anleitung von mathematischen verstehensprozessen im unterricht</source>. <publisher-loc>M&#x00FC;nster</publisher-loc>: <publisher-name>Waxmann</publisher-name>.</citation></ref>
<ref id="ref28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Emmer</surname> <given-names>E. T.</given-names></name> <name><surname>Stough</surname> <given-names>L. M.</given-names></name></person-group> (<year>2001</year>). <article-title>Classroom management: a critical part of educational psychology, with implications for teacher education</article-title>. <source>Educ. Psychol.</source> <volume>36</volume>, <fpage>103</fpage>&#x2013;<lpage>112</lpage>. doi: <pub-id pub-id-type="doi">10.1207/S15326985EP3602_5</pub-id></citation></ref>
<ref id="ref29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fenstermacher</surname> <given-names>G. D.</given-names></name> <name><surname>Richardson</surname> <given-names>V.</given-names></name></person-group> (<year>2005</year>). <article-title>On making determinations of quality in teaching</article-title>. <source>Teachers College Record</source> <volume>107</volume>, <fpage>186</fpage>&#x2013;<lpage>213</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1467-9620.2005.00462.x</pub-id></citation></ref>
<ref id="ref30"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Ferguson</surname> <given-names>R.</given-names></name> <name><surname>Danielson</surname> <given-names>C.</given-names></name></person-group> (<year>2015</year>). &#x201C;<article-title>How framework for teaching and tripod 7Cs evidence distinguish key components of effective teaching</article-title>&#x201D; in <source>Designing teacher evaluation systems: New guidance from the measures of effective teaching project</source>. eds. <person-group person-group-type="editor"><name><surname>Kane</surname> <given-names>T. J.</given-names></name> <name><surname>Kerr</surname> <given-names>K. A.</given-names></name> <name><surname>Pianta</surname> <given-names>R. C.</given-names></name></person-group> (<publisher-loc>Hoboken, NJ</publisher-loc>: <publisher-name>Wiley</publisher-name>), <fpage>98</fpage>&#x2013;<lpage>143</lpage>.</citation></ref>
<ref id="ref31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Green</surname> <given-names>R. A.</given-names></name></person-group> (<year>2014</year>). <article-title>The Delphi technique in educational research</article-title>. <source>SAGE Open</source> <volume>4</volume>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>. doi: <pub-id pub-id-type="doi">10.1177/2158244014529773</pub-id></citation></ref>
<ref id="ref32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grossman</surname> <given-names>P.</given-names></name> <name><surname>Loeb</surname> <given-names>S.</given-names></name> <name><surname>Cohen</surname> <given-names>J.</given-names></name> <name><surname>Wyckoff</surname> <given-names>J.</given-names></name></person-group> (<year>2013</year>). <article-title>Measure for measure: the relationship between measures of instructional practice in middle school English language arts and teachers&#x2019; value-added scores</article-title>. <source>Am. J. Educ.</source> <volume>119</volume>, <fpage>445</fpage>&#x2013;<lpage>470</lpage>. doi: <pub-id pub-id-type="doi">10.1086/669901</pub-id></citation></ref>
<ref id="ref33"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Helmer</surname> <given-names>O.</given-names></name></person-group> (<year>1966</year>). <source>The use of the Delphi technique in problems of educational innovations (P-3499)</source>. <publisher-loc>Santa Monica, CA</publisher-loc>: <publisher-name>The RAND Corporation</publisher-name>.</citation></ref>
<ref id="ref34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ho</surname> <given-names>S. S.</given-names></name> <name><surname>McLeod</surname> <given-names>D. M.</given-names></name></person-group> (<year>2008</year>). <article-title>Social-psychological influences on opinion expression in face-to-face and computer-mediated communication</article-title>. <source>Commun. Res.</source> <volume>35</volume>, <fpage>190</fpage>&#x2013;<lpage>207</lpage>. doi: <pub-id pub-id-type="doi">10.1177/0093650207313159</pub-id></citation></ref>
<ref id="ref35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>B. Y.</given-names></name> <name><surname>Fan</surname> <given-names>X.</given-names></name> <name><surname>Gu</surname> <given-names>C.</given-names></name> <name><surname>Yang</surname> <given-names>N.</given-names></name></person-group> (<year>2016</year>). <article-title>Applicability of the classroom assessment scoring system in Chinese preschools based on psychometric evidence</article-title>. <source>Early Educ. Dev.</source> <volume>27</volume>, <fpage>714</fpage>&#x2013;<lpage>734</lpage>. doi: <pub-id pub-id-type="doi">10.1080/10409289.2016.1113069</pub-id></citation></ref>
<ref id="ref36"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Janik</surname> <given-names>T.</given-names></name> <name><surname>Seidel</surname> <given-names>T.</given-names></name> <name><surname>Najvar</surname> <given-names>P.</given-names></name></person-group> (<year>2009</year>). &#x201C;<article-title>Introduction: on the power of video studies in investigating teaching and learning</article-title>&#x201D; in <source>The power of video studies in investigating teaching and learning in the classroom</source>. eds. <person-group person-group-type="editor"><name><surname>Janik</surname> <given-names>T.</given-names></name> <name><surname>Seidel</surname> <given-names>T.</given-names></name></person-group> (<publisher-loc>M&#x00FC;nster</publisher-loc>: <publisher-name>Waxmann</publisher-name>), <fpage>7</fpage>&#x2013;<lpage>19</lpage>.</citation></ref>
<ref id="ref37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jentsch</surname> <given-names>A.</given-names></name> <name><surname>Casale</surname> <given-names>G.</given-names></name> <name><surname>Schlesinger</surname> <given-names>L.</given-names></name> <name><surname>Kaiser</surname> <given-names>G.</given-names></name> <name><surname>K&#x00F6;nig</surname> <given-names>J.</given-names></name> <name><surname>Bl&#x00F6;meke</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>Variabilit&#x00E4;t und generalisierbarkeit von ratings zur qualit&#x00E4;t von mathematikunterricht zwischen und innerhalb von unterrichtsstunden</article-title>. <source>Unterrichtswissenschaft</source> <volume>48</volume>, <fpage>179</fpage>&#x2013;<lpage>197</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s42010-019-00061-8</pub-id></citation></ref>
<ref id="ref38"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Jentsch</surname> <given-names>A.</given-names></name> <name><surname>Heinrichs</surname> <given-names>H.</given-names></name> <name><surname>Schlesinger</surname> <given-names>L.</given-names></name> <name><surname>Kaiser</surname> <given-names>G.</given-names></name> <name><surname>K&#x00F6;nig</surname> <given-names>J.</given-names></name> <name><surname>Bl&#x00F6;meke</surname> <given-names>S.</given-names></name></person-group> (<year>2021a</year>). &#x201C;<article-title>Multi-group measurement invariance and generalizability analyses for an instructional quality observational instrument</article-title>&#x201D; in <source>Ways of analyzing teaching quality</source>. eds. <person-group person-group-type="editor"><name><surname>Blikstad-Balas</surname> <given-names>M.</given-names></name> <name><surname>Klette</surname> <given-names>K.</given-names></name> <name><surname>Tengberg</surname> <given-names>M.</given-names></name></person-group> (<publisher-loc>Oslo</publisher-loc>: <publisher-name>Scandinavian University Press</publisher-name>), <fpage>121</fpage>&#x2013;<lpage>139</lpage>.</citation></ref>
<ref id="ref39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jentsch</surname> <given-names>A.</given-names></name> <name><surname>Schlesinger</surname> <given-names>L.</given-names></name> <name><surname>Heinrichs</surname> <given-names>H.</given-names></name> <name><surname>Kaiser</surname> <given-names>G.</given-names></name> <name><surname>K&#x00F6;nig</surname> <given-names>J.</given-names></name> <name><surname>Bl&#x00F6;meke</surname> <given-names>S.</given-names></name></person-group> (<year>2021b</year>). <article-title>Erfassung der fachspezifischen qualit&#x00E4;t von mathematikunterricht: Faktorenstruktur und zusammenh&#x00E4;nge zur professionellen kompetenz von mathematiklehrpersonen</article-title>. <source>J. Math.-Didakt.</source> <volume>42</volume>, <fpage>97</fpage>&#x2013;<lpage>121</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s13138-020-00168-x</pub-id></citation></ref>
<ref id="ref40"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Jentsch</surname> <given-names>A.</given-names></name> <name><surname>Senden</surname> <given-names>B.</given-names></name></person-group> (<year>n.d.</year>). <source>Hybrid frameworks to capture teaching quality in secondary classrooms &#x2013; The case of the TEDS-Instruct observation system.</source></citation></ref>
<ref id="ref41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kaiser</surname> <given-names>G.</given-names></name> <name><surname>Bl&#x00F6;meke</surname> <given-names>S.</given-names></name> <name><surname>K&#x00F6;nig</surname> <given-names>J.</given-names></name> <name><surname>Busse</surname> <given-names>A.</given-names></name> <name><surname>D&#x00F6;hrmann</surname> <given-names>M.</given-names></name> <name><surname>Hoth</surname> <given-names>J.</given-names></name></person-group> (<year>2017</year>). <article-title>Professional competencies of (prospective) mathematics teachers&#x2014;cognitive versus situated approaches</article-title>. <source>Educ. Stud. Math.</source> <volume>94</volume>, <fpage>161</fpage>&#x2013;<lpage>182</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10649-016-9713-8</pub-id></citation></ref>
<ref id="ref42"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Kaiser</surname> <given-names>G.</given-names></name> <name><surname>K&#x00F6;nig</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). &#x201C;<article-title>Analyses and validation of central assessment instruments of the research program TEDS-M</article-title>&#x201D; in <source>Student learning in German higher education</source>. eds. <person-group person-group-type="editor"><name><surname>Zlatkin-Troitschanskaia</surname> <given-names>O.</given-names></name> <name><surname>Pant</surname> <given-names>H. A.</given-names></name> <name><surname>Toepper</surname> <given-names>M.</given-names></name> <name><surname>Lautenbach</surname> <given-names>C.</given-names></name></person-group> (<publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer</publisher-name>).</citation></ref>
<ref id="ref43"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Kane</surname> <given-names>T. J.</given-names></name> <name><surname>Cantrell</surname> <given-names>S.</given-names></name></person-group> (<year>2010</year>). Learning about teaching: initial findings from the measures of effective teaching project [MET project research paper]. Bill &#x0026; Melinda Gates Foundation. Available online at: <ext-link xlink:href="https://docs.gatesfoundation.org/Documents/preliminary-findings-research-paper.pdf" ext-link-type="uri">https://docs.gatesfoundation.org/Documents/preliminary-findings-research-paper.pdf</ext-link></citation></ref>
<ref id="ref44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kleickmann</surname> <given-names>T.</given-names></name> <name><surname>Vehmeyer</surname> <given-names>J.</given-names></name> <name><surname>M&#x00F6;ller</surname> <given-names>K.</given-names></name></person-group> (<year>2010</year>). <article-title>Zusammenh&#x00E4;nge zwischen lehrervorstellungen und kognitivem Strukturieren im unterricht am beispiel von scaffolding-ma&#x00DF;nahmen</article-title>. <source>Unterrichtswissenschaft</source> <volume>38</volume>, <fpage>210</fpage>&#x2013;<lpage>228</lpage>.</citation></ref>
<ref id="ref45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Klette</surname> <given-names>K.</given-names></name> <name><surname>Blikstad-Balas</surname> <given-names>M.</given-names></name> <name><surname>Roe</surname> <given-names>A.</given-names></name></person-group> (<year>2017</year>). <article-title>Linking instruction and student achievement. A research design for a new generation of classroom studies</article-title>. <source>Acta Didactica Norge</source> <volume>11</volume>, <fpage>1</fpage>&#x2013;<lpage>19</lpage>. doi: <pub-id pub-id-type="doi">10.5617/adno.4729</pub-id></citation></ref>
<ref id="ref46"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Klieme</surname> <given-names>E.</given-names></name> <name><surname>Pauli</surname> <given-names>C.</given-names></name> <name><surname>Reusser</surname> <given-names>K.</given-names></name></person-group> (<year>2009</year>). &#x201C;<article-title>The Pythagoras study: investigating effects of teaching and learning in Swiss and German mathematics classrooms</article-title>&#x201D; in <source>The power of video studies in investigating teaching and learning in the classroom</source>. eds. <person-group person-group-type="editor"><name><surname>Janik</surname> <given-names>T.</given-names></name> <name><surname>Seidel</surname> <given-names>T.</given-names></name></person-group> (<publisher-loc>M&#x00FC;nster</publisher-loc>: <publisher-name>Waxmann</publisher-name>), <fpage>137</fpage>&#x2013;<lpage>160</lpage>.</citation></ref>
<ref id="ref47"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Klieme</surname> <given-names>E.</given-names></name> <name><surname>Sch&#x00FC;mer</surname> <given-names>G.</given-names></name> <name><surname>Knoll</surname> <given-names>S.</given-names></name></person-group> (<year>2001</year>). &#x201C;<article-title>Mathematikunterricht in der sekundarstufe I: "aufgabenkultur" und unterrichtsgestaltung</article-title>&#x201D; in <source>TIMSS-Impulse f&#x00FC;r schule und unterricht</source>. eds. <person-group person-group-type="editor"><name><surname>Klieme</surname> <given-names>E.</given-names></name> <name><surname>Baumert</surname> <given-names>J.</given-names></name></person-group> (<publisher-loc>Bonn</publisher-loc>: <publisher-name>Bundesministerium f&#x00FC;r Bildung und Forschung</publisher-name>), <fpage>43</fpage>&#x2013;<lpage>57</lpage>.</citation></ref>
<ref id="ref48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Koo</surname> <given-names>T. K.</given-names></name> <name><surname>Li</surname> <given-names>M. Y.</given-names></name></person-group> (<year>2016</year>). <article-title>A guideline of selecting and reporting Intraclass correlation coefficients for reliability research</article-title>. <source>J. Chiropr. Med.</source> <volume>15</volume>, <fpage>155</fpage>&#x2013;<lpage>163</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jcm.2016.02.012</pub-id>, PMID: <pub-id pub-id-type="pmid">27330520</pub-id></citation></ref>
<ref id="ref49"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Kounin</surname> <given-names>J. S.</given-names></name></person-group> (<year>1970</year>). <source>Discipline and group management in classrooms</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Holt, Rinehart and Winston</publisher-name>.</citation></ref>
<ref id="ref50"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Kunter</surname> <given-names>M.</given-names></name> <name><surname>Voss</surname> <given-names>T.</given-names></name></person-group> (<year>2013</year>). &#x201C;<article-title>The model of instructional quality in COACTIV: a multicriteria analysis</article-title>&#x201D; in <source>Cognitive activation in the mathematics classroom and professional competence of teachers. Results from the COACTIV project</source>. eds. <person-group person-group-type="editor"><name><surname>Kunter</surname> <given-names>M.</given-names></name> <name><surname>Baumert</surname> <given-names>J.</given-names></name> <name><surname>Blum</surname> <given-names>W.</given-names></name> <name><surname>Klusmann</surname> <given-names>U.</given-names></name> <name><surname>Krauss</surname> <given-names>S.</given-names></name> <name><surname>Neubrand</surname> <given-names>M.</given-names></name></person-group> (<publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>97</fpage>&#x2013;<lpage>124</lpage>.</citation></ref>
<ref id="ref51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Leyva</surname> <given-names>D.</given-names></name> <name><surname>Weiland</surname> <given-names>C.</given-names></name> <name><surname>Barata</surname> <given-names>M.</given-names></name> <name><surname>Yoshikawa</surname> <given-names>H.</given-names></name> <name><surname>Snow</surname> <given-names>C.</given-names></name> <name><surname>Trevi&#x00F1;o</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Teacher&#x2013;child interactions in Chile and their associations with prekindergarten outcomes</article-title>. <source>Child Dev.</source> <volume>86</volume>, <fpage>781</fpage>&#x2013;<lpage>799</lpage>. doi: <pub-id pub-id-type="doi">10.1111/cdev.12342</pub-id>, PMID: <pub-id pub-id-type="pmid">25626642</pub-id></citation></ref>
<ref id="ref52"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Linstone</surname> <given-names>H.</given-names></name> <name><surname>Turoff</surname> <given-names>M.</given-names></name></person-group> (<year>1975</year>). &#x201C;<article-title>Introduction</article-title>&#x201D; in <source>The Delphi method: Techniques and applications</source>. eds. <person-group person-group-type="editor"><name><surname>Linstone</surname> <given-names>H.</given-names></name> <name><surname>Turoff</surname> <given-names>M.</given-names></name></person-group> (<publisher-loc>Boston</publisher-loc>: <publisher-name>Addison-Wesley Publishing Company</publisher-name>), <fpage>3</fpage>&#x2013;<lpage>12</lpage>.</citation></ref>
<ref id="ref53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lipowsky</surname> <given-names>F.</given-names></name> <name><surname>Rakoczy</surname> <given-names>K.</given-names></name> <name><surname>Pauli</surname> <given-names>C.</given-names></name> <name><surname>Drollinger-Vetter</surname> <given-names>B.</given-names></name> <name><surname>Klieme</surname> <given-names>E.</given-names></name> <name><surname>Reusser</surname> <given-names>K.</given-names></name></person-group> (<year>2009</year>). <article-title>Quality of geometry instruction and its short-term impact on students' understanding of the Pythagorean theorem</article-title>. <source>Learn. Instr.</source> <volume>19</volume>, <fpage>527</fpage>&#x2013;<lpage>537</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.learninstruc.2008.11.001</pub-id></citation></ref>
<ref id="ref54"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Bell</surname> <given-names>C. A.</given-names></name> <name><surname>Jones</surname> <given-names>N. D.</given-names></name> <name><surname>McCaffrey</surname> <given-names>D. F.</given-names></name></person-group> (<year>2019</year>). <article-title>Classroom observation systems in context: a case for the validation of observation systems</article-title>. <source>Educ. Assess. Eval. Account.</source> <volume>31</volume>, <fpage>61</fpage>&#x2013;<lpage>95</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11092-018-09291-3</pub-id></citation></ref>
<ref id="ref55"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Luoto</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). Exploring, understanding, and problematizing patterns of instructional quality: A study of instructional quality in Finnish&#x2013;Swedish and Norwegian lower secondary mathematics classrooms [Doctoral dissertation, University of Oslo]. DUO Research Archive. Available online at: <ext-link xlink:href="http://urn.nb.no/URN:NBN:no-88324" ext-link-type="uri">http://urn.nb.no/URN:NBN:no-88324</ext-link></citation></ref>
<ref id="ref56"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luoto</surname> <given-names>J.</given-names></name> <name><surname>Klette</surname> <given-names>K.</given-names></name> <name><surname>Blikstad-Balas</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>Possible biases in observation systems when applied across contexts: conceptualizing, operationalizing, and sequencing instructional quality</article-title>. <source>Educ. Assess. Eval. Account.</source> <volume>35</volume>, <fpage>105</fpage>&#x2013;<lpage>128</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11092-022-09394-y</pub-id>, PMID: <pub-id pub-id-type="pmid">39867833</pub-id></citation></ref>
<ref id="ref57"><citation citation-type="other"><person-group person-group-type="author"><collab id="coll3">Mangold</collab></person-group>. (<year>2023</year>). Interact (version 18.7.7.17) [Computer software]. Available online at: <ext-link xlink:href="https://www.mangold-international.com/en/products/software/behavior-research-with-mangold-interact.html" ext-link-type="uri">https://www.mangold-international.com/en/products/software/behavior-research-with-mangold-interact.html</ext-link></citation></ref>
<ref id="ref58"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mashburn</surname> <given-names>A. J.</given-names></name> <name><surname>Meyer</surname> <given-names>J. P.</given-names></name> <name><surname>Allen</surname> <given-names>J. P.</given-names></name> <name><surname>Pianta</surname> <given-names>R. C.</given-names></name></person-group> (<year>2014</year>). <article-title>The effect of observation length and presentation order on the reliability and validity of an observational measure of teaching quality</article-title>. <source>Educ. Psychol. Meas.</source> <volume>74</volume>, <fpage>400</fpage>&#x2013;<lpage>422</lpage>. doi: <pub-id pub-id-type="doi">10.1177/0013164413515882</pub-id></citation></ref>
<ref id="ref59"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>McKenna</surname> <given-names>H. P.</given-names></name></person-group> (<year>1994</year>). <article-title>The Delphi technique: a worthwhile research approach for nursing?</article-title> <source>J. Adv. Nurs.</source> <volume>19</volume>, <fpage>1221</fpage>&#x2013;<lpage>1225</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1365-2648.1994.tb01207.x</pub-id>, PMID: <pub-id pub-id-type="pmid">7930104</pub-id></citation></ref>
<ref id="ref60"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mengual-Andr&#x00E9;s</surname> <given-names>S.</given-names></name> <name><surname>Roig-Vila</surname> <given-names>R.</given-names></name> <name><surname>Mira</surname> <given-names>J. B.</given-names></name></person-group> (<year>2016</year>). <article-title>Delphi study for the design and validation of a questionnaire about digital competences in higher education</article-title>. <source>Intern. J. Edu. Technol. High.</source> <volume>13</volume>:<fpage>12</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s41239-016-0009-y</pub-id>, PMID: <pub-id pub-id-type="pmid">39825265</pub-id></citation></ref>
<ref id="ref61"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Minner</surname> <given-names>D.</given-names></name> <name><surname>DeLisi</surname> <given-names>J.</given-names></name></person-group> (<year>2012</year>). <source>Inquiring into science instruction observation protocol (ISIOP): Codebook</source>. <publisher-loc>Waltham, MA</publisher-loc>: <publisher-name>Education Development Center, Inc</publisher-name>.</citation></ref>
<ref id="ref62"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Muijs</surname> <given-names>D.</given-names></name> <name><surname>Kyriakides</surname> <given-names>L.</given-names></name> <name><surname>van der Werf</surname> <given-names>G.</given-names></name> <name><surname>Creemers</surname> <given-names>B.</given-names></name> <name><surname>Timperley</surname> <given-names>H.</given-names></name> <name><surname>Earl</surname> <given-names>L.</given-names></name></person-group> (<year>2014</year>). <article-title>State of the art &#x2013; teacher effectiveness and professional learning</article-title>. <source>Sch. Eff. Sch. Improv.</source> <volume>25</volume>, <fpage>231</fpage>&#x2013;<lpage>256</lpage>. doi: <pub-id pub-id-type="doi">10.1080/09243453.2014.885451</pub-id></citation></ref>
<ref id="ref63"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Muijs</surname> <given-names>D.</given-names></name> <name><surname>Reynolds</surname> <given-names>D.</given-names></name> <name><surname>Sammons</surname> <given-names>P.</given-names></name> <name><surname>Kyriakides</surname> <given-names>L.</given-names></name> <name><surname>Creemers</surname> <given-names>B. P. M.</given-names></name> <name><surname>Teddlie</surname> <given-names>C.</given-names></name></person-group> (<year>2018</year>). <article-title>Assessing individual lessons using a generic teacher observation instrument: how useful is the international system for teacher observation and feedback (ISTOF)?</article-title> <source>ZDM</source> <volume>50</volume>, <fpage>395</fpage>&#x2013;<lpage>406</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11858-018-0921-9</pub-id></citation></ref>
<ref id="ref64"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Mullis</surname> <given-names>I. V. S.</given-names></name> <name><surname>Martin</surname> <given-names>M. O.</given-names></name> <name><surname>Foy</surname> <given-names>P.</given-names></name> <name><surname>Kelly</surname> <given-names>D. L.</given-names></name> <name><surname>Fishbein</surname> <given-names>B.</given-names></name></person-group> (<year>2020</year>). TIMSS 2019 international results in mathematics and science. TIMSS &#x0026; PIRLS International Study Center, Lynch School of Education and Human Development, Boston College, and International Association for the Evaluation of Educational Achievement (IEA). Available online at: <ext-link xlink:href="https://timssandpirls.bc.edu/timss2019/international-results/" ext-link-type="uri">https://timssandpirls.bc.edu/timss2019/international-results/</ext-link></citation></ref>
<ref id="ref65"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ng</surname> <given-names>E. L.</given-names></name> <name><surname>Bull</surname> <given-names>R.</given-names></name> <name><surname>Bautista</surname> <given-names>A.</given-names></name> <name><surname>Poon</surname> <given-names>K.</given-names></name></person-group> (<year>2021</year>). <article-title>A bifactor model of the classroom assessment scoring system in preschool and early intervention classrooms in Singapore</article-title>. <source>Int. J. Early Child.</source> <volume>53</volume>, <fpage>197</fpage>&#x2013;<lpage>218</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s13158-021-00292-w</pub-id></citation></ref>
<ref id="ref67"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oliver</surname> <given-names>R. M.</given-names></name> <name><surname>Wehby</surname> <given-names>J. H.</given-names></name> <name><surname>Reschly</surname> <given-names>D. J.</given-names></name></person-group> (<year>2011</year>). <article-title>Teacher classroom management practices: effects on disruptive or aggressive student behavior</article-title>. <source>Campbell Syst. Rev.</source> <volume>7</volume>, <fpage>1</fpage>&#x2013;<lpage>55</lpage>. doi: <pub-id pub-id-type="doi">10.4073/csr.2011.4</pub-id></citation></ref>
<ref id="ref68"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Pacheco</surname> <given-names>A.</given-names></name></person-group> (<year>2009</year>). &#x201C;<article-title>Mapping the terrain of teacher quality</article-title>&#x201D; in <source>Measurement issues and assessment for teaching quality</source>. ed. <person-group person-group-type="editor"><name><surname>Gitomer</surname> <given-names>D. H.</given-names></name></person-group> (<publisher-loc>Thousand Oaks, CA</publisher-loc>: <publisher-name>SAGE Publications</publisher-name>), <fpage>160</fpage>&#x2013;<lpage>178</lpage>.</citation></ref>
<ref id="ref69"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Palinkas</surname> <given-names>L. A.</given-names></name> <name><surname>Horwitz</surname> <given-names>S. M.</given-names></name> <name><surname>Green</surname> <given-names>C. A.</given-names></name> <name><surname>Wisdom</surname> <given-names>J. P.</given-names></name> <name><surname>Duan</surname> <given-names>N.</given-names></name> <name><surname>Hoagwood</surname> <given-names>K.</given-names></name></person-group> (<year>2015</year>). <article-title>Purposeful sampling for qualitative data collection and analysis in mixed method implementation research</article-title>. <source>Adm. Policy Ment. Health Ment. Health Serv. Res.</source> <volume>42</volume>, <fpage>533</fpage>&#x2013;<lpage>544</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10488-013-0528-y</pub-id>, PMID: <pub-id pub-id-type="pmid">24193818</pub-id></citation></ref>
<ref id="ref70"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Pianta</surname> <given-names>R. C.</given-names></name> <name><surname>La Paro</surname> <given-names>K. M.</given-names></name> <name><surname>Hamre</surname> <given-names>B. K.</given-names></name></person-group> (<year>2008</year>). <source>Classroom assessment scoring system&#x2122;: Manual K-3</source>. <publisher-loc>Towson, MD</publisher-loc>: <publisher-name>Paul H Brookes Publishing</publisher-name>.</citation></ref>
<ref id="ref71"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Praetorius</surname> <given-names>A.-K.</given-names></name> <name><surname>Charalambous</surname> <given-names>C. Y.</given-names></name></person-group> (<year>2018</year>). <article-title>Classroom observation frameworks for studying instructional quality: looking back and looking forward</article-title>. <source>ZDM</source> <volume>50</volume>, <fpage>535</fpage>&#x2013;<lpage>553</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11858-018-0946-0</pub-id></citation></ref>
<ref id="ref72"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Praetorius</surname> <given-names>A. K.</given-names></name> <name><surname>Charalambous</surname> <given-names>C.</given-names></name></person-group> (<year>2023</year>). &#x201C;<article-title>Creating practical theories of teaching</article-title>&#x201D; in <source>Theorizing teaching: Current status and open issues</source>. eds. <person-group person-group-type="editor"><name><surname>Praetorius</surname> <given-names>A.-K.</given-names></name> <name><surname>Charalambous</surname> <given-names>C.</given-names></name></person-group> (<publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>23</fpage>&#x2013;<lpage>56</lpage>.</citation></ref>
<ref id="ref73"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Praetorius</surname> <given-names>A.-K.</given-names></name> <name><surname>Klieme</surname> <given-names>E.</given-names></name> <name><surname>Herbert</surname> <given-names>B.</given-names></name> <name><surname>Pinger</surname> <given-names>P.</given-names></name></person-group> (<year>2018</year>). <article-title>Generic dimensions of teaching quality: the German framework of three basic dimensions</article-title>. <source>Math. Educ.</source> <volume>50</volume>, <fpage>407</fpage>&#x2013;<lpage>426</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11858-018-0918-4</pub-id>, PMID: <pub-id pub-id-type="pmid">39867833</pub-id></citation></ref>
<ref id="ref74"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Praetorius</surname> <given-names>A.-K.</given-names></name> <name><surname>Klieme</surname> <given-names>E.</given-names></name> <name><surname>Kleickmann</surname> <given-names>T.</given-names></name> <name><surname>Brunner</surname> <given-names>E.</given-names></name> <name><surname>Lindmeier</surname> <given-names>A.</given-names></name> <name><surname>Taut</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Towards developing a theory of generic teaching quality: origin, current status, and necessary next steps regarding the three basic dimensions model</article-title>. <publisher-name>Zeitschrift f&#x00FC;r P&#x00E4;dagogik</publisher-name>. <volume>66</volume>, <fpage>15</fpage>&#x2013;<lpage>36</lpage>.</citation></ref>
<ref id="ref75"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Praetorius</surname> <given-names>A.-K.</given-names></name> <name><surname>Pauli</surname> <given-names>C.</given-names></name> <name><surname>Reusser</surname> <given-names>K.</given-names></name> <name><surname>Rakoczy</surname> <given-names>K.</given-names></name> <name><surname>Klieme</surname> <given-names>E.</given-names></name></person-group> (<year>2014</year>). <article-title>One lesson is all you need? Stability of instructional quality across lessons</article-title>. <source>Learn. Instr.</source> <volume>31</volume>, <fpage>2</fpage>&#x2013;<lpage>12</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.learninstruc.2013.12.002</pub-id></citation></ref>
<ref id="ref76"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Praetorius</surname> <given-names>A.-K.</given-names></name> <name><surname>Vieluf</surname> <given-names>S.</given-names></name> <name><surname>Sa&#x00DF;</surname> <given-names>S.</given-names></name> <name><surname>Bernholt</surname> <given-names>A.</given-names></name> <name><surname>Klieme</surname> <given-names>E.</given-names></name></person-group> (<year>2016</year>). <article-title>The same in German as in English? Investigating the subject-specificity of teaching quality</article-title>. <source>Z. Erzieh.</source> <volume>19</volume>, <fpage>191</fpage>&#x2013;<lpage>209</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11618-015-0660-4</pub-id></citation></ref>
<ref id="ref77"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Sabornie</surname> <given-names>E. J.</given-names></name> <name><surname>Espelage</surname> <given-names>D. L.</given-names></name></person-group> (<year>2022</year>). <source>Handbook of classroom management</source>. <edition>3rd</edition> Edn. <publisher-loc>London</publisher-loc>: <publisher-name>Routledge</publisher-name>.</citation></ref>
<ref id="ref78"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Scheerens</surname> <given-names>J.</given-names></name> <name><surname>Luyten</surname> <given-names>J. W.</given-names></name> <name><surname>Steen</surname> <given-names>R.</given-names></name> <name><surname>de Thouars</surname> <given-names>Y. C. H.</given-names></name></person-group> (<year>2007</year>). <source>Review and meta-analyses of school and teaching effectiveness</source>. <publisher-loc>Enschede</publisher-loc>: <publisher-name>University of Twente</publisher-name>.</citation></ref>
<ref id="ref79"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schlesinger</surname> <given-names>L.</given-names></name> <name><surname>Jentsch</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). <article-title>Theoretical and methodological challenges in measuring instructional quality in mathematics education using classroom observations</article-title>. <source>ZDM</source> <volume>48</volume>, <fpage>29</fpage>&#x2013;<lpage>40</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11858-016-0765-0</pub-id></citation></ref>
<ref id="ref80"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schlesinger</surname> <given-names>L.</given-names></name> <name><surname>Jentsch</surname> <given-names>A.</given-names></name> <name><surname>Kaiser</surname> <given-names>G.</given-names></name> <name><surname>K&#x00F6;nig</surname> <given-names>J.</given-names></name> <name><surname>Bl&#x00F6;meke</surname> <given-names>S.</given-names></name></person-group> (<year>2018</year>). <article-title>Subject-specific characteristics of instructional quality in mathematics education</article-title>. <source>ZDM</source> <volume>50</volume>, <fpage>475</fpage>&#x2013;<lpage>490</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11858-018-0917-5</pub-id></citation></ref>
<ref id="ref81"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Seidel</surname> <given-names>T.</given-names></name> <name><surname>Shavelson</surname> <given-names>R. J.</given-names></name></person-group> (<year>2007</year>). <article-title>Teaching effectiveness research in the past decade: the role of theory and research design in disentangling meta-analysis results</article-title>. <source>Rev. Educ. Res.</source> <volume>77</volume>, <fpage>454</fpage>&#x2013;<lpage>499</lpage>. doi: <pub-id pub-id-type="doi">10.3102/0034654307310317</pub-id></citation></ref>
<ref id="ref82"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Senden</surname> <given-names>B.</given-names></name> <name><surname>Nilsen</surname> <given-names>T.</given-names></name> <name><surname>Bl&#x00F6;meke</surname> <given-names>S.</given-names></name></person-group> (<year>2022</year>). &#x201C;<article-title>5. Instructional quality: a review of conceptualizations, measurement approaches, and research findings</article-title>&#x201D; in <source>Ways of analyzing teaching quality: Potentials and pitfalls</source>. eds. <person-group person-group-type="editor"><name><surname>Blikstad-Balas</surname> <given-names>M.</given-names></name> <name><surname>Klette</surname> <given-names>K.</given-names></name> <name><surname>Tengberg</surname> <given-names>M.</given-names></name></person-group> (<publisher-loc>Oslo</publisher-loc>: <publisher-name>Scandinavian University Press</publisher-name>), <fpage>140</fpage>&#x2013;<lpage>172</lpage>.</citation></ref>
<ref id="ref83"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Shavelson</surname> <given-names>R. J.</given-names></name> <name><surname>Webb</surname> <given-names>N. M.</given-names></name></person-group> (<year>1991</year>). <source>Generalizability theory: A primer</source>. <publisher-loc>Thousand Oaks, CA</publisher-loc>: <publisher-name>SAGE Publications</publisher-name>.</citation></ref>
<ref id="ref84"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Slot</surname> <given-names>P. L.</given-names></name> <name><surname>Boom</surname> <given-names>J.</given-names></name> <name><surname>Verhagen</surname> <given-names>J.</given-names></name> <name><surname>Leseman</surname> <given-names>P. P. M.</given-names></name></person-group> (<year>2017</year>). <article-title>Measurement properties of the CLASS toddler in ECEC in the Netherlands</article-title>. <source>J. Appl. Dev. Psychol.</source> <volume>48</volume>, <fpage>79</fpage>&#x2013;<lpage>91</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.appdev.2016.11.008</pub-id></citation></ref>
<ref id="ref85"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>K. S.</given-names></name> <name><surname>Simpson</surname> <given-names>R. D.</given-names></name></person-group> (<year>1995</year>). <article-title>Validating teaching competencies for faculty members in higher education: a national study using the Delphi method</article-title>. <source>Innov. High. Educ.</source> <volume>19</volume>, <fpage>223</fpage>&#x2013;<lpage>234</lpage>. doi: <pub-id pub-id-type="doi">10.1007/BF01191221</pub-id></citation></ref>
<ref id="ref86"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Teddlie</surname> <given-names>C.</given-names></name> <name><surname>Creemers</surname> <given-names>B.</given-names></name> <name><surname>Kyriakides</surname> <given-names>L.</given-names></name> <name><surname>Muijs</surname> <given-names>D.</given-names></name> <name><surname>Yu</surname> <given-names>F.</given-names></name></person-group> (<year>2006</year>). <article-title>The international system for teacher observation and feedback: evolution of an international study of teacher effectiveness constructs</article-title>. <source>Educ. Res. Eval.</source> <volume>12</volume>, <fpage>561</fpage>&#x2013;<lpage>582</lpage>. doi: <pub-id pub-id-type="doi">10.1080/13803610600874067</pub-id></citation></ref>
<ref id="ref87"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thorpe</surname> <given-names>K.</given-names></name> <name><surname>Houen</surname> <given-names>S.</given-names></name> <name><surname>Rankin</surname> <given-names>P.</given-names></name> <name><surname>Pattinson</surname> <given-names>C.</given-names></name> <name><surname>Staton</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <article-title>Do the numbers add up? Questioning measurement that places Australian ECEC teaching as &#x2018;low quality&#x2019;</article-title>. <source>Aust. Educ. Res.</source> <volume>50</volume>, <fpage>781</fpage>&#x2013;<lpage>800</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s13384-022-00525-4</pub-id></citation></ref>
<ref id="ref88"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Turner</surname> <given-names>J. C.</given-names></name> <name><surname>Meyer</surname> <given-names>D. K.</given-names></name> <name><surname>Cox</surname> <given-names>K. E.</given-names></name> <name><surname>Logan</surname> <given-names>C.</given-names></name> <name><surname>DiCintio</surname> <given-names>M.</given-names></name> <name><surname>Thomas</surname> <given-names>C. T.</given-names></name></person-group> (<year>1998</year>). <article-title>Creating contexts for involvement in mathematics</article-title>. <source>J. Educ. Psychol.</source> <volume>90</volume>, <fpage>730</fpage>&#x2013;<lpage>745</lpage>. doi: <pub-id pub-id-type="doi">10.1037/0022-0663.90.4.730</pub-id></citation></ref>
<ref id="ref89"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Van de Pol</surname> <given-names>J.</given-names></name> <name><surname>Volman</surname> <given-names>M.</given-names></name> <name><surname>Beishuizen</surname> <given-names>J.</given-names></name></person-group> (<year>2010</year>). <article-title>Scaffolding in teacher&#x2013;student interaction: a decade of research</article-title>. <source>Educ. Psychol. Rev.</source> <volume>22</volume>, <fpage>271</fpage>&#x2013;<lpage>296</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10648-010-9127-6</pub-id>, PMID: <pub-id pub-id-type="pmid">39867833</pub-id></citation></ref>
<ref id="ref90"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Virtanen</surname> <given-names>T. E.</given-names></name> <name><surname>Pakarinen</surname> <given-names>E.</given-names></name> <name><surname>Lerkkanen</surname> <given-names>M.-K.</given-names></name> <name><surname>Poikkeus</surname> <given-names>A.-M.</given-names></name> <name><surname>Siekkinen</surname> <given-names>M.</given-names></name> <name><surname>Nurmi</surname> <given-names>J.-E.</given-names></name></person-group> (<year>2018</year>). <article-title>A validation study of classroom assessment scoring system&#x2013;secondary in the Finnish school context</article-title>. <source>J. Early Adolesc.</source> <volume>38</volume>, <fpage>849</fpage>&#x2013;<lpage>880</lpage>. doi: <pub-id pub-id-type="doi">10.1177/0272431617699944</pub-id></citation></ref>
<ref id="ref91"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Westerg&#x00E5;rd</surname> <given-names>E.</given-names></name> <name><surname>Ertesv&#x00E5;g</surname> <given-names>S. K.</given-names></name> <name><surname>Rafaelsen</surname> <given-names>F.</given-names></name></person-group> (<year>2019</year>). <article-title>A preliminary validity of the classroom assessment scoring system in Norwegian lower-secondary schools</article-title>. <source>Scand. J. Educ. Res.</source> <volume>63</volume>, <fpage>566</fpage>&#x2013;<lpage>584</lpage>. doi: <pub-id pub-id-type="doi">10.1080/00313831.2017.1415964</pub-id></citation></ref>
<ref id="ref1009"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>White</surname> <given-names>M.</given-names></name> <name><surname>Luoto</surname> <given-names>J.</given-names></name> <name><surname>Klette</surname> <given-names>K.</given-names></name> <name><surname>Blikstad-Balas</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>Bringing the conceptualization and measurement of teaching into alignment</article-title>. <source>Studies Educat. Evalu.</source> <volume>75</volume>:<fpage>101204</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.stueduc.2022.101204</pub-id>, PMID: <pub-id pub-id-type="pmid">39867833</pub-id></citation></ref>
<ref id="ref92"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wittek</surname> <given-names>L.</given-names></name> <name><surname>Kvernbekk</surname> <given-names>T.</given-names></name></person-group> (<year>2011</year>). <article-title>On the problems of asking for a definition of quality in education</article-title>. <source>Scand. J. Educ. Res.</source> <volume>55</volume>, <fpage>671</fpage>&#x2013;<lpage>684</lpage>. doi: <pub-id pub-id-type="doi">10.1080/00313831.2011.594618</pub-id></citation></ref>
</ref-list>
</back>
</article>