<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Med.</journal-id>
<journal-title>Frontiers in Medicine</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Med.</abbrev-journal-title>
<issn pub-type="epub">2296-858X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmed.2023.1201517</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Medicine</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Inter-rater reliability of the extended Composite Quality Score (CQS-2)</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Mickenautsch</surname>
<given-names>Steffen</given-names>
</name>
<xref rid="aff1" ref-type="aff"><sup>1</sup></xref>
<xref rid="aff2" ref-type="aff"><sup>2</sup></xref>
<xref rid="aff3" ref-type="aff"><sup>3</sup></xref>
<xref rid="c001" ref-type="corresp"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/246728/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Rupf</surname>
<given-names>Stefan</given-names>
</name>
<xref rid="aff4" ref-type="aff"><sup>4</sup></xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mileti&#x0107;</surname>
<given-names>Ivana</given-names>
</name>
<xref rid="aff5" ref-type="aff"><sup>5</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/893229/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Str&#x00E4;hle</surname>
<given-names>Ulf Tilman</given-names>
</name>
<xref rid="aff4" ref-type="aff"><sup>4</sup></xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sturm</surname>
<given-names>Richard</given-names>
</name>
<xref rid="aff6" ref-type="aff"><sup>6</sup></xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kimmie-Dhansay</surname>
<given-names>Faheema</given-names>
</name>
<xref rid="aff7" ref-type="aff"><sup>7</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1417265/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Vidosusi&#x0107;</surname>
<given-names>Kata</given-names>
</name>
<xref rid="aff5" ref-type="aff"><sup>5</sup></xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yengopal</surname>
<given-names>Veerasamy</given-names>
</name>
<xref rid="aff1" ref-type="aff"><sup>1</sup></xref>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Faculty of Dentistry, University of the Western Cape</institution>, <addr-line>Bellville</addr-line>, <country>South Africa</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Community Dentistry, Faculty of Health Sciences, School of Oral Health Sciences, University of the Witwatersrand</institution>, <addr-line>Johannesburg</addr-line>, <country>South Africa</country></aff>
<aff id="aff3"><sup>3</sup><institution>Review Centre for Health Science Research</institution>, <addr-line>Johannesburg</addr-line>, <country>South Africa</country></aff>
<aff id="aff4"><sup>4</sup><institution>Synoptic Dentistry, Saarland University</institution>, <addr-line>Homburg</addr-line>, <country>Germany</country></aff>
<aff id="aff5"><sup>5</sup><institution>Department of Endodontics and Restorative Dentistry, School of Dental Medicine, University of Zagreb</institution>, <addr-line>Zagreb</addr-line>, <country>Croatia</country></aff>
<aff id="aff6"><sup>6</sup><institution>Department of Operative, Preventive and Paediatric Dentistry, Charit&#x00E9; &#x2013; Universit&#x00E4;tsmedizin Berlin</institution>, <addr-line>Berlin</addr-line>, <country>Germany</country></aff>
<aff id="aff7"><sup>7</sup><institution>Department of Community Oral Health, Faculty of Dentistry, University of the Western Cape</institution>, <addr-line>Bellville</addr-line>, <country>South Africa</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0001">
<p>Edited by: Lise Aagaard, Independent Researcher, Copenhagen, Denmark</p>
</fn>
<fn fn-type="edited-by" id="fn0002">
<p>Reviewed by: Frits Lekkerkerker, Consultant, Amsterdam, Netherlands; Segundo Mariz, European Medicines Agency, Netherlands</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Steffen Mickenautsch, <email>neem@global.co.za</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>17</day>
<month>08</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>10</volume>
<elocation-id>1201517</elocation-id>
<history>
<date date-type="received">
<day>06</day>
<month>04</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>08</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2023 Mickenautsch, Rupf, Mileti&#x0107;, Str&#x00E4;hle, Sturm, Kimmie-Dhansay, Vidosusi&#x0107; and Yengopal.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Mickenautsch, Rupf, Mileti&#x0107;, Str&#x00E4;hle, Sturm, Kimmie-Dhansay, Vidosusi&#x0107; and Yengopal</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec id="sec1">
<title>Aim</title>
<p>To establish the inter-rater reliability of the Composite Quality Score (CQS-2) and to test the null hypothesis that it did not differ significantly from that of the first CQS version (CQS-1).</p>
</sec>
<sec id="sec2">
<title>Materials and methods</title>
<p>Four independent raters were selected to rate 45 clinical trial reports using CQS-1 and CQS-2. The raters remained unaware of each other&#x2019;s participation in this study until all rating had been completed. Each rater received only one rating template at a time in a random sequence for CQS-1 and CQS-2 rating. Raters completed each template and sent these back to the principal investigator. Each rater received their next template 2 weeks after submission of the completed previous template. The inter-rater reliabilities for the overall appraisal score of the CQS-1 and the CQS-2 were established by using the Brennan-Prediger coefficient (BPC). The coefficients of both CQS versions were compared by using the two-sample <italic>z</italic>-test. During secondary analysis, the BPCs for every criterion and each corroboration level for both CQS versions were established.</p>
</sec>
<sec id="sec3">
<title>Results</title>
<p>The BPC for the CQS-1 was 0.85 (95% CI: 0.64&#x2013;1.00) and for the CQS-2 it was 1.00 (95% CI: 0.94&#x2013;1.00), suggesting a very high inter-rater reliability for both. The difference between the two CQS versions was statistically not significant (<italic>p</italic> = 0.17). The null hypothesis was accepted.</p>
</sec>
<sec id="sec4">
<title>Conclusion</title>
<p>The CQS-2 is still under development, This study shows that it is associated with a very high inter-rater reliability, which did not statistically significantly differ from that of the CQS-1. The promising results of this study warrant further investigation in the applicability of the CQS-2 as an appraisal tool for prospective controlled clinical therapy trials.</p>
</sec>
</abstract>
<kwd-group>
<kwd>Composite Quality Score</kwd>
<kwd>clinical rial</kwd>
<kwd>trial appraisal</kwd>
<kwd>inter-rater reliability</kwd>
<kwd>systematic review</kwd>
</kwd-group>
<counts>
<fig-count count="0"/>
<table-count count="3"/>
<equation-count count="0"/>
<ref-count count="17"/>
<page-count count="6"/>
<word-count count="4867"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Regulatory Science</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec5">
<label>1.</label>
<title>Introduction</title>
<p>The Composite Quality Score (CQS) is a recently established appraisal tool for prospective, controlled, clinical therapy trials based on the deductive falsification approach (<xref ref-type="bibr" rid="ref1">1</xref>). Trial appraisal that follows such an approach assumes that any trial design characteristic (or the lack thereof) which lies outside a particular set of applied trial appraisal criteria, such as that of the Jadad scale (<xref ref-type="bibr" rid="ref2">2</xref>) or Cochrane&#x2019;s Risk of Bias (RoB) tool (<xref ref-type="bibr" rid="ref3">3</xref>, <xref ref-type="bibr" rid="ref4">4</xref>), may completely falsify the truthfulness of trial results. It therefore rejects any confidence in &#x201C;low bias risk.&#x201D; Consequently, the approach accepts that, in principle, it is impossible to establish &#x201C;low bias risk&#x201D; for any trial. Instead, the CQS follows the concept that, although &#x201C;low bias risk&#x201D; cannot be proven, it is possible to establish with high certainty whether bias risk is high. High bias risk is recognized when essential characteristics are absent for a therapy trial to reflect the true effect estimate (<xref ref-type="bibr" rid="ref5">5</xref>).</p>
<p>The first version of the CQS (CQS-1) was developed as a composite of trial appraisal categories for both systematic and random error. The CQS-1 appeared to have been sufficient for trial appraisal in the field of restorative dentistry, where 681 from the total of 683 trial reports could be rated with high confidence as of high bias risk (<xref ref-type="bibr" rid="ref6">6</xref>). In addition, Mickenautsch et al. investigated the CQS-1 inter-rater reliability (<xref ref-type="bibr" rid="ref7">7</xref>). The results showed a very high inter-rater reliability, based on an &#x201C;almost perfect&#x201D; strength of inter-rater agreement, according to the Landis/Koch Kappa&#x2019;s Benchmark Scale (Brennan-Prediger coefficient (BPC) 0.95; 95% CI: 0.87&#x2013;1.00) (<xref ref-type="bibr" rid="ref8">8</xref>) that was statistically significantly higher than that of the first version of Cochrane&#x2019;s RoB tool (<xref ref-type="bibr" rid="ref7">7</xref>).</p>
<p>However, while the current CQS-1 appeared to have been sufficient for clinical trial appraisal in the field of restorative dentistry, other fields of clinical therapy may contain a higher number of trials that would pass its three simple, non-restrictive criteria. For that reason, a new CQS version (CQS-2) was developed based on meta-epidemiological study evidence. Subsequently, one new criterion concerning double-blinding was added and criteria II and III of the original CQS version were amended (<xref ref-type="bibr" rid="ref9">9</xref>). These changes raise the questions whether the CQS-2 is associated with a high inter-rater reliability, too, and whether such reliability would statistically significantly differ from that of the CQS-1.</p>
<p>Therefore, this study aimed to establish the inter-rater reliability of the CQS-1 and of the CQS-2 and test the null hypothesis that the inter-rater reliability of the CQS-2 does not differ significantly from that of the CQS-1.</p>
</sec>
<sec sec-type="methods" id="sec6">
<label>2.</label>
<title>Methods</title>
<p>The methodology of this study was pre-specified in a protocol, which was made available online prior to the start of the study (<xref ref-type="bibr" rid="ref10">10</xref>). The final report is given in line with the Guidelines for Reporting Reliability and Agreement Studies (GRRAS) (<xref ref-type="bibr" rid="ref11">11</xref>).</p>
<sec id="sec7">
<label>2.1.</label>
<title>Rater selection</title>
<p>Each investigator (SM, SR, IM, and VY) selected one independent rater based on the following criteria (to the best of each investigator&#x2019;s knowledge):</p>
<list list-type="roman-lower">
<list-item>
<p>Knowledge of research methodology;</p>
</list-item>
<list-item>
<p>Potential and/or demonstrated past interest in conducting systematic reviews of clinical trials;</p>
</list-item>
<list-item>
<p>Independent from each other and from the investigators (e.g., no joint publication listed in PubMed or other known prior academic collaboration);</p>
</list-item>
<list-item>
<p>Positive response to the written invitation for participation as rater.</p>
</list-item>
</list>
<p>From the potential number of raters contacted, the first raters who agreed to participate were selected. Hence, a total of four independent raters participated in this study. Each investigator (SR, IM, and VY) revealed the identity of their chosen rater to the principal investigator (SM) only and remained unaware of each other&#x2019;s rater selection until all ratings had been completed.</p>
<p>The number of raters was determined in accordance with a similar study to assess the inter-rater reliability of the CQS-1, published elsewhere (<xref ref-type="bibr" rid="ref7">7</xref>). Rater selection was quasi-random; that is, although no selection according to a random sequence was conducted, each rater&#x2019;s acceptance to participate was left to chance. Raters were free to accept or decline a once-off written invitation without any further effort by the investigators to secure study participation.</p>
</sec>
<sec id="sec8">
<label>2.2.</label>
<title>Rater blinding</title>
<p>In order to assure rater independence, no rater interaction took place during the rating process, thus avoiding any interaction effect on the results. The raters remained unaware of each other&#x2019;s participation in this study until all rating had been completed. However, in order to investigate the use of the CQS-2 under conditions as close as possible to the practical routine of trial appraisal, the raters were not blinded to the references of the trial reports, the author names and affiliations, nor to acknowledgements and funding sources. In addition, to obtain raters&#x2019; informed consent regarding their participation in this study, they received information about the full content of the study protocol. Hence, each rater was aware that their judgment was compared with those of other raters.</p>
</sec>
<sec id="sec9">
<label>2.3.</label>
<title>Sample size calculation</title>
<p>The number of required trial reports was calculated based on a minimum expected agreement between raters of 70%, and a 95% confidence interval (CI) of 15%, using the appropriate formula for sample size calculation: N = 1/E<sup>2</sup> (with N = number of required articles and E = confidence interval) (<xref ref-type="bibr" rid="ref12">12</xref>). In line with the applied sample size calculation method, a minimum number of 44 (rounded to 45) required trial reports were determined.</p>
</sec>
<sec id="sec10">
<label>2.4.</label>
<title>Trial report selection</title>
<p>All 45 trial reports were selected from PubMed. The references are listed in <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>/Section 1. The database was searched by the principal investigator (SM) using the search term &#x201C;prospective AND clinical AND controlled AND trial&#x201D; with the set limits: &#x201C;Abstract,&#x201D; &#x201C;Free full text&#x201D; [Text availability], &#x201C;Clinical trial&#x201D; [Article type], &#x201C;From 2022/1/1 to 2022/05/31&#x201D; [Publication date] and &#x201C;Best match&#x201D; [Display options]. Citation abstracts were checked whether they described a prospective, clinical, controlled trial, published in the English language. Trials were quasi-randomly selected by choosing the first 45 relevant citations from the resulting search list (trial protocols or trials in publication languages other than English were not included).</p>
</sec>
<sec id="sec11">
<label>2.5.</label>
<title>Trial rating process</title>
<p>The raters had no extensive expertise in the conduct of systematic reviews of randomized controlled trials. One rater was an epidemiologist and statistician with 8 years&#x2019; experience; two were dentists employed in academic institutions with 2&#x2013;3 years of work experience (one with 2 years&#x2019; experience in bias risk assessment), and one was a statistician with 25 years of experience and experience in bias risk assessment but not in the use of trials appraisal tools during systematic reviews.</p>
<p>The rater&#x2019;s content knowledge of the trials was not assessed. However, due to the quasi-random nature of the trial selection, it was assumed to be slight. No calibration or training in using both CQS versions was carried out. All raters received the study protocol (<xref ref-type="bibr" rid="ref10">10</xref>) for information about how to apply the CQS-1 and 2 only.</p>
<p>From the principal investigator (SM), each rater received a download link for the 45 trial reports via email and a MS Excel assessment template for both CQS versions was prepared in line with published specifications for each appraisal method (<xref ref-type="bibr" rid="ref7">7</xref>, <xref ref-type="bibr" rid="ref9">9</xref>). Each rater received only one template at a time in a random sequence for CQS-1 and CQS-2 rating. The random sequence (<xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>/Section 2) was generated using block randomization (Block size = 2) out of a total of eight rating events. Raters entered their rating results into the template and sent these back to the principal investigator via email. Each rater received their next template 2 weeks after submission of the completed previous template.</p>
</sec>
<sec id="sec12">
<label>2.6.</label>
<title>The composite quality score</title>
<p>The CQS includes: (i) binary trial report rating per appraisal criterion (Scores: 0 = invalid/falsified, 1 = corroborated); (ii) multiplication of individual rating scores to an overall appraisal score, and (iii) identification of invalid/falsified trial reports based on a zero overall appraisal score.</p>
<sec id="sec13">
<label>2.6.1.</label>
<title>CQS-1</title>
<p>(a) Systematic error (randomization)</p>
<p>Criterion I: &#x201C;Randomization&#x201D; for allocation to treatment groups is in some form reported in the text (Yes = 1/No = 0);</p>
<p>Criterion II: Concealing of the random allocation is in some form reported in the text (Yes = 1/No = 0).</p>
<p>(b) Random error (sample size)</p>
<p>Criterion III: The sample size of any particular treatment group reported in the trial report is not less than N = 200 (Yes = 1/No = 0).</p>
<p>The minimum sample size limit (N) was calculated using the formula: N = {([P1 &#x00D7; (100 &#x2212; P1)] + [P2 &#x00D7; (100 &#x2212; P2)])/(P2 &#x2212; P1)<sup>2</sup>} &#x00D7; f(&#x03B1;,&#x03B2;) (<xref ref-type="bibr" rid="ref13">13</xref>) and was based on the assumption that the difference in intervention effect between study groups (P1&#x2013;P2) is not less than 10%, with <italic>&#x03B1;</italic> = 5% and <italic>&#x03B2;</italic> = 20%, that is: f(&#x03B1;,&#x03B2;) = 7.9 (<xref ref-type="bibr" rid="ref14">14</xref>).</p>
</sec>
<sec id="sec14">
<label>2.6.2.</label>
<title>CQS-2</title>
<p>The CQS-2 is an update of the CQS-1 and based on a systematic review with meta-analysis of meta-epidemiological study evidence, concerning the lack of trial design characteristics associated with over- or under-estimation of the correct effect estimate due to systematic error alone (<xref ref-type="bibr" rid="ref9">9</xref>). In contrast to the CQS-1, the CQS-2 does not include a category for random error. The following criteria were set:</p>
<p>Criterion I: &#x201C;Randomization&#x201D; for allocation to treatment groups is in some form reported in the text (Yes = 1/No = 0);</p>
<p>Criterion II:</p>
<list list-type="roman-lower">
<list-item>
<p>Keeping the random allocation sequence in a locked computer file; and</p>
</list-item>
<list-item>
<p>Translation of the sequence into identical, coded, serially administered containers and/or sealed, opaque envelopes; and</p>
</list-item>
<list-item>
<p>Reassurance that the person who generated the sequence did not administer it.</p>
</list-item>
</list>
<p>are in some form reported in the text (Yes = 1/No = 0);</p>
<p>Criterion III: Double-blinding or the blinding of at least two out of the three groups: trial participants, trial personnel, and trial outcome assessors in some form reported in the text (Yes = 1/No = 0); and.</p>
<p>Criterion IV: The sample size of any particular treatment group reported in the trial is not less than N = 100 (Yes = 1/No = 0).</p>
</sec>
</sec>
<sec id="sec15">
<label>2.7.</label>
<title>Statistical analysis</title>
<p>The inter-rater reliabilities for the overall appraisal score of the CQS-1 and the CQS-2 were established by use of the Brennan-Prediger coefficient (BPC) (<xref ref-type="bibr" rid="ref12">12</xref>). This BPC is given by the ratio (p<sub>a</sub> &#x2212; 1/q)/(1&#x2013;1/q), with p<sub>a</sub> being the percent agreement and q the number of nominal categories in the rating scale. As in a previous study (<xref ref-type="bibr" rid="ref7">7</xref>), this study did not use Cohen&#x2019;s Kappa for inter-rater reliability analysis. Cohen&#x2019;s Kappa is still the most used agreement measure, mainly due to its correction of agreement expected merely by play of chance. However, it is affected by a paradox that returns biased estimates of the statistic itself (situations where high strength of inter-rater agreement actually produce low values for Kappa). This paradox is generated, because marginal values are not independent from the prevalence of the subject under study and this causes an imbalance in case distribution, resulting in lower kappa values. Hence, Cohen&#x2019;s Kappa is increasingly being replaced by several newer coefficients, such as the BPC that does not suffer from this shortcoming since it ignores the marginal values (<xref ref-type="bibr" rid="ref12">12</xref>).</p>
<p>The BPCs of both CQS versions were compared using the two-sample <italic>z</italic>-test. All data analyses were carried out using SAS statistical software (<xref ref-type="bibr" rid="ref15">15</xref>). A 5% significance level was used.</p>
<p>During secondary analysis, the BPC for each single criterion and each corroboration level for both CQS versions was established. The corroboration levels indicate the number of consecutive criteria a trial has complied with (e.g., level C2 indicates Criterion I and II; level C3 indicates Criterion I, II and III, etc.). After a criterion has been rated with a 0-score, the corroboration level remains the same, even if a following criterion is rated with a 1-score, for example Corroboration level C2: Criterion I and II = 1-score, Criterion III = 0-score, Criterion IV = 1-score (<xref ref-type="bibr" rid="ref5">5</xref>).</p>
</sec>
</sec>
<sec sec-type="results" id="sec16">
<label>3.</label>
<title>Results</title>
<p>All four selected raters (FK, KV, RS, and US) completed the rating of all 45 trials with both CQS versions, thus completing a total of 360 evaluations. The rated trials originated from 16 different clinical specialities of which non represented more than 25% of the trials and thus assured a relative even distribution among a variety of medical fields. Most trials were related to surgery (24.4%), followed by internal medicine (15.6%) and dentistry (13.3%). All other 13 clinical specialities contributed less than 10% of the rated trials, each (<xref rid="tab1" ref-type="table">Table 1</xref>). The resulting rating data are presented in <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>/Section 3.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Characteristics of rated trials.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Clinical specialty</th>
<th align="center" valign="top">No.</th>
<th align="center" valign="top">%</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Anesthesiology</td>
<td align="center" valign="top">4</td>
<td align="center" valign="top">9.0</td>
</tr>
<tr>
<td align="left" valign="top">Cardiology</td>
<td align="center" valign="top">4</td>
<td align="center" valign="top">9.0</td>
</tr>
<tr>
<td align="left" valign="top">Clinical immunology</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">2.2</td>
</tr>
<tr>
<td align="left" valign="top">Clinical nutrition</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">2.2</td>
</tr>
<tr>
<td align="left" valign="top">Dentistry</td>
<td align="center" valign="top">6</td>
<td align="center" valign="top">13.3</td>
</tr>
<tr>
<td align="left" valign="top">Dermatology</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">2.2</td>
</tr>
<tr>
<td align="left" valign="top">Gynecology</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">2.2</td>
</tr>
<tr>
<td align="left" valign="top">Internal medicine</td>
<td align="center" valign="top">7</td>
<td align="center" valign="top">15.6</td>
</tr>
<tr>
<td align="left" valign="top">Neurology</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">2.2</td>
</tr>
<tr>
<td align="left" valign="top">Obstetrics</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">2.2</td>
</tr>
<tr>
<td align="left" valign="top">Oncology</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">2.2</td>
</tr>
<tr>
<td align="left" valign="top">Ophthalmology</td>
<td align="center" valign="top">3</td>
<td align="center" valign="top">6.7</td>
</tr>
<tr>
<td align="left" valign="top">Psychotherapy</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">2.2</td>
</tr>
<tr>
<td align="left" valign="top">Reproductive science</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">2.2</td>
</tr>
<tr>
<td align="left" valign="top">Surgery</td>
<td align="center" valign="top">11</td>
<td align="center" valign="top">24.4</td>
</tr>
<tr>
<td align="left" valign="top">Urology</td>
<td align="center" valign="top">1</td>
<td align="center" valign="top">2.2</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The BPC for the CQS-1 was 0.85 (95% CI: 0.64&#x2013;1.00) and for the CQS-2 it was 1.00 (95% CI: 0.94&#x2013;1.00). The difference was not statistically significant (<italic>p</italic> = 0.17) and the null hypothesis was accepted. The BPCs for each criterion and each corroboration level are shown in <xref rid="tab2" ref-type="table">Table 2</xref>: For the CQS-1, the BPC for criterion III was the highest (0.86; 95% CI: 0.73&#x2013;0.99) followed by criterion I (0.71; 95% CI: 0.12&#x2013;1.00) and criterion II (0.29; 95% CI: 0.00&#x2013;0.59). For the CQS-2, the highest BPC value was established for criterion I (0.89; 95% CI: 0.70&#x2013;1.00) followed by criterion IV (0.87; 95% CI: 0.72&#x2013;1.00), criterion II (0.69; 95% CI: 0.38&#x2013;1.00) and criterion III (0.54, 95% CI: 0.32&#x2013;0.76). The results for criterion III of the CQS-1 and criteria I and IV of the CQS-2 reflected an &#x2018;almost perfect&#x2019; inter-rater agreement with their upper confidence levels even reaching the maximum value 1.00. The coefficient for the single criteria concerning random allocation, allocation concealment and sample size limit was higher for the CQS-2 than for the CQS-1. However, these differences were not statistically significant (<xref rid="tab3" ref-type="table">Table 3</xref>).</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Brennan-Prediger coefficients with 95% Confidence interval (CI) of the two CQS versions.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th/>
<th align="center" valign="top">Brennan-Prediger coefficient</th>
<th align="center" valign="top">95% CI</th>
<th align="left" valign="top">Strength of inter-rater agreement according to the Landis/Koch Kappa&#x2019;s Benchmark Scale&#x002A;</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="bottom" colspan="4">Single criterion/CQS-1</td>
</tr>
<tr>
<td align="left" valign="bottom">Criterion I &#x2013; Random allocation</td>
<td align="center" valign="bottom">0.71</td>
<td align="center" valign="bottom">0.12&#x2013;1.00</td>
<td align="left" valign="top">Substantial</td>
</tr>
<tr>
<td align="left" valign="bottom">Criterion II &#x2013; Allocation concealment</td>
<td align="center" valign="bottom">0.29</td>
<td align="center" valign="bottom">0.00&#x2013;0.59</td>
<td align="left" valign="top">Fair</td>
</tr>
<tr>
<td align="left" valign="bottom">Criterion III &#x2013; Sample size limit</td>
<td align="center" valign="bottom">0.86</td>
<td align="center" valign="bottom">0.73&#x2013;0.99</td>
<td align="left" valign="top">Almost perfect</td>
</tr>
<tr>
<td align="left" valign="bottom" colspan="4">Single criterion/CQS-2</td>
</tr>
<tr>
<td align="left" valign="bottom">Criterion I &#x2013; Random allocation</td>
<td align="center" valign="bottom">0.89</td>
<td align="center" valign="bottom">0.70&#x2013;1.00</td>
<td align="left" valign="top">Almost perfect</td>
</tr>
<tr>
<td align="left" valign="bottom">Criterion II &#x2013; Allocation concealment</td>
<td align="center" valign="bottom">0.69</td>
<td align="center" valign="bottom">0.38&#x2013;1.00</td>
<td align="left" valign="top">Substantial</td>
</tr>
<tr>
<td align="left" valign="bottom">Criterion III &#x2013; Double blinding</td>
<td align="center" valign="bottom">0.54</td>
<td align="center" valign="bottom">0.32&#x2013;0.76</td>
<td align="left" valign="top">Moderate</td>
</tr>
<tr>
<td align="left" valign="bottom">Criterion IV &#x2013; Sample size limit</td>
<td align="center" valign="bottom">0.87</td>
<td align="center" valign="bottom">0.72&#x2013;1.00</td>
<td align="left" valign="top">Almost perfect</td>
</tr>
<tr>
<td align="left" valign="bottom" colspan="4">Corroboration levels/CQS-1</td>
</tr>
<tr>
<td align="left" valign="bottom">C1: Criterion I</td>
<td align="center" valign="bottom">0.71</td>
<td align="center" valign="bottom">0.12&#x2013;1.00</td>
<td align="left" valign="top">Substantial</td>
</tr>
<tr>
<td align="left" valign="bottom">C2: Criterion I + II</td>
<td align="center" valign="bottom">0.24</td>
<td align="center" valign="bottom">0.00&#x2013;0.61</td>
<td align="left" valign="top">Fair</td>
</tr>
<tr>
<td align="left" valign="bottom">C3: Criterion I + II + III</td>
<td align="center" valign="bottom">0.85</td>
<td align="center" valign="bottom">0.64&#x2013;1.00</td>
<td align="left" valign="top">Almost perfect</td>
</tr>
<tr>
<td align="left" valign="bottom" colspan="4">Corroboration levels/CQS-2</td>
</tr>
<tr>
<td align="left" valign="bottom">C1: Criterion I</td>
<td align="center" valign="bottom">0.89</td>
<td align="center" valign="bottom">0.70&#x2013;1.00</td>
<td align="left" valign="top">Almost perfect</td>
</tr>
<tr>
<td align="left" valign="bottom">C2: Criterion I + II</td>
<td align="center" valign="bottom">0.69</td>
<td align="center" valign="bottom">0.38&#x2013;1.00</td>
<td align="left" valign="top">Substantial</td>
</tr>
<tr>
<td align="left" valign="bottom">C3: Criterion I + II + III</td>
<td align="center" valign="bottom">0.81</td>
<td align="center" valign="bottom">0.57&#x2013;1.00</td>
<td align="left" valign="top">Almost perfect</td>
</tr>
<tr>
<td align="left" valign="bottom">C4: Criterion I + II + III + IV</td>
<td align="center" valign="bottom">1.00</td>
<td align="center" valign="bottom">0.94&#x2013;1.00</td>
<td align="left" valign="top">Almost perfect</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>&#x002A;</italic>Poor: &#x003C;0; Slight: 0&#x2013;0.20; Fair: 0.21&#x2013;0.40; Moderate: 0.41&#x2013;0.60; Substantial: 061&#x2013;0.80; Almost perfect: 0.81&#x2013;1.00 (<xref ref-type="bibr" rid="ref8">8</xref>).</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Differences in the Brennan-Prediger coefficients between the components of the different rating tools.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Appraisal category</th>
<th align="left" valign="top">CQS-1</th>
<th align="left" valign="top">CQS-2</th>
<th align="center" valign="top"><italic>p</italic>-value</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Random allocation</td>
<td align="left" valign="top">Criterion I</td>
<td align="left" valign="top">Criterion I</td>
<td align="center" valign="top">0.57</td>
</tr>
<tr>
<td align="left" valign="top">Allocation concealment</td>
<td align="left" valign="top">Criterion II</td>
<td align="left" valign="top">Criterion II</td>
<td align="center" valign="top">0.071</td>
</tr>
<tr>
<td align="left" valign="top">Sample size limit</td>
<td align="left" valign="top">Criterion III</td>
<td align="left" valign="top">Criterion IV</td>
<td align="center" valign="top">0.92</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The BPC for most corroboration levels suggested &#x201C;substantial&#x201D; or &#x201C;almost perfect&#x201D; strength of inter-rater agreement for both CQS versions, particularly for the CQS-2; except for level C2 of CQS-1 (BPC 0.24; 95% CI: 0.00&#x2013;0.61), which indicated &#x201C;fair&#x201D; strength of inter-rater agreement only. Notwithstanding, the difference between the coefficient value to that of the CQS-2 was not statistically significant (<italic>p</italic> = 0.069).</p>
</sec>
<sec sec-type="discussions" id="sec17">
<label>4.</label>
<title>Discussion</title>
<sec id="sec18">
<label>4.1.</label>
<title>Study results</title>
<p>The results of this study show that the CQS-2 is associated with a very high inter-rater reliability (BPC 1.00; 95% CI: 0.94&#x2013;1.00), which did not statistically significantly differ from that of the CQS-1 (<italic>p</italic> = 0.17). In addition, this study replicated the very high inter-rater reliability for the CQS-1 (BPC 0.85, 95% CI: 0.64&#x2013;1.00), thus confirming previous results (BPC 0.95, 95% CI: 0.87&#x2013;1.00) (<xref ref-type="bibr" rid="ref7">7</xref>).</p>
<p>These results compare favorably to that of previously established inter-rater reliabilities of other evidence appraisal tools: the Jadad scale (BPC 0.70; 95% CI: 0.58&#x2013;0.82) (<xref ref-type="bibr" rid="ref7">7</xref>), the Grading of Recommendations, Assessment, Development and Evaluation (GRADE) approach (Intraclass correlation coefficient 0.84; 95% CI: 0.78&#x2013;0.89) (<xref ref-type="bibr" rid="ref16">16</xref>) and the second version of Cochrane&#x2019;s Risk of Bias tool (for overall judgment: Fleiss&#x2019;s Kappa 0.16; 95% CI: 0.08&#x2013;0.24) (<xref ref-type="bibr" rid="ref17">17</xref>).</p>
<p>The results of this study also show that adding one further criterion (the new criterion III) and amending two existing criteria (new criterion II and IV) to the CQS (<xref ref-type="bibr" rid="ref9">9</xref>) did not negatively affect its inter-rater reliability. The BPC for the added CQS-2 criterion III, regarding double-blinding, was found to be 0.54 (95%CI: 0.32&#x2013;0.76) only. However, this result still compares favorably to that of previously established results for the bias risk domains &#x201C;operator blinding&#x201D; and &#x201C;evaluator blinding&#x201D; of the RoB-1 tool [BPC 0.03; 95% CI: &#x2212;0.22 to 0.28 and 0.27; 95%CI: &#x2212;0.08 to 0.62, respectively (<xref ref-type="bibr" rid="ref7">7</xref>)].</p>
<p>It has been observed from previous data (<xref ref-type="bibr" rid="ref5">5</xref>, <xref ref-type="bibr" rid="ref7">7</xref>) that higher corroboration levels were associated with higher Brennan-Prediger coefficient values. The higher the corroboration levels, the more single binary (0/1) scores from single appraisal criteria are multiplied into an overall trial appraisal score. A higher number of multiplied single scores increase the chance of multiplication by a single 0-score, which subsequently would render the overall score as zero. This higher chance of an overall 0-score increases the chance that an independent rater will agree on a 0-score in the overall appraisal of a trial, even when they differ in the rating of a single criterion. Such possible mechanism may explain the consistently very high inter-rater reliability of the CQS. It may also indicate that, in that way, rating errors between individual raters are canceled out and thus a high inter-rater reliability is retained.</p>
<p>However, in this study, a consistent pattern of increasing Brennan-Prediger coefficient per corroboration level was not observed. Both CQS versions showed a decrease in the coefficient at corroboration level C2 (<xref rid="tab2" ref-type="table">Table 2</xref>). Such a decrease may be explained on basis that the coefficient for criterion I was high in both CQS versions and subsequently reduced at C2 level by combination with a lower coefficient for criterion II. The difference between the current results and results from a previous study (<xref ref-type="bibr" rid="ref7">7</xref>) may have been due to variations in the characteristics of the rated trials. In a previous study by Mickenautsch et al. only trials related to restorative dentistry were rated. Only a small number of these trials reported the application of allocation concealment (CQS-1/criterion II). It thus may have been easier for all raters to agree on a 0-score for this criterion, resulting in a higher Brennan-Prediger coefficient (<xref ref-type="bibr" rid="ref7">7</xref>). In this study, clinical trials from various medical fields were included instead. In these trials, allocation concealment was more frequently applied but reported in different ways. This may have made trial appraisal more challenging and thus negatively affected the inter-rater reliability.</p>
<p>Notwithstanding such observed differences, the Brennan-Prediger coefficient and its lower confidence limit for the CQS-2 increased steadily from level C2 upwards to corroboration level C4 (i.e., the overall CQS-2 score): BPC C2: 0.69 (95% CI: 0.38&#x2013;1.00); C3: BPC 0.81 (95% CI: 0.57&#x2013;1.00) and C4: BPC 1.00 (95% CI: 0.94&#x2013;1.00) (<xref rid="tab2" ref-type="table">Table 2</xref>).</p>
<p>It was further observed that, although the difference was not statically significant (<italic>p</italic> = 0.071), the Brennan-Prediger coefficient for criterion II for the CQS-2 was higher than that of the CQS-1 (<xref rid="tab2" ref-type="table">Table 2</xref>), despite the former having a more restrictive nature. However, it is possible that, because of the higher restriction level for this criterion, it was easier for raters to agree on a 0-score, thus resulting in a higher Brennan-Prediger coefficient. Our data show that raters agreed far more often on a 0-score for criterion II using the CQS-2 (with no agreement for a 1-score) than with the CQS-1. There was also overall less agreement for both 1- and 0-scores combined when using criterion II with the CQS-1 than with the CQS-2 (<xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>).</p>
</sec>
<sec id="sec19">
<label>4.2.</label>
<title>Study limitations and recommendations for further research</title>
<p>The fact that none of the 45 trials received an overall 1-score by any of the four raters may indicate that a too low sample size may have been calculated. A higher sample size may have resulted in at least a few overall 1-score judgment by some of the raters and, thus, a higher precision of the study results. Further inter-rater reliability studies may include a larger number of trials based on a higher expected agreement percentage than was used in this study (70%). Also, the quasi-random sampling method for trials used in this study caused that a heterogeneous range of different medical fields was included, resulting in an overall slight rater content knowledge only. Usually, raters who participate in a systematic review of trials are experts in the particular field of medicine study and appraise trials of homogeneous content that are related to a specific clinical review question. Such high content knowledge would most likely assist in a higher strength of inter-rater agreement than observed in this study.</p>
<p>The CQS-2 as a trial appraisal tool is still under development. Trials from systematic reviews that have applied the 2nd version of Cochrane&#x2019;s RoB tool may be re-appraised using the CQS-2 in order to establish whether the direction and magnitude of any pooled effect estimates remain the same. Based on the results of these further investigations, the CQS-2 may be piloted as part of the regular, systematic review methodology for the appraisal of prospective, controlled clinical therapy trials.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="sec20">
<label>5.</label>
<title>Conclusion</title>
<p>This study shows that the CQS-2 is associated with a very high inter-rater reliability, which did not statistically significantly differ from that of the previous CQS-1. The promising results of this study warrant further investigation into the applicability of the CQS-2 as an appraisal tool for prospective controlled clinical therapy trials in systematic reviews.</p>
</sec>
<sec sec-type="data-availability" id="sec21">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="sec22">
<title>Ethics statement</title>
<p>Ethical review and approval was not required for the study on human participants in accordance with the local legislation and institutional requirements. Written informed consent from the participants was obtained to take part in this study.</p>
</sec>
<sec id="sec23">
<title>Author contributions</title>
<p>SM contributed to conception and design of the study and wrote the first draft of the manuscript. US, RS, FK-D, and KV contributed to the investigation. All authors commented, improved the manuscript, read, and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="COI-statement" id="sec24">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="sec100" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<ack>
<p>The authors thank Petra Gaylard from DMSA for her valuable advice concerning data statistics and for conducting the data analysis.</p>
</ack>
<sec sec-type="supplementary-material" id="sec25">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fmed.2023.1201517/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fmed.2023.1201517/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.xls" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><label>1.</label> <citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mickenautsch</surname><given-names>S</given-names></name></person-group>. <article-title>Is the deductive falsification approach a better basis for clinical trial appraisal?</article-title> <source>Rev Recent Clin Trials</source>. (<year>2019</year>) <volume>14</volume>:<fpage>224</fpage>&#x2013;<lpage>8</lpage>. doi: <pub-id pub-id-type="doi">10.2174/1574887114666190313170400</pub-id>, PMID: <pub-id pub-id-type="pmid">30868960</pub-id></citation></ref>
<ref id="ref2"><label>2.</label> <citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jadad</surname><given-names>AR</given-names></name> <name><surname>Moore</surname><given-names>RA</given-names></name> <name><surname>Carroll</surname><given-names>D</given-names></name> <name><surname>Jenkinson</surname><given-names>C</given-names></name> <name><surname>Reynolds</surname><given-names>DJ</given-names></name> <name><surname>Gavaghan</surname><given-names>DJ</given-names></name> <etal/></person-group>. <article-title>Assessing the quality of reports of randomized clinical trials: is blinding necessary?</article-title> <source>Control Clin Trials</source>. (<year>1996</year>) <volume>17</volume>:<fpage>1</fpage>&#x2013;<lpage>12</lpage>. doi: <pub-id pub-id-type="doi">10.1016/0197-2456(95)00134-4</pub-id>, PMID: <pub-id pub-id-type="pmid">8721797</pub-id></citation></ref>
<ref id="ref3"><label>3.</label> <citation citation-type="journal"><person-group person-group-type="author"><name><surname>Higgins</surname><given-names>JP</given-names></name> <name><surname>Altman</surname><given-names>DG</given-names></name> <name><surname>G&#x00F8;tzsche</surname><given-names>PC</given-names></name> <name><surname>J&#x00FC;ni</surname><given-names>P</given-names></name> <name><surname>Moher</surname><given-names>D</given-names></name> <name><surname>Oxman</surname><given-names>AD</given-names></name> <etal/></person-group>. <article-title>The Cochrane collaboration's tool for assessing risk of bias in randomised trials</article-title>. <source>BMJ</source>. (<year>2011</year>) <volume>343</volume>:<fpage>d5928</fpage>. doi: <pub-id pub-id-type="doi">10.1136/bmj.d5928</pub-id>, PMID: <pub-id pub-id-type="pmid">22008217</pub-id></citation></ref>
<ref id="ref4"><label>4.</label> <citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sterne</surname><given-names>JAC</given-names></name> <name><surname>Savovi&#x0107;</surname><given-names>J</given-names></name> <name><surname>Page</surname><given-names>MJ</given-names></name> <name><surname>Elbers</surname><given-names>RG</given-names></name> <name><surname>Blencowe</surname><given-names>NS</given-names></name> <name><surname>Boutron</surname><given-names>I</given-names></name> <etal/></person-group>. <article-title>RoB 2: a revised tool for assessing risk of bias in randomised trials</article-title>. <source>BMJ</source>. (<year>2019</year>) <volume>366</volume>:<fpage>l4898</fpage>. doi: <pub-id pub-id-type="doi">10.1136/bmj.l4898</pub-id>, PMID: <pub-id pub-id-type="pmid">31462531</pub-id></citation></ref>
<ref id="ref5"><label>5.</label> <citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mickenautsch</surname><given-names>S</given-names></name> <name><surname>Rupf</surname><given-names>S</given-names></name> <name><surname>Mileti&#x0107;</surname><given-names>I</given-names></name> <name><surname>Yengopal</surname><given-names>V</given-names></name></person-group>. <article-title>The composite quality score (CQS) as appraisal tool for prospective, controlled, clinical therapy trials: rationale and current evidence</article-title>. <source>Rev Recent Clin Trials</source>. (<year>2023</year>) <volume>18</volume>:<fpage>28</fpage>&#x2013;<lpage>33</lpage>. doi: <pub-id pub-id-type="doi">10.2174/1574887118666230104152245</pub-id>, PMID: <pub-id pub-id-type="pmid">36600618</pub-id></citation></ref>
<ref id="ref6"><label>6.</label> <citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mickenautsch</surname><given-names>S</given-names></name></person-group>. <article-title>Are most of the published clinical trial results in restorative dentistry invalid? An empirical investigation</article-title>. <source>Rev Recent Clin Trials</source>. (<year>2020</year>) <volume>15</volume>:<fpage>122</fpage>&#x2013;<lpage>30</lpage>. doi: <pub-id pub-id-type="doi">10.2174/1574887115666200421110732</pub-id>, PMID: <pub-id pub-id-type="pmid">32316900</pub-id></citation></ref>
<ref id="ref7"><label>7.</label> <citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mickenautsch</surname><given-names>S</given-names></name> <name><surname>Mileti&#x0107;</surname><given-names>I</given-names></name> <name><surname>Rupf</surname><given-names>S</given-names></name> <name><surname>Renteria</surname><given-names>J</given-names></name> <name><surname>G&#x00F6;stemeyer</surname><given-names>G</given-names></name></person-group>. <article-title>The composite quality score (CQS) as a trial appraisal tool: inter-rater reliability and rating time</article-title>. <source>Clin Oral Investig</source>. (<year>2021</year>) <volume>25</volume>:<fpage>6015</fpage>&#x2013;<lpage>23</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s00784-021-04099-w</pub-id>, PMID: <pub-id pub-id-type="pmid">34379191</pub-id></citation></ref>
<ref id="ref8"><label>8.</label> <citation citation-type="journal"><person-group person-group-type="author"><name><surname>Landis</surname><given-names>JR</given-names></name> <name><surname>Koch</surname><given-names>GG</given-names></name></person-group>. <article-title>The measurement of observer agreement for categorical data</article-title>. <source>Biometrics</source>. (<year>1977</year>) <volume>33</volume>:<fpage>159</fpage>&#x2013;<lpage>74</lpage>. doi: <pub-id pub-id-type="doi">10.2307/2529310</pub-id>, PMID: <pub-id pub-id-type="pmid">843571</pub-id></citation></ref>
<ref id="ref9"><label>9.</label> <citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mickenautsch</surname><given-names>S</given-names></name> <name><surname>Rupf</surname><given-names>S</given-names></name> <name><surname>Mileti&#x0107;</surname><given-names>I</given-names></name> <name><surname>Yengopal</surname><given-names>V</given-names></name></person-group>. <article-title>Extension of the composite quality score (CQS) as an appraisal tool for prospective, controlled clinical therapy trials</article-title>. <source>PLoS One</source>. (<year>2022</year>) <volume>17</volume>:<fpage>e0279645</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0279645</pub-id>, PMID: <pub-id pub-id-type="pmid">36584067</pub-id></citation></ref>
<ref id="ref10"><label>10.</label> <citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mickenautsch</surname><given-names>S</given-names></name> <name><surname>Rupf</surname><given-names>S</given-names></name> <name><surname>Mileti&#x0107;</surname><given-names>I</given-names></name> <name><surname>Yengopal</surname><given-names>V</given-names></name></person-group>. <article-title>Inter-rater reliability of the extended composite quality score (CQS-2) &#x2013; protocol</article-title>. <source>Res. Square</source>. (<year>2022</year>). doi: <pub-id pub-id-type="doi">10.21203/rs.3.rs-1763870/v1</pub-id></citation></ref>
<ref id="ref11"><label>11.</label> <citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kottner</surname><given-names>J</given-names></name> <name><surname>Audig&#x00E9;</surname><given-names>L</given-names></name> <name><surname>Brorson</surname><given-names>S</given-names></name> <name><surname>Donner</surname><given-names>A</given-names></name> <name><surname>Gajewski</surname><given-names>BJ</given-names></name> <name><surname>Hr&#x00F3;bjartsson</surname><given-names>A</given-names></name> <etal/></person-group>. <article-title>Guidelines for reporting reliability and agreement studies (GRRAS) were proposed</article-title>. <source>J Clin Epidemiol</source>. (<year>2011</year>) <volume>64</volume>:<fpage>96</fpage>&#x2013;<lpage>106</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jclinepi.2010.03.002</pub-id>, PMID: <pub-id pub-id-type="pmid">21130355</pub-id></citation></ref>
<ref id="ref12"><label>12.</label> <citation citation-type="book"><person-group person-group-type="author"><name><surname>Gwet</surname><given-names>KL</given-names></name></person-group>. <source>Handbook of inter-rater reliability</source>. <edition>2nd</edition> ed. <publisher-loc>Gainsburg, MD</publisher-loc>: <publisher-name>Advanced analytics, LLC</publisher-name> (2010).</citation></ref>
<ref id="ref13"><label>13.</label> <citation citation-type="book"><person-group person-group-type="author"><name><surname>Pocock</surname><given-names>SJ</given-names></name></person-group>. <source>Clinical trials: a practical approach</source>. <publisher-loc>Chichester</publisher-loc>: <publisher-name>Wiley</publisher-name> (<year>1988</year>).</citation></ref>
<ref id="ref14"><label>14.</label> <citation citation-type="book"><person-group person-group-type="author"><collab id="coll1">Geigy</collab></person-group> ed. <source>Scientific tables</source>. <edition>7th</edition> ed. <publisher-loc>Basel</publisher-loc>: <publisher-name>Geigy</publisher-name> (1970).</citation></ref>
<ref id="ref15"><label>15.</label> <citation citation-type="book"><person-group person-group-type="author"><collab id="coll2">SAS Institute Inc</collab></person-group>. <source>SAS software, version 9.4 for windows</source>. <publisher-loc>Cary, NC</publisher-loc>: <publisher-name>SAS Institute Inc.</publisher-name> (<year>2002&#x2013;2012</year>).</citation></ref>
<ref id="ref16"><label>16.</label> <citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mustafa</surname><given-names>RA</given-names></name> <name><surname>Santesso</surname><given-names>N</given-names></name> <name><surname>Brozek</surname><given-names>J</given-names></name> <name><surname>Akl</surname><given-names>EA</given-names></name> <name><surname>Walter</surname><given-names>SD</given-names></name> <name><surname>Norman</surname><given-names>G</given-names></name> <etal/></person-group>. <article-title>The GRADE approach is reproducible in assessing the quality of evidence of quantitative evidence syntheses</article-title>. <source>J Clin Epidemiol</source>. (<year>2013</year>) <volume>66</volume>:<fpage>736</fpage>&#x2013;<lpage>42</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jclinepi.2013.02.004</pub-id>, PMID: <pub-id pub-id-type="pmid">23623694</pub-id></citation></ref>
<ref id="ref17"><label>17.</label> <citation citation-type="journal"><person-group person-group-type="author"><name><surname>Minozzi</surname><given-names>S</given-names></name> <name><surname>Cinquini</surname><given-names>M</given-names></name> <name><surname>Gianola</surname><given-names>S</given-names></name> <name><surname>Gonzalez-Lorenzo</surname><given-names>M</given-names></name> <name><surname>Banzi</surname><given-names>R</given-names></name></person-group>. <article-title>The revised Cochrane risk of bias tool for randomized trials (RoB 2) showed low interrater reliability and challenges in its application</article-title>. <source>J Clin Epidemiol</source>. (<year>2020</year>) <volume>126</volume>:<fpage>37</fpage>&#x2013;<lpage>44</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jclinepi.2020.06.015</pub-id>, PMID: <pub-id pub-id-type="pmid">32562833</pub-id></citation></ref>
</ref-list>
</back>
</article>