<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="brief-report" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Educ.</journal-id>
<journal-title>Frontiers in Education</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Educ.</abbrev-journal-title>
<issn pub-type="epub">2504-284X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">731763</article-id>
<article-id pub-id-type="doi">10.3389/feduc.2021.731763</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Education</subject>
<subj-group>
<subject>Brief Research Report</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>The Position of Distractors in Multiple-Choice Test Items: The Strongest Precede the Weakest</article-title>
<alt-title alt-title-type="left-running-head">Lions et&#x20;al.</alt-title>
<alt-title alt-title-type="right-running-head">Distractors Are Not Uniformly Distributed</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Lions</surname>
<given-names>S&#xe9;verin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1376941/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Monsalve</surname>
<given-names>Carlos</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dartnell</surname>
<given-names>Pablo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/919698/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Godoy</surname>
<given-names>Mar&#xed;a In&#xe9;s</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1197676/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>C&#xf3;rdova</surname>
<given-names>Nora</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jim&#xe9;nez</surname>
<given-names>Daniela</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Blanco</surname>
<given-names>Mar&#xed;a Paz</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ortega</surname>
<given-names>Gabriel</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1490309/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lemari&#xe9;</surname>
<given-names>Julie</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
</contrib>
</contrib-group>
<aff id="aff1">
<label>
<sup>1</sup>
</label>Center for Advanced Research in Education (FB0003), Institute of Education, Universidad de Chile, <addr-line>Santiago</addr-line>, <country>Chile</country>
</aff>
<aff id="aff2">
<label>
<sup>2</sup>
</label>Center for Mathematical Modeling (AFB170001), Universidad de Chile, <addr-line>Santiago</addr-line>, <country>Chile</country>
</aff>
<aff id="aff3">
<label>
<sup>3</sup>
</label>Department of Mathematical Engineering, Universidad de Chile, <addr-line>Santiago</addr-line>, <country>Chile</country>
</aff>
<aff id="aff4">
<label>
<sup>4</sup>
</label>Departamento de Evaluaci&#xf3;n, Medici&#xf3;n y Registro Educacional, Universidad de Chile, <addr-line>Santiago</addr-line>, <country>Chile</country>
</aff>
<aff id="aff5">
<label>
<sup>5</sup>
</label>CLLE (Cognition, Langues, Langage, Ergonomie), UT2J CNRS, University of Toulouse, <addr-line>Toulouse</addr-line>, <country>France</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/468707/overview">Yong Luo</ext-link>, Educational Testing Service, United&#x20;States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/356821/overview">Georgios Sideridis</ext-link>, Harvard Medical School, United&#x20;States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1428675/overview">Duy Pham</ext-link>, Educational Testing Service, United&#x20;States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: S&#xe9;verin Lions, <email>severin.lions@ciae.uchile.cl</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Assessment, Testing and Applied Measurement, a section of the journal Frontiers in Education</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>28</day>
<month>10</month>
<year>2021</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>6</volume>
<elocation-id>731763</elocation-id>
<history>
<date date-type="received">
<day>28</day>
<month>06</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>16</day>
<month>09</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2021 Lions, Monsalve, Dartnell, Godoy, C&#xf3;rdova, Jim&#xe9;nez, Blanco, Ortega and Lemari&#xe9;.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Lions, Monsalve, Dartnell, Godoy, C&#xf3;rdova, Jim&#xe9;nez, Blanco, Ortega and Lemari&#xe9;</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these&#x20;terms.</p>
</license>
</permissions>
<abstract>
<p>Middle bias has been reported for responses to multiple-choice test items used in educational assessment. It has been claimed that this response bias probably occurs because test developers tend to place correct responses among middle options, tests thus presenting a middle-biased distribution of answer keys. However, this response bias could be driven by strong distractors being more frequently located among middle options. In this study, the frequency of responses to a Chilean national examination used to rank students wanting to access higher education was used to categorize distractors based on attractiveness level. The distribution of different distractor types (best distractor, non-functioning distractors&#x2026;) was analyzed across 110 tests of 80&#x20;five-option items administered to assess several disciplines in five consecutive years. Results showed that the strongest distractors were more frequently found among middle options, most commonly at option C. In contrast, the weakest distractors were more frequently found at the last option (E). This pattern did not substantially vary across disciplines or years. Supplementary analyses revealed that a similar position bias for distractors could be observed in tests administered in countries other than Chile. Thus, the location of different types of distractors might provide an alternative explanation for the middle bias reported in literature for tests&#x2019; responses. Implications for test developers, test takers, and researchers in the field are discussed.</p>
</abstract>
<kwd-group>
<kwd>assessment</kwd>
<kwd>educational tests</kwd>
<kwd>multiple-choice</kwd>
<kwd>response placement</kwd>
<kwd>distractors</kwd>
</kwd-group>
<contract-num rid="cn001">3190273</contract-num>
<contract-num rid="cn002">ID16I10090</contract-num>
<contract-sponsor id="cn001">Fondo Nacional de Desarrollo Cient&#xed;fico y Tecnol&#xf3;gico<named-content content-type="fundref-id">10.13039/501100002850</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">Fondo de Fomento al Desarrollo Cient&#xed;fico y Tecnol&#xf3;gico<named-content content-type="fundref-id">10.13039/501100008736</named-content>
</contract-sponsor>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Multiple-choice tests are widely used in educational assessment, students&#x2019; performance on these tests being sometimes highly consequential (<xref ref-type="bibr" rid="B13">Gierl et&#x20;al., 2017</xref>). Thus, it may become critical that tests do provide valid and reliable learning measures (<xref ref-type="bibr" rid="B16">Haladyna and Downing, 2004</xref>). Even if item-writing guidelines have been advanced in literature to help test developers design better multiple-choice instruments (<xref ref-type="bibr" rid="B15">Haladyna and Downing, 1989a</xref>; <xref ref-type="bibr" rid="B17">Haladyna et&#x20;al., 2002</xref>; <xref ref-type="bibr" rid="B19">Haladyna and Rodriguez, 2013</xref>), item-writing flaws are still commonly found, impacting tests&#x2019; psychometric properties, students&#x2019; scores, and even pass-fail outcomes (<xref ref-type="bibr" rid="B10">Downing, 2005</xref>; <xref ref-type="bibr" rid="B27">Tarrant and Ware, 2008</xref>; <xref ref-type="bibr" rid="B1">Ali and Ruit, 2015</xref>).</p>
<p>One rather common test construction flaw is that the placement of correct responses (also called answer keys) across a test is middle-biased, key position providing an unwanted strategic clue to examinees (<xref ref-type="bibr" rid="B23">Metfessel and Sax, 1958</xref>; <xref ref-type="bibr" rid="B18">Haladyna and Downing, 1989b</xref>; <xref ref-type="bibr" rid="B3">Attali and Bar-Hillel, 2003</xref>). Empirical results have confirmed that students do consider option position when taking a test (<xref ref-type="bibr" rid="B7">Carnegie, 2017</xref>) and that students&#x2019; responses themselves present a middle-bias pattern, which can lead to less discriminative items with high accuracy rates when middle-keyed (<xref ref-type="bibr" rid="B3">Attali and Bar-Hillel, 2003</xref>). One recent explanation for students&#x2019; response bias lies in the test developers&#x2019; own middle bias when positioning answer keys (<xref ref-type="bibr" rid="B4">Bar-Hillel, 2015</xref>).</p>
<p>However, it might be distractors, not keys, what really drives middle-biased responses among students. If the strongest distractors were to be more frequently positioned as middle options, examinees would consequently select middle options more frequently than edge options when responding inaccurately (<xref ref-type="bibr" rid="B14">Gustav, 1963</xref>). This would be consistent with the fact that the reported students&#x2019; response bias is sometimes more robust for incorrect responses than for correct ones (see, for example, <xref ref-type="bibr" rid="B3">Attali and Bar-Hillel, 2003</xref>).</p>
<p>Literature has shown that strong distractors&#x2019; position impacts item difficulty (<xref ref-type="bibr" rid="B12">Friel and Johnstone, 1979</xref>; <xref ref-type="bibr" rid="B2">Ambu Saidi and Khamis, 2000</xref>; <xref ref-type="bibr" rid="B22">Kiat et&#x20;al., 2018</xref>; <xref ref-type="bibr" rid="B24">Shin et&#x20;al., 2019</xref>). However, the distribution of strong distractors across a test has not been addressed. In a systematic research synthesis examining test developers&#x2019; practice regarding options placement (Authors, 2021, under review), more than 50 relevant studies were identified, none of them considering strong distractors&#x2019; arrangement. Neither did any of these studies focus on weak distractors&#x2019; placement. Interestingly, however, one study noticed that most of non-selected distractors from a sample of 151&#x20;five-option items were located as last option (<xref ref-type="bibr" rid="B25">Siddiqui, 2018</xref>). Since an unbalanced distribution of strong/weak distractors may provide students with valuable information to reject some options when strategically solving items, studying the overall arrangement of distractors might prove to be enlightening.</p>
<p>This study was conceived to examine the distribution of different types of distractors in multiple-choice tests. Previous studies have shown that many tests present either a middle-keying bias (<xref ref-type="bibr" rid="B3">Attali and Bar-Hillel, 2003</xref>) or an overbalanced distribution of answer keys (<xref ref-type="bibr" rid="B5">Bar-Hillel and Attali, 2002</xref>), suggesting that test developers rarely randomize options order during test assembly. We thus expected results to provide new insights into both test development and item creation processes. Our study was guided by the working hypothesis that when test developers design items, they tend to generate distractors following a plausibility order, which ultimately correlates with distractors&#x2019; placement within the options list, with strong distractors being positioned before weak&#x20;ones.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>Materials and Methods</title>
<sec id="s2-1">
<title>Data Collection</title>
<p>All of the examinees&#x2019; responses to Chilean national examination PSU (Prueba de Selecci&#xf3;n Universitaria) from 2016, 2017, 2018, 2019, and 2020 were gathered. PSU is a paper-and-pencil, high-stakes, standardized examination which students must take to enter most universities in Chile. The assessment comprises two mandatory exams that all students must take (one in mathematics and one in language) and several other optional exams belonging to different domains that students voluntarily take depending on the program they apply to (such as chemistry, history, or physics). Observed tests were from four domains: language, mathematics, science, and history. Final data set included 8,800&#x20;multiple-choice items from 110&#x20;eighty-item&#x20;tests.</p>
<p>Individual responses per item ranged from 1,567 to 66,821, totaling 318,859,763&#x20;single-item responses. All items had five options and were designed and field-tested by DEMRE (Departamento de Evaluaci&#xf3;n, Medici&#xf3;n y Registro Educacional), the Chilean state institution in charge of developing and administering national university admission exams. All participants signed a written informed consent stating that their responses could be used for research purposes.</p>
<p>A second data set, obtained from a previous systematic research synthesis (Authors, 2021, under review), was also used. Data consisted of 421 items (108&#x20;five-option and 313&#x20;four-option items) from 13 item sets (four for five-option and nine for four-option items), obtained from 11 studies. Studies were identified during the selection process of the systematic research synthesis and were included because they provided not only answer keys&#x2019; distribution but also test-takers responses to multiple-choice items for each option position separately, making it possible to identify the different types of distractors. Items from this second data set were from tests used in countries other than Chile.</p>
</sec>
<sec id="s2-2">
<title>Data Analysis</title>
<p>The first set of analyses consisted of examining the distribution of the two most classically studied distractor types: best distractor and non-functioning distractors. The best distractor (also called the most attractive distractor) for each item was defined as the erroneous response most frequently selected by examinees, following previous studies (e.g., <xref ref-type="bibr" rid="B24">Shin et&#x20;al., 2019</xref>). A non-functioning distractor was defined as an erroneous response selected by less than five percent of examinees, as standardly defined in most previous studies (e.g., <xref ref-type="bibr" rid="B28">Tarrant et&#x20;al., 2009</xref>). It is worth mentioning that an item can have various non-functioning distractors but no more than one best distractor. Occasionally, items had no best distractor (because two distractors of one item received the same number of responses) or no non-functioning distractors at all (because all distractors of one item received more than five percent of responses).</p>
<p>For the purposes of the first set of analyses, the best distractor for every single item was identified based on examinees&#x2019; responses. Once identified, its position within the options list (A, B, C, D, E) was registered. This allowed determining best distractor&#x2019;s position at item level. On a second step, at a single-test level, each test taken by examinees was individually inspected to determine the frequency of best distractors at A, B, C, D, and E across all test items. Since not all items of a given test had indeed a best distractor, the absolute frequency of best distractors per position was converted, per test, to a percentage relative to the exact number of items containing an actual best distractor. Finally, a one-way repeated-measures ANOVA was conducted including all tests (regardless of domain and year), with Option Position as within factor (five levels: A, B, C, D, E) and percentage of best distractor&#x2019;s presence (hereinafter called frequency) as dependent variable. The same procedure was implemented for non-functioning distractors.</p>
<p>In a second set of analyses, all distractors were ranked by attractiveness level for every single item, based on response frequency (best distractor &#x3e; distractor 2&#x20;&#x3e; distractor 3&#x20;&#x3e; worst distractor), registering the position of each kind of distractor within the options list (A, B, C, D, E). At test level, distractors&#x2019; frequencies were compared at every position. Since totals varied per test and per positions, raw frequencies were again converted to percentages. For instance, if for a given test distractors were found 60&#x20;times (out of 80) for option A, raw frequencies of best distractor and remaining distractors were converted to percentages relative to a total of 60. Correct answers were excluded from all counts. Once this was completed, five one-way repeated-measures ANOVAs were conducted (one for each of the five option positions), with Distractor Type as within factor (four levels: best distractor, distractor 2, distractor 3, worst distractor) and percentage of occurrence (hereinafter called frequency) as dependent variable. ANOVAs assumptions were inspected and met by all conducted tests. Bonferroni post-hoc tests were conducted and were reported when relevant. Partial eta squared was reported as size effect measure.</p>
<p>Supplementary analyses were implemented to make sure that observed results were robust and generalizable. First, the distributions (percentages) of best and worst distractors were analyzed again, after defining distractors more conservatively, to make sure that observed results were not spurious. At this point, the most frequently selected erroneous response was labeled <italic>best distractor</italic> only when having received at least five percent more responses than the <italic>second-best</italic> distractor (distractor 2), and the least frequently selected erroneous response was labeled <italic>worst distractor</italic> only when having been selected five percent less than the <italic>second-worst</italic> distractor (distractor 3). This was done to confirm that findings were not attributable to the influence of option position on test-takers behavior (this influence being modest, with option position effects being generally smaller than five percent). Second, the distributions of best distractor and non-functioning distractors were analyzed for each tested domain (language, math, science, history) and year of test administration (2016, 2017, 2018, 2019, 2020) separately, to evaluate the generalizability and replicability of findings. Finally, the second data set was used to determine whether tests used in countries other than Chile presented similar distributions of distractors.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<p>A statistically significant difference was observed when comparing the frequency of best distractor between different option positions: F (4,436) &#x3d; 50.267, <italic>p</italic>&#x20;&#x3c; 0.001, &#x19e;<sup>2</sup>p &#x3d; 0.316. Best distractor was found the most frequently at option C and the least frequently at option E (all p<sub>s</sub> &#x2264; 0.004 in post-hoc tests, <xref ref-type="fig" rid="F1">Figure&#x20;1A</xref>). When comparing the frequency of non-functioning distractors across option positions, a significant difference was also observed: F (4,420) &#x3d; 41.598, <italic>p</italic>&#x20;&#x3c; 0.001, &#x19e;<sup>2</sup>p &#x3d; 0.284. Non-functioning distractors were found the most frequently at option E and the least frequently at option C, an exact inversion of the pattern observed for best distractor (all p<sub>s</sub> &#x3c; 0.001 in post-hoc tests, <xref ref-type="fig" rid="F1">Figure&#x20;1B</xref>). In short, while frequencies for options A, B, and D did not hugely differ either when observing best distractor or non-functioning distractors, frequencies for options C and E did differ importantly and were completely reversed, with an eloquent bias towards option C for best distractors and an eloquent bias towards option E for non-functioning distractors. A visual inspection of these frequencies&#x2019; distribution showed that the strongest distractors were, in general, more likely to be found among middle options, whereas the weakest ones were mostly found at the last option.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Distribution of different distractor types in multiple-choice tests. The distribution of best distractor, non-functioning distractors, and ranked distractors (best distractor, distractor 2, distractor 3, worst distractor) used in Chilean national examination to access higher education is presented in <bold>(A&#x2013;C)</bold>, respectively. All presented percentages are means across tests. In <bold>(A,B)</bold>, percentages are calculated for every single analysed test by dividing the number of best distractors and non-functioning distractors found in each option position throughout the test by the total number of best and non-functioning distractors in test, respectively. In <bold>(C)</bold>, percentages are computed differently: They are calculated for every single analysed test by counting the number of each distractor type found in each option position throughout the test and then dividing this number by the total number of distractors in that position in test. Error bars represent 95% confidence intervals.</p>
</caption>
<graphic xlink:href="feduc-06-731763-g001.tif"/>
</fig>
<p>When inspecting the frequency of distractor types (best distractor, distractor 2, distractor 3, worst distractor) at each option position (A, B, C, D, E), statistically significant differences were observed for all five positions: F (3,327) &#x3d; 3.483, 21.177, 95.690, 14.726, and 245.512, respectively, all p<sub>s</sub> &#x3c; 0.016, &#x19e;<sup>2</sup>p &#x3d; 0.031, 0.163, 0.467, 0.119, and 0.693, respectively. Post-hoc analyses revealed that the worst distractor was found less frequently than the other distractors at options B, C, and D, but much more frequently at option E (all p<sub>s</sub> &#x3c; 0.001). The best and second-best distractors were more frequently found at option C than the second-worst distractor was, and, conversely, they were both less frequently found at option E than the second-worst distractor (all p<sub>s</sub> &#x3c; 0.012).</p>
<p>Taken together, these results clearly revealed a bias in terms of how strong distractors and weak distractors spread between option positions. Strongest distractors were more likely to be found among middle options, preferentially at option C, whereas the weakest distractors were more likely to be found at last option, E. Supplementary analyses confirmed that these results were robust and generalizable. Frequencies for the best and worst distractors were biased even when distractors were defined more conservatively (see Data Analysis section and <xref ref-type="sec" rid="s10">Supplementary Figure S1</xref>), revealing that these position biases cannot be explained (at least not wholly explained) by the fact that examinees tended to more frequently select any specific option position(s). Frequencies for the best distractor and for non-functioning distractors were found to be similarly biased in the four tested domains and in the 5&#xa0;years exams were administered (<xref ref-type="sec" rid="s10">Supplementary Figure S2</xref>), which confirmed generalizability and replicability of findings. Critically, a similarly biased pattern for distractors was observed again when inspecting multiple-choice tests used in countries other than Chile (<xref ref-type="sec" rid="s10">Supplementary Figure S3</xref>), suggesting that the involved phenomenon is probably not cultural. Note that in this last analysis, bias was not only observed for five-option items, but also for four-option&#x20;items.</p>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>Previous studies about response options placement have shown that answer keys are not uniformly distributed in many multiple-choice tests, keys being more frequently positioned as a middle option than as an edge option (<xref ref-type="bibr" rid="B3">Attali and Bar-Hillel, 2003</xref>; Authors, 2021, under review; <xref ref-type="bibr" rid="B23">Metfessel and Sax, 1958</xref>). This keying bias reveals that test developers do not balance (or randomize) the position of answer keys in tests, ignoring guidelines provided by item-writing guides for decades now (<xref ref-type="bibr" rid="B29">Trump and Haggerty, 1952</xref>; <xref ref-type="bibr" rid="B15">Haladyna and Downing, 1989a</xref>; <xref ref-type="bibr" rid="B17">Haladyna et&#x20;al., 2002</xref>; <xref ref-type="bibr" rid="B19">Haladyna and Rodriguez, 2013</xref>). Implications for the validity of test scores may be critical: if test takers become aware that answer keys are more frequent among middle options, they can develop position-based strategies to make more accurate guesses and provide correct responses by selecting more central positions (<xref ref-type="bibr" rid="B5">Bar-Hillel and Attali, 2002</xref>; <xref ref-type="bibr" rid="B6">Bar-Hillel et&#x20;al., 2005</xref>).</p>
<p>Results from this study showed that neither strong nor weak distractors were uniformly distributed in tests: while the strongest distractors were most frequently found among middle options, the weakest distractor was most likely to be found at the end of the options list. These distribution biases are independent of the keying bias. Put differently, the best distractor of multiple-choice items tends to present itself before the worst one. This bias does not imply non-adherence to item-writing guidelines because no guide has provided any specific recommendations about distractors&#x2019; placement. However, it confirms that test developers do not usually randomize options order, contrary to recommendations from recent guides (<xref ref-type="bibr" rid="B32">Xu et&#x20;al., 2016</xref>; <xref ref-type="bibr" rid="B13">Gierl et&#x20;al., 2017</xref>).</p>
<p>Present findings have several implications. Most importantly, they have apparent implications for research exploring the effects of key position on item accuracy. Empirical literature about this topic reports conflicting results: While some studies have claimed that items are easier when key is placed in the middle (<xref ref-type="bibr" rid="B3">Attali and Bar-Hillel, 2003</xref>; <xref ref-type="bibr" rid="B9">DeVore et&#x20;al., 2016</xref>) or among the first options (<xref ref-type="bibr" rid="B20">Hohensinn and Baghaei, 2017</xref>; <xref ref-type="bibr" rid="B21">Holzknecht et&#x20;al., 2020</xref>), others have concluded that item performance is hardly affected by options position (<xref ref-type="bibr" rid="B26">Sonnleitner et&#x20;al., 2016</xref>; <xref ref-type="bibr" rid="B30">Wang, 2019</xref>). Since position of distractors has been shown to impact item accuracy (<xref ref-type="bibr" rid="B22">Kiat et&#x20;al., 2018</xref>; <xref ref-type="bibr" rid="B24">Shin et&#x20;al., 2019</xref>) and since the present study shows that the distribution of distractors may be significantly biased, the above inconsistency in reported results may ultimately be driven by the fact that numerous studies about key position did not control for distractors&#x2019; position. In other words, the middle bias observed in the past among examinees&#x2019; responses might not always have been a correlate of keying bias but the result of placing the strongest distractors within the middle options. Future studies inspecting the effects of key position on item performance and test scores might need to consider distractors&#x2019; position as a potential confounding factor.</p>
<p>Implications for test takers are less clear. Indeed, it remains uncertain how examinees would adapt their item-solving strategies if they knew that strongest distractors are more likely to be found among earlier options than weaker ones. Examinees might assume that the last option(s) is (are) not worth being read with care and focus their cognitive efforts on the first options in the list, which would be consistent with the claim that test takers do not always read all the alternatives before responding (<xref ref-type="bibr" rid="B8">Clark, 1956</xref>; <xref ref-type="bibr" rid="B11">Fagley, 1987</xref>; <xref ref-type="bibr" rid="B31">Willing, 2013</xref>) and with the fact that they most frequently explore options in order (<xref ref-type="bibr" rid="B21">Holzknecht et&#x20;al., 2020</xref>). Further research is needed to better understand the link between belief or awareness about options placement and how test takers read and solve multiple-choice&#x20;items.</p>
<p>One possible explanation for presented results is that distractors are generated and listed in order of plausibility during item-design stage, this order remaining unaltered by test developers during the process of assembling a test once items have been designed. If this is the case, it is only natural that the weakest distractors are to be found at the last option, because a highly plausible, strong distractor is more likely to be retrieved from memory during the item-writing process than a less plausible, weak distractor (<xref ref-type="bibr" rid="B3">Attali and Bar-Hillel, 2003</xref>). Ultimately, then, distractors&#x2019; prominence/cognitive availability shapes the options order, consistently with our working hypothesis. Test developers might thus be highly interested in the results presented here because they provide, to the best of our knowledge, the first evidence supporting the claim that options are generated in order of plausibility. Future studies might analyze in much more depth the creation process of single items to explore&#x20;this.</p>
<p>Finally, there are some limitations to be mentioned. First, most of the results presented in this article were based on data gathered from five-option items. A large sample of four-option items and three-option items should be analyzed to determine whether (and how) the number of options modulates the distribution bias of distractors. More generally, items with a different set of traits (such as items with ordered numbers as options or with algebraic expressions as options) should be studied to confirm whether this position bias is present in all kinds of multiple-choice items or not. Second, the distribution of distractors was mainly analyzed in a set of real-life high-stakes tests. More in-house tests should be analyzed to confirm that the distribution bias of distractors exists at all educational levels and gauge the impact of test developers&#x2019; training/experience at item writing on this phenomenon. Finally, studies identifying different distractor types by means of a method not solely based on response frequency are needed to disentangle developers&#x2019; placement bias more clearly from examinees&#x2019; response bias. One interesting possibility is working on item sets having distractor types clearly identified by test developers&#x2019; boards before administration. Although predicting which distractors will be most or least selected by examinees is not an easy task that will probably be not 100% accurate, such an approach would possibly bring decisive evidence in favor or against our hypothesis.</p>
<p>In sum, this is the first study showing that a clear and widespread bias can be observed in the distribution of distractors in multiple-choice tests, suggesting that distractors were probably sequenced in a plausibility order when developers created items. Considering that distractors&#x2019; relative position and distance to correct response affect item performance and test scores (<xref ref-type="bibr" rid="B22">Kiat et&#x20;al., 2018</xref>; <xref ref-type="bibr" rid="B24">Shin et&#x20;al., 2019</xref>), test developers should be aware that the order of distractors could introduce noise on test results, especially when options order is scrambled to generate equivalent test forms. Researchers interested in conducting empirical studies focused on option position effects should consider controlling distractors position if they want to adequately capture the effects of key position on the item performance and/or test outcomes. In short, this study should draw educators and researchers&#x2019; attention to an item trait they have probably never or rarely considered.</p>
</sec>
</body>
<back>
<sec id="s5">
<title>Data Availability Statement</title>
<p>The raw data supporting the conclusion of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s6">
<title>Author Contributions</title>
<p>SL, PD, and JL developed the study concept; MG, NC, and DJ handled the main data collection; SL and CM performed data analyses; MB and GO provided crucial information about item-writing guides and results&#x2019; presentation, respectively. SL drafted the manuscript, and all the other authors provided critical revisions. All authors have approved the final version of this manuscript.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>This research was supported by the following grants from ANID: Fondecyt postdoctorado &#x23;3190273 and FONDEF ID16I10090. Support from ANID/PIA/Basal Funds for Centers of Excellence FB0003 (Center for Advanced Research in Education) and AFB170001 (Center for Mathematical Modeling) is also gratefully acknowledged.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>We thank Mar&#xed;a Leonor Varas, director of the Departamento de Evaluaci&#xf3;n, Medici&#xf3;n y Registro Educacional (DEMRE), for her unconditional support and for making this collaborative research possible. We also thank Camilo Quezada Gaponov for editing the manuscript.</p>
</ack>
<sec id="s10">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/feduc.2021.731763/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/feduc.2021.731763/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Presentation1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ali</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Ruit</surname>
<given-names>K. G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>The Impact of Item Flaws, Testing at Low Cognitive Level, and Low Distractor Functioning on Multiple-Choice Question Quality</article-title>. <source>Perspect. Med. Educ.</source> <volume>4</volume> (<issue>5</issue>), <fpage>244</fpage>&#x2013;<lpage>251</lpage>. <pub-id pub-id-type="doi">10.1007/s40037-015-0212-x</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ambu-Saidi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Khamis</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2000</year>). <source>An Investigation into Fixed Response Questions in Science at Secondary and Tertiary Levels</source>. <comment>Doctoral dissertation</comment>. <publisher-loc>Glasgow</publisher-loc>: <publisher-name>University of Glasgow</publisher-name>. </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Attali</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bar-Hillel</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Guess where: The Position of Correct Answers in Multiple-Choice Test Items as a Psychometric Variable</article-title>. <source>J.&#x20;Educ. Meas.</source> <volume>40</volume> (<issue>2</issue>), <fpage>109</fpage>&#x2013;<lpage>128</lpage>. <pub-id pub-id-type="doi">10.1111/j.1745-3984.2003.tb01099.x</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bar-Hillel</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Position Effects in Choice from Simultaneous Displays: A Conundrum Solved</article-title>. <source>Perspect. Psychol. Sci.</source> <volume>10</volume> (<issue>4</issue>), <fpage>419</fpage>&#x2013;<lpage>433</lpage>. <pub-id pub-id-type="doi">10.1177/1745691615588092</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bar-Hillel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Attali</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Seek Whence</article-title>. <source>The Am. Statistician</source> <volume>56</volume> (<issue>4</issue>), <fpage>299</fpage>&#x2013;<lpage>303</lpage>. <pub-id pub-id-type="doi">10.1198/000313002623</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bar-Hillel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Budescu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Attali</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Scoring and Keying Multiple Choice Tests: A Case Study in Irrationality</article-title>. <source>Mind Soc.</source> <volume>4</volume> (<issue>1</issue>), <fpage>3</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1007/s11299-005-0001-z</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carnegie</surname>
<given-names>J.&#x20;A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Does Correct Answer Distribution Influence Student Choices when Writing Multiple Choice Examinations</article-title>. <source>cjsotl-rcacea</source> <volume>8</volume> (<issue>1</issue>), <fpage>11</fpage>. <pub-id pub-id-type="doi">10.5206/cjsotl-rcacea.2017.1.11</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clark</surname>
<given-names>E. L.</given-names>
</name>
</person-group> (<year>1956</year>). <article-title>General Response Patterns to Five-Choice Items</article-title>. <source>J.&#x20;Educ. Psychol.</source> <volume>47</volume> (<issue>2</issue>), <fpage>110</fpage>&#x2013;<lpage>117</lpage>. <pub-id pub-id-type="doi">10.1037/h0043113</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>DeVore</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Stewart</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Stewart</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Examining the Effects of Testwiseness in Conceptual Physics Evaluations</article-title>. <source>Phys. Rev. Phys. Educ. Res.</source> <volume>12</volume> (<issue>2</issue>), <fpage>020138</fpage>. <pub-id pub-id-type="doi">10.1103/PhysRevPhysEducRes.12.020138</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Downing</surname>
<given-names>S. M.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>The Effects of Violating Standard Item Writing Principles on Tests and Students: the Consequences of Using Flawed Test Items on Achievement Examinations in Medical Education</article-title>. <source>Adv. Health Sci. Educ. Theor. Pract.</source> <volume>10</volume> (<issue>2</issue>), <fpage>133</fpage>&#x2013;<lpage>143</lpage>. <pub-id pub-id-type="doi">10.1007/s10459-004-4019-5</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fagley</surname>
<given-names>N. S.</given-names>
</name>
</person-group> (<year>1987</year>). <article-title>Positional Response Bias in Multiple-Choice Tests of Learning: Its Relation to Testwiseness and Guessing Strategy</article-title>. <source>J.&#x20;Educ. Psychol.</source> <volume>79</volume> (<issue>1</issue>), <fpage>95</fpage>&#x2013;<lpage>97</lpage>. <pub-id pub-id-type="doi">10.1037/0022-0663.79.1.95</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Friel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Johnstone</surname>
<given-names>A. H.</given-names>
</name>
</person-group> (<year>1979</year>). <article-title>Does the Position of the Answer in a Multiple-Choice Test Matter</article-title>. <source>Educ. Chem.</source> <volume>16</volume>, <fpage>175</fpage>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://eric.ed.gov/?id=EJ213396">https://eric.ed.gov/?id&#x3d;EJ213396</ext-link>
</comment>. </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gierl</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Bulut</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Developing, Analyzing, and Using Distractors for Multiple-Choice Tests in Education: a Comprehensive Review</article-title>. <source>Rev. Educ. Res.</source> <volume>87</volume> (<issue>6</issue>), <fpage>1082</fpage>&#x2013;<lpage>1116</lpage>. <pub-id pub-id-type="doi">10.3102/0034654317726529</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gustav</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>1963</year>). <article-title>Response Set in Objective Achievement Tests</article-title>. <source>J.&#x20;Psychol.</source> <volume>56</volume> (<issue>2</issue>), <fpage>421</fpage>&#x2013;<lpage>427</lpage>. <pub-id pub-id-type="doi">10.1080/00223980.1963.9916657</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haladyna</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Downing</surname>
<given-names>S. M.</given-names>
</name>
</person-group> (<year>1989a</year>). <article-title>A Taxonomy of Multiple-Choice Item-Writing Rules</article-title>. <source>Appl. Meas. Educ.</source> <volume>2</volume> (<issue>1</issue>), <fpage>37</fpage>&#x2013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1207/s15324818ame0201_3</pub-id> </citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haladyna</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Downing</surname>
<given-names>S. M.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Construct-Irrelevant Variance in High-Stakes Testing</article-title>. <source>Educ. Meas. Issues Pract.</source> <volume>23</volume> (<issue>1</issue>), <fpage>17</fpage>&#x2013;<lpage>27</lpage>. <pub-id pub-id-type="doi">10.1111/j.1745-3992.2004.tb00149.x</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haladyna</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Downing</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Rodriguez</surname>
<given-names>M. C.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>A Review of Multiple-Choice Item-Writing Guidelines for Classroom Assessment</article-title>. <source>Appl. Meas. Educ.</source> <volume>15</volume> (<issue>3</issue>), <fpage>309</fpage>&#x2013;<lpage>333</lpage>. <pub-id pub-id-type="doi">10.1207/s15324818ame1503_5</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haladyna</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Downing</surname>
<given-names>S. M.</given-names>
</name>
</person-group> (<year>1989b</year>). <article-title>Validity of a Taxonomy of Multiple-Choice Item-Writing Rules</article-title>. <source>Appl. Meas. Educ.</source> <volume>2</volume> (<issue>1</issue>), <fpage>51</fpage>&#x2013;<lpage>78</lpage>. <pub-id pub-id-type="doi">10.1207/s15324818ame0201_4</pub-id> </citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Haladyna</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Rodriguez</surname>
<given-names>M. C.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Developing and Validating Test Items</article-title>,&#x201d; in <source>Developing and Validating Test Items</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Haladyna</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Rodriguez</surname>
<given-names>M. C.</given-names>
</name>
</person-group> (<publisher-loc>New York</publisher-loc>: <publisher-name>Routledge</publisher-name>), <fpage>89</fpage>&#x2013;<lpage>110</lpage>. <pub-id pub-id-type="doi">10.4324/9780203850381</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hohensinn</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Baghaei</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Does the Position of Response Options in Multiple-Choice Tests Matter</article-title>. <source>Psicol&#xf3;gica</source> <volume>38</volume> (<issue>1</issue>), <fpage>93</fpage>&#x2013;<lpage>109</lpage>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://files.eric.ed.gov/fulltext/EJ1125979.pdf">https://files.eric.ed.gov/fulltext/EJ1125979.pdf</ext-link>
</comment>. </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Holzknecht</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>McCray</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Eberharter</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kremmel</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zehentner</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Spiby</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>The Effect of Response Order on Candidate Viewing Behaviour and Item Difficulty in a Multiple-Choice Listening Test</article-title>. <source>Lang. Test.</source> <volume>38</volume>, <fpage>41</fpage>&#x2013;<lpage>61</lpage>. <pub-id pub-id-type="doi">10.1177/0265532220917316</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kiat</surname>
<given-names>J.&#x20;E.</given-names>
</name>
<name>
<surname>Ong</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Ganesan</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>The Influence of Distractor Strength and Response Order on MCQ Responding</article-title>. <source>Educ. Psychol.</source> <volume>38</volume> (<issue>3</issue>), <fpage>368</fpage>&#x2013;<lpage>380</lpage>. <pub-id pub-id-type="doi">10.1080/01443410.2017.1349877</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Metfessel</surname>
<given-names>N. S.</given-names>
</name>
<name>
<surname>Sax</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1958</year>). <article-title>Systematic Biases in the Keying of Correct Responses on Certain Standardized Tests</article-title>. <source>Educ. Psychol. Meas.</source> <volume>18</volume> (<issue>4</issue>), <fpage>787</fpage>&#x2013;<lpage>790</lpage>. <pub-id pub-id-type="doi">10.1177/001316445801800411</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bulut</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Gierl</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>The Effect of the Most-Attractive-Distractor Location on Multiple-Choice Item Difficulty</article-title>. <source>J.&#x20;Exp. Educ.</source> <volume>88</volume>, <fpage>643</fpage>&#x2013;<lpage>659</lpage>. <pub-id pub-id-type="doi">10.1080/00220973.2019.1629577</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Siddiqui</surname>
<given-names>Z. S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Errors in the Construction of Multi-Choice Questions: An Analysis</article-title>. <source>The Pakistan J.&#x20;Med. Dentistry</source> <volume>7</volume> (<issue>4</issue>), <fpage>4</fpage>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://research-repository.uwa.edu.au/files/39825784/ERRORS_IN_THE_CONSTRUCTION_OF_MULTI_CHOI.pdf">https://research-repository.uwa.edu.au/files/39825784/ERRORS_IN_THE_CONSTRUCTION_OF_MULTI_CHOI.pdf</ext-link>
</comment>. </citation>
</ref>
<ref id="B26">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Sonnleitner</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Guill</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hohensinn</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Effects of Correct Answer Position on Multiplechoice Item Difficulty in Educational Settings: Where Would You Go</article-title>,&#x201d; in <conf-name>International Test Commission Conference</conf-name>, <conf-loc>Vancouver, Canada</conf-loc>, <conf-date>August 2, 2016</conf-date>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="http://hdl.handle.net/10993/29469">http://hdl.handle.net/10993/29469</ext-link>
</comment>. </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tarrant</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ware</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Impact of Item-Writing Flaws in Multiple-Choice Questions on Student Achievement in High-Stakes Nursing Assessments</article-title>. <source>Med. Educ.</source> <volume>42</volume> (<issue>2</issue>), <fpage>198</fpage>&#x2013;<lpage>206</lpage>. <pub-id pub-id-type="doi">10.1111/j.1365-2923.2007.02957.x</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tarrant</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ware</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mohammed</surname>
<given-names>A. M.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>An Assessment of Functioning and Non-functioning Distractors in Multiple-Choice Questions: a Descriptive Analysis</article-title>. <source>BMC Med. Educ.</source> <volume>9</volume> (<issue>1</issue>), <fpage>40</fpage>&#x2013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.1186/1472-6920-9-40</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Trump</surname>
<given-names>J.&#x20;B.</given-names>
</name>
<name>
<surname>Haggerty</surname>
<given-names>H. R.</given-names>
</name>
</person-group> (<year>1952</year>). <source>Basic Principles in Achievement Test Item Construction</source>. <publisher-loc>Washington</publisher-loc>: <publisher-name>The Adjutant General&#x2019;s Office</publisher-name>. <comment>Personnel Research Section Report 979</comment>. <pub-id pub-id-type="doi">10.1037/e523682009-001</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Does Rearranging Multiple&#x2010;Choice Item Response Options Affect Item and Test Performance</article-title>. <source>ETS Res. Rep. Ser.</source> <volume>2019</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1002/ets2.12238</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Willing</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2013</year>). <source>Discrete-option Multiple-Choice: Evaluating The Psychometric Properties of a New Method of Knowledge Assessment</source>. <comment>Doctoral dissertation</comment>. <publisher-loc>Duesseldorf</publisher-loc>: <publisher-name>University of D&#x00FC;sseldorf</publisher-name>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://docserv.uni-duesseldorf.de/servlets/DerivateServlet/Derivate-29719/Dissertation%20Sonja%20Willing.pdf">https://docserv.uni-duesseldorf.de/servlets/DerivateServlet/Derivate-29719/Dissertation%20Sonja%20Willing.pdf</ext-link>
</comment>. </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Kauer</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tupy</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Multiple-choice Questions: Tips for Optimizing Assessment In-Seat and Online</article-title>. <source>Scholarship Teach. Learn. Psychol.</source> <volume>2</volume> (<issue>2</issue>), <fpage>147</fpage>&#x2013;<lpage>158</lpage>. <pub-id pub-id-type="doi">10.1037/stl0000062</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>