<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Vet. Sci.</journal-id>
<journal-title>Frontiers in Veterinary Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Vet. Sci.</abbrev-journal-title>
<issn pub-type="epub">2297-1769</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fvets.2024.1385681</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Veterinary Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Identification of parameters for electronic distance examinations</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Richter</surname> <given-names>Robin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2649732/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Tipold</surname> <given-names>Andrea</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/137044/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Schaper</surname> <given-names>Elisabeth</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/971130/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Centre for E-Learning, Didactics and Educational Research (ZELDA), University of Veterinary Medicine Hannover, Foundation</institution>, <addr-line>Hanover</addr-line>, <country>Germany</country></aff>
<aff id="aff2"><sup>2</sup><institution>Clinic for Small Animals, Neurology, University of Veterinary Medicine Hannover, Foundation</institution>, <addr-line>Hanover</addr-line>, <country>Germany</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0001">
<p>Edited by: Sheena Warman, University of Bristol, United Kingdom</p>
</fn>
<fn fn-type="edited-by" id="fn0002">
<p>Reviewed by: Deirdre P. Campion, University College Dublin, Ireland</p>
<p>Yolanda Martinez Pereira, University of Edinburgh, United Kingdom</p>
<p>Emma Fishbourne, University of Liverpool, United Kingdom</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Robin Richter, <email>robin.richter@tiho-hannover.de</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>19</day>
<month>06</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>11</volume>
<elocation-id>1385681</elocation-id>
<history>
<date date-type="received">
<day>13</day>
<month>02</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>06</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2024 Richter, Tipold and Schaper.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Richter, Tipold and Schaper</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>This study investigates the log data and response behavior from invigilated in-person electronic timed exams at the University of Veterinary Medicine Hannover, Foundation, Germany. The primary focus is on understanding how various factors influence the time needed per exam item, including item format, item difficulty, item discrimination and character count. The aim was to use these results to derive recommendations for designing timed online distance examinations, an examination format that has become increasingly important in recent years.</p>
</sec>
<sec>
<title>Methods</title>
<p>Data from 216,625 log entries of five electronic exams, taken by a total of 1,241 veterinary medicine students in 2021 and 2022, were analyzed. Various statistical methods were employed to assess the correlations between the recorded parameters.</p>
</sec>
<sec>
<title>Results</title>
<p>The analysis revealed that different item formats require varying amounts of time. For instance, image-based question formats and Kprim necessitated more than 60 s per item, whereas one-best-answer multiple-choice questions (MCQs) and individual Key Feature items were effectively completed in less than 60 s. Furthermore, there was a positive correlation between character count and response time, suggesting that longer items require more time. A negative correlation could be verified for the parameters &#x201C;difficulty&#x201D; and &#x201C;discrimination index&#x201D; towards response time, indicating that more challenging items and those that are less able to differentiate between high- and low-performing students take longer to answer.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>The findings highlight the need for careful consideration of the ratio of item formats when defining time limits for exams. Regarding exam design, the literature mentions that time pressure is a critical factor, since it can negatively impact students&#x2019; exam performance and some students, such as those with disabilities, are particularly disadvantaged. Therefore, this study emphasizes finding the right time limits to provide sufficient time for answering questions and reducing time pressure. In the context of unsupervised online exams, the findings of this study support previous recommendations that implementation of a stringent time limit might be a useful strategy to reduce cheating.</p>
</sec>
</abstract>
<kwd-group>
<kwd>E-assessment</kwd>
<kwd>veterinary education</kwd>
<kwd>examinations</kwd>
<kwd>open-book</kwd>
<kwd>item formats</kwd>
<kwd>log data</kwd>
<kwd>response time</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="4"/>
<equation-count count="0"/>
<ref-count count="74"/>
<page-count count="13"/>
<word-count count="9050"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Veterinary Humanities and Social Sciences</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>The digitalization of university teaching has been an ongoing process for decades, which was significantly accelerated and promoted by the COVID-19 pandemic and the associated infection prevention measures (<xref ref-type="bibr" rid="ref1 ref2 ref3 ref4 ref5">1&#x2013;5</xref>). This was also the case for digitalization measures introduced in veterinary medical education (<xref ref-type="bibr" rid="ref4 ref5 ref6">4&#x2013;6</xref>). One of the most challenging aspects during the pandemic was the issue of conducting examinations. In the light of the infection prevention measures, the formats of traditional in-person written exams and Objective Structured Clinical Examinations were no longer seen as practical for examinations involving large cohorts (<xref ref-type="bibr" rid="ref4">4</xref>, <xref ref-type="bibr" rid="ref7">7</xref>). Distance examinations became the focus as a solution to this challenge.</p>
<p>However, especially in the area of digital distance examinations, there are numerous hurdles in the context of examination regulations and data protection law (<xref ref-type="bibr" rid="ref3">3</xref>, <xref ref-type="bibr" rid="ref7">7</xref>, <xref ref-type="bibr" rid="ref8">8</xref>). Particularly concerning online proctoring, which is meant to ensure constant monitoring of the examinees&#x2019; identity, protection against attempted cheating, and prevention of the usage of unauthorized tools, many of the currently available technical solutions and tools must be rejected due to European General Data Regulation (GDPR) requirements (<xref ref-type="bibr" rid="ref8">8</xref>). As a result, the concept of open-book examinations is the main focus of exam design, since in this examination format, apart from direct exchange between candidates, the examinees are allowed to use any form of resources to solve the tasks. Consequently, the need for continuous monitoring to guard against cheating attempts using external sources is limited, and hence the open-book format alleviates the challenging implementation of online proctoring for distance examinations (<xref ref-type="bibr" rid="ref9">9</xref>).</p>
<p>At the University of Veterinary Medicine Hannover, Foundation (TiHo), Hannover, Germany, electronic examinations have been conducted since 2008 (<xref ref-type="bibr" rid="ref10">10</xref>), which means that digital performance assessments are already established in both a didactic and technical sense. In the light of the pandemic, the potential of online open-book distance examinations for future examination procedures needed to be assessed. This entailed a review of existing formats and the identification of suitable evaluation parameters. The aim of this study was to evaluate log data from electronic examinations and the examinees&#x2019; response selection behavior to determine whether recommendations for the design of online distance examinations can be derived from these data.</p>
</sec>
<sec sec-type="materials|methods" id="sec2">
<label>2</label>
<title>Materials and methods</title>
<sec id="sec3">
<label>2.1</label>
<title>Selection of data sets</title>
<p>Log data and response selection behavior were examined. A total of 216,625 log records from a cohort of 1,241 students were analyzed, including 225 participants who took multiple exams. Data were derived from one exam from each of five different departments. Examinations from the years 2021 and 2022 were used, with an average participation of 248 individuals per exam (range: 234&#x2013;282). Care was taken in the selection process to ensure that the summative exams were from various stages of the curriculum, the allocated time per item was varied, and the exams exhibited good characteristic values (Difficulty, Cronbach&#x2019;s &#x03B1;). The five exams were invigilated electronic state examinations conducted in-person using the Q-Examiner<sup>&#x00AE;</sup> software (IQuL GmbH, Bergisch Gladbach, Germany); additional details about the exams are provided in <xref ref-type="table" rid="tab1">Table 1</xref>.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Details on the examinations selected for the analysis of log data sets, including information on the characteristic values of the exams (Cronbach&#x2019;s &#x03B1;, overall exam difficulty) and item analysis values (Difficulty, Discrimination index).</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top" colspan="2">Subjects</th>
<th align="center" valign="top">A</th>
<th align="center" valign="top">B</th>
<th align="center" valign="top">C</th>
<th align="center" valign="top">D</th>
<th align="center" valign="top">E</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" colspan="2">Section</td>
<td align="center" valign="middle">Clinical sciences</td>
<td align="center" valign="middle">Clinical sciences</td>
<td align="center" valign="middle">Clinical sciences</td>
<td align="center" valign="middle">Veterinary public health</td>
<td align="center" valign="middle">Basic sciences</td>
</tr>
<tr>
<td align="left" valign="middle" colspan="2">Number of items</td>
<td align="center" valign="middle">73</td>
<td align="center" valign="middle">90</td>
<td align="center" valign="middle">73</td>
<td align="center" valign="middle">50</td>
<td align="center" valign="middle">60</td>
</tr>
<tr>
<td align="left" valign="middle" colspan="2">Exam time (in minutes)</td>
<td align="center" valign="middle">90</td>
<td align="center" valign="middle">120</td>
<td align="center" valign="middle">90</td>
<td align="center" valign="middle">100</td>
<td align="center" valign="middle">90</td>
</tr>
<tr>
<td align="left" valign="middle" colspan="2">Cronbach&#x2019;s &#x03B1;</td>
<td align="center" valign="middle">0.79</td>
<td align="center" valign="middle">0.81</td>
<td align="center" valign="middle">0.81</td>
<td align="center" valign="middle">0.76</td>
<td align="center" valign="middle">0.86</td>
</tr>
<tr>
<td align="left" valign="middle" colspan="2">Overall exam difficulty (%)</td>
<td align="center" valign="middle">69.6</td>
<td align="center" valign="middle">68.5</td>
<td align="center" valign="middle">77.5</td>
<td align="center" valign="middle">79.7</td>
<td align="center" valign="middle">65.3</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="3">Difficulty (P) of the items (%)</td>
<td align="left" valign="middle">Median</td>
<td align="center" valign="middle">75.0</td>
<td align="center" valign="middle">72.9</td>
<td align="center" valign="middle">82.1</td>
<td align="center" valign="middle">86.2</td>
<td align="center" valign="middle">67.7</td>
</tr>
<tr>
<td align="left" valign="middle">IQR</td>
<td align="center" valign="middle">47.0</td>
<td align="center" valign="middle">42.8</td>
<td align="center" valign="middle">26.9</td>
<td align="center" valign="middle">24.1</td>
<td align="center" valign="middle">29.6</td>
</tr>
<tr>
<td align="left" valign="middle">Number of items in optimal range <italic>p</italic> =&#x2009;40&#x2013;94%</td>
<td align="center" valign="middle">41</td>
<td align="center" valign="middle">60</td>
<td align="center" valign="middle">56</td>
<td align="center" valign="middle">35</td>
<td align="center" valign="middle">51</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="3">Discrimination index (r) of the items</td>
<td align="left" valign="middle">Median</td>
<td align="center" valign="middle">0.15</td>
<td align="center" valign="middle">0.2</td>
<td align="center" valign="middle">0.25</td>
<td align="center" valign="middle">0.27</td>
<td align="center" valign="middle">0.3</td>
</tr>
<tr>
<td align="left" valign="middle">IQR</td>
<td align="center" valign="middle">0.15</td>
<td align="center" valign="middle">0.16</td>
<td align="center" valign="middle">0.22</td>
<td align="center" valign="middle">0.14</td>
<td align="center" valign="middle">0.16</td>
</tr>
<tr>
<td align="left" valign="middle">Number of items in optimal range <italic>r</italic> &#x2265;&#x2009;0.2</td>
<td align="center" valign="middle">26</td>
<td align="center" valign="middle">46</td>
<td align="center" valign="middle">48</td>
<td align="center" valign="middle">38</td>
<td align="center" valign="middle">47</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">Item response time (in seconds)</td>
<td align="left" valign="middle">Median</td>
<td align="center" valign="middle">56</td>
<td align="center" valign="middle">56</td>
<td align="center" valign="middle">47</td>
<td align="center" valign="middle">57</td>
<td align="center" valign="middle">56</td>
</tr>
<tr>
<td align="left" valign="middle">IQR</td>
<td align="center" valign="middle">44</td>
<td align="center" valign="middle">37</td>
<td align="center" valign="middle">29</td>
<td align="center" valign="middle">37</td>
<td align="center" valign="middle">28</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Overall exam difficulty is calculated as the mean value of all scores achieved by the students and is given as a percentage of the maximum score. IQR, interquartile range.</p>
</table-wrap-foot>
</table-wrap>
<p>The item formats utilized in this study comprised of multiple-choice question (MCQ) in single-choice format, Kprim, Key Feature, picture diagnosis, and picture mapping, which are described in more detail in <xref ref-type="table" rid="tab2">Table 2</xref>.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Characteristics and scoring scheme of the five item formats used at the University of Veterinary Medicine Hannover.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Item format</th>
<th align="left" valign="top">Description</th>
<th align="left" valign="top">Evaluation</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">MCQ</td>
<td align="left" valign="top">A one-best-answer format featuring one correct option (attractor) and two to four incorrect choices (distractors).</td>
<td align="left" valign="top">Attractor chosen: One point.<break/>Distractor chosen: No points.</td>
</tr>
<tr>
<td align="left" valign="top">Kprim</td>
<td align="left" valign="top">A true-false selection item with exactly four answer options. Each of these options must be marked as &#x201C;correct&#x201D; or &#x201C;incorrect.&#x201D;</td>
<td align="left" valign="top">Four correct matches receive one point, three correct matches earn half a point, and less than three correct matches result in no points.</td>
</tr>
<tr>
<td align="left" valign="top">Key feature</td>
<td align="left" valign="top">Three individual items with a predetermined order, that are designed to build on each other regarding content, case study or topic. After response selection and choice confirmation, selected options cannot be changed.</td>
<td align="left" valign="top">Each correctly answered subquestion awards one point.</td>
</tr>
<tr>
<td align="left" valign="top">Picture diagnosis</td>
<td align="left" valign="top">A marker must be placed on a picture.</td>
<td align="left" valign="top">If the marker was positioned within the predefined area, one point is awarded.</td>
</tr>
<tr>
<td align="left" valign="top">Picture mapping</td>
<td align="left" valign="top">Terms are matched to predefined, specific marks on an image.</td>
<td align="left" valign="top">One point is rewarded for a completely correct assignment of the terms. Half a point is awarded if at least half of the terms were assigned correctly.</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec4">
<label>2.2</label>
<title>Log data analysis</title>
<p>Based on the log data, the response time per item of each examinee was determined, and subsequently, the mean response time for each individual item was calculated. For the respective subjects and examinees, the median time spent on each item and the submission time of the exam was recorded. To standardize and facilitate comparability, the submission time was calculated as a percentage of the maximum available time for the examinations.</p>
<p>The overall difficulty of the examinations was calculated as the mean value of all examinees&#x2019; scores and is presented as a percentage of the maximum attainable score.</p>
<p>For each of the 346 individual items, the parameters considered were the item format, response time, character count as well as the two psychometric characteristics: discrimination index and difficulty. The character count was defined as the sum of all letters and numbers, including special characters, punctuation, and spaces from the item stem and the answer options. These parameters were then checked for correlation. Correlations between the response time and the parameters difficulty, discrimination index and character count were examined separately for every item format in order to exclude the influence of the variances between the formats. Only the statistical data for the two formats MCQ and Kprim are presented, as a sufficiently high number of items for a meaningful statistical evaluation of the other formats was not achieved. As a next step, the ratio between the length of the question stem and the answer options was examined. For this purpose, items of the two item formats MCQ (<italic>n</italic>&#x2009;=&#x2009;231) and Kprim (<italic>n</italic>&#x2009;=&#x2009;81) were considered. The relative proportion of the character count of the question stem to the total character count of the item was calculated for each, and its relationship with the parameters difficulty and discrimination index was assessed.</p>
<p>Regarding the response selection behavior, data on changes made to the selected answer option were only available for items of the MCQ format. This meant that a total of 231 items from the five examinations were available for evaluation. For every item, the number of changes in answer selection, along with the corresponding switch between the originally selected option and the newly chosen answer were recorded for each examinee. Furthermore, in cases where the response selection was changed several times, the last change was identified, as it was the one that was ultimately graded. Due to variations in the exam conditions concerning item and participant numbers, the analyses were conducted based on relative proportions. For each exam, the following parameters were calculated:</p><list list-type="order">
<list-item>
<p>The proportion of items where the chosen answer option was changed by at least one examinee.</p>
</list-item>
<list-item>
<p>The proportion of examinees who altered the originally selected option to a different response for at least one item.</p>
</list-item>
<list-item>
<p>The average proportion of items in which individual examinees modified the original answer.</p>
</list-item>
</list>
<p>To assess the quality of answer modifications in detail, the number of changes between distractors and attractor or between different distractors was examined. For clarity, distractors are referred to as incorrect answer options (incorrect) and the attractor as the correct answer option (correct).</p>
</sec>
<sec id="sec5">
<label>2.3</label>
<title>Difficulty and discrimination index</title>
<p>The difficulty of an item is defined as the percentage of participants who answered the task correctly and can thus range from 0 to 100% (<xref ref-type="bibr" rid="ref11">11</xref>). The recommended range for item difficulty is 40&#x2013;94% (<xref ref-type="bibr" rid="ref12">12</xref>). Discrimination index describes an item&#x2019;s ability to differentiate between participants with high performance and those with low performance. Items with good discrimination are answered correctly by good candidates and incorrectly by poorer candidates (<xref ref-type="bibr" rid="ref13">13</xref>). The values of the discrimination index can vary from &#x2212;1 to +1, with values above 0.2 considered adequate (<xref ref-type="bibr" rid="ref12">12</xref>).</p>
</sec>
<sec id="sec6">
<label>2.4</label>
<title>Statistics and data privacy</title>
<p>This study was approved in advance by the Data Protection Officer at the TiHo. All utilized and collected data were processed and analyzed anonymously. Students had to agree a data protection declaration upon matriculation, which permits the use of data collected during examinations in anonymized form in accordance with the requirements of Art. 6(1)(e), 89 GDPR in conjunction with &#x00A7; 13 Lower Saxony Data Protection Law (Nieders&#x00E4;chsisches Datenschutzgesetz, NDSG).</p>
<p>Access to the raw data was restricted to the authors of this paper only, and all data was stored and processed on secure servers within the institution. To protect students&#x2019; data privacy, all personal identifiers were removed from the data before analysis, including matriculation numbers and any other information that could potentially be used to identify individual students. Instead, a unique, anonymized identifier was assigned to each data point to maintain the integrity of the dataset. Descriptive and statistical analysis was performed using aggregated data to prevent the identification of students based on their response behavior during the exams.</p>
<p>The descriptive analysis was performed using the spreadsheet software Microsoft&#x00AE; Office Excel 2010 (Microsoft Corporation, Redmond, WA, USA), while advanced statistical analysis was conducted using SAS&#x00AE; Software, Version 9.4, and SAS&#x00AE; Enterprise Guide&#x00AE; 7.1 (SAS Institute Inc., Cary, NC, USA).</p>
<p>Concerning quantitative data, all normally distributed numerical values are presented as mean values, including the standard deviation (SD) where applicable. For non-normally distributed values, the median and interquartile range (IQR) are provided.</p>
<p>To examine the correlations among the quantitative parameters, these were initially tested for normal distribution using the Kolmogorov&#x2013;Smirnov test. Subsequently, a Spearman&#x2019;s rank correlation analysis was performed on non-normally distributed data. For correlations between qualitative with quantitative parameters, the Kruskal-Wallis test was applied, followed by the Dwass-Steel-Critchlow-Fligner pairwise comparison method. A significance level of 5% was used, indicating that a <italic>p</italic>-value &#x003C;0.05 implied that the influence of the parameters was significant.</p>
</sec>
</sec>
<sec sec-type="results" id="sec7">
<label>3</label>
<title>Results</title>
<sec id="sec8">
<label>3.1</label>
<title>Time of submission of the exams</title>
<p>Log data was initially analyzed by examining the time of submission of exams by the participants in their respective subjects. The absolute number of submissions per submission time, along with the available time in minutes, the number of items, and the overall difficulty of each exam are presented in <xref ref-type="fig" rid="fig1">Figure 1</xref>. Across all five of the analyzed exams, half of the candidates completed the exams within 64% (IQR: 11%, range: 46&#x2013;77%) of the maximum available time. Three out of four students had submitted their exams within 77% (IRQ: 17%, range: 58&#x2013;90%) of the exam time. Additionally, 90% of the participants finished the exams within 92% (IQR: 17%, range: 70&#x2013;98%) of the exam time.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Absolute number of examinees submitting their exams at the same point in time for each of five anonymized academic subjects of veterinary medicine <bold>(A&#x2013;E)</bold> including information on number of items and exam time limit. Submission times of exams by individual students were calculated as a percentage of the maximum available exam time. Examination difficulty is determined as the mean examination score of all examinees of the respective exam and shown as a percentage of the highest achievable score. Spearman&#x2019;s rank correlation analysis shows a significant (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001) negative relationship (<italic>r</italic><sub>s</sub>&#x2009;=&#x2009;&#x2212;0.2738) between examination difficulty and submission time of the students.</p>
</caption>
<graphic xlink:href="fvets-11-1385681-g001.tif"/>
</fig>
<p>For each exam, the average allocated time per item was calculated for further analysis. Subsequently, correlations among the parameters &#x201C;available time per item,&#x201D; &#x201C;submission time,&#x201D; and &#x201C;overall difficulty&#x201D; were examined. A significant (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001) negative correlation (<italic>r</italic><sub>s</sub>&#x2009;=&#x2009;&#x2212;0.4312) was observed between the available time per item and the submission time. Hence, participants did not utilize the additional available time per item in exams that allocated more time per item.</p>
<p>Furthermore, a significant (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001) positive correlation (<italic>r</italic><sub>s</sub>&#x2009;=&#x2009;0.1106) was found between the available time per item and overall exam difficulty. This implies that the more time students had to answer the items, the higher the frequency of correct answers. There existed a significant (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001) negative correlation (<italic>r</italic><sub>s</sub>&#x2009;=&#x2009;&#x2212;0.2738) between the submission time and overall exam difficulty. Accordingly, students completed easier exams earlier.</p>
</sec>
<sec id="sec9">
<label>3.2</label>
<title>Item response time</title>
<p>The median item response time including range for each of the five academic subjects is shown in <xref ref-type="table" rid="tab1">Table 1</xref>.</p>
<p>Regarding the five item formats MCQ, Kprim, Key Feature, picture mapping, and picture diagnosis, the median response time was evaluated. For items of the MCQ format (<italic>n</italic>&#x2009;=&#x2009;231), 43&#x2009;s (IQR: 26&#x2009;s) were spent, for Kprim (<italic>n</italic>&#x2009;=&#x2009;81) 74&#x2009;s (IQR: 26&#x2009;s), for Key Feature per subquestion (<italic>n</italic>&#x2009;=&#x2009;21) 44&#x2009;s (IQR: 30&#x2009;s), for picture mapping (<italic>n</italic>&#x2009;=&#x2009;8) 87&#x2009;s (IQR: 84&#x2009;s), and for picture diagnosis (<italic>n</italic>&#x2009;=&#x2009;5) 77&#x2009;s (IQR: 35&#x2009;s). The effect of item format on item response time was significant (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001). A more detailed distribution of the required time per item and format is depicted in <xref ref-type="fig" rid="fig2">Figure 2</xref>. Pairwise comparisons show significant differences between the formats MCQ and Kprim (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001) as well as between Key Feature and Kprim (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001). There are no significant differences between any of the other formats. Further analysis reveals significant differences between item formats referring to their difficulty (<italic>p</italic>&#x2009;=&#x2009;0.0003) and discrimination index (<italic>p</italic>&#x2009;=&#x2009;0.0243). Pairwise comparisons show significant variations in difficulty for MCQ and Kprim (<italic>p</italic>&#x2009;=&#x2009;0.0009) as well as for Kprim and Key Feature (<italic>p</italic>&#x2009;=&#x2009;0.0026). For variations in discrimination index the test was significant for MCQ and Key Feature (<italic>p</italic>&#x2009;=&#x2009;0.0072) as well as Kprim and Key Feature (<italic>p</italic>&#x2009;=&#x2009;0.0228).</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Item response times of the five item formats MCQ, Kprim, Key Feature, picture diagnosis, and picture mapping. Kruskal-Wallis test displays significant differences (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001) between the response times of item formats. Pairwise comparisons using the Dwass-Steel-Critchlow-Fligner method reveal statistically significant differences between MCQ and Kprim (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001) as well as between Key Feature and Kprim (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001); <italic>n</italic>&#x2009;=&#x2009;346.</p>
</caption>
<graphic xlink:href="fvets-11-1385681-g002.tif"/>
</fig>
<p>In addition to the influence of the item format on item response time an assessment of the impact of the character count of an item was included. Median character count of the item formats was 246 characters (IQR: 200 characters) for MCQ items (<italic>n</italic>&#x2009;=&#x2009;231), 268 characters (IQR: 222 characters) for Kprim (<italic>n</italic>&#x2009;=&#x2009;81), 446 characters (IQR: 243 characters) per subquestion for Key Feature (<italic>n</italic>&#x2009;=&#x2009;21), 253 characters (IQR: 113 characters) for picture mapping (<italic>n</italic>&#x2009;=&#x2009;8), and 292 characters (IQR: 109 characters) for picture diagnosis (<italic>n</italic>&#x2009;=&#x2009;5). Variations in character count of the various item formats were significantly different (<italic>p</italic>&#x2009;=&#x2009;0.0003) but only for the pairwise comparison of MCQ and Key Feature (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001). In <xref ref-type="fig" rid="fig3">Figure 3</xref> the formats MCQ (<italic>n</italic>&#x2009;=&#x2009;231) and Kprim (<italic>n</italic>&#x2009;=&#x2009;81) were included separately to display the time spent on each item based on individual character count. A significant positive correlation was found between the two parameters item response time and character count for both MCQ (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001, <italic>r</italic><sub>s</sub>&#x2009;=&#x2009;0.3809) and Kprim (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001, <italic>r</italic><sub>s</sub>&#x2009;=&#x2009;0.5986), which indicated that more time is needed to complete items the more characters they contain.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Item response times in minutes depending on the respective character count and separated by item format MCQ (<italic>n</italic>&#x2009;=&#x2009;231, blue) and Kprim (<italic>n</italic>&#x2009;=&#x2009;81, gray). Spearman&#x2019;s rank correlation analysis indicates a statistically significant positive correlation between item response time and character count for MCQ (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001, <italic>r</italic><sub>s</sub>&#x2009;=&#x2009;0.3809) and Kprim (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001, <italic>r</italic><sub>s</sub>&#x2009;=&#x2009;0.5986); <italic>n</italic>&#x2009;=&#x2009;312.</p>
</caption>
<graphic xlink:href="fvets-11-1385681-g003.tif"/>
</fig>
<p><xref ref-type="fig" rid="fig4">Figure 4</xref> illustrates both the difficulty of individual items separated by item format and their respective average response time, revealing a significant and negative relationship between the parameters &#x201C;difficulty&#x201D; and &#x201C;item response time&#x201D; for MCQ (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001, <italic>r</italic><sub>s</sub>&#x2009;=&#x2009;&#x2212;0.6607) and Kprim (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001, <italic>r</italic><sub>s</sub>&#x2009;=&#x2009;&#x2212;0.5038). Accordingly, answering more difficult questions required more time.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Item response times in minutes depending on the respective item difficulty index and separated by item format MCQ (<italic>n</italic>&#x2009;=&#x2009;231, blue) and Kprim (<italic>n</italic>&#x2009;=&#x2009;81, gray). Spearman&#x2019;s rank correlation analysis indicates a statistically significant negative correlation between item response time and difficulty index for MCQ (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001, <italic>r</italic><sub>s</sub>&#x2009;=&#x2009;&#x2212;0.6607) and Kprim (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001, <italic>r</italic><sub>s</sub>&#x2009;=&#x2009;&#x2212;0.5038); <italic>n</italic>&#x2009;=&#x2009;312.</p>
</caption>
<graphic xlink:href="fvets-11-1385681-g004.tif"/>
</fig>
<p>However, no significant correlation (MCQ: <italic>p</italic>&#x2009;=&#x2009;0.2101; Kprim: <italic>p</italic>&#x2009;=&#x2009;0.7390) was found for the relationship between character count and difficulty.</p>
<p><xref ref-type="fig" rid="fig5">Figure 5</xref> depicts the discrimination index of individual items separated by item format and their respective average response time. A significant negative correlation was identified between the parameters &#x201C;discrimination index&#x201D; and &#x201C;item response time&#x201D; for MCQ (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001, <italic>r</italic><sub>s</sub>&#x2009;=&#x2009;&#x2212;0.3779) and Kprim (<italic>p</italic>&#x2009;=&#x2009;0.0099, <italic>r</italic><sub>s</sub>&#x2009;=&#x2009;&#x2212;0.2851). This indicates that questions on which students spent more time generally showed poorer discrimination, suggesting that the time students took to respond to an item might be linked to its ability to discriminate effectively between high- and lower-performing students.</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>Item response times in minutes depending on the respective item discrimination index and separated by item format MCQ (<italic>n</italic>&#x2009;=&#x2009;231, blue) and Kprim (<italic>n</italic>&#x2009;=&#x2009;81, gray). Spearman&#x2019;s rank correlation analysis shows a statistically significant negative correlation between item response time and discrimination index for MCQ (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001, <italic>r</italic><sub>s</sub>&#x2009;=&#x2009;&#x2212;0.3779) and Kprim (<italic>p</italic>&#x2009;=&#x2009;0.0099, <italic>r</italic><sub>s</sub>&#x2009;=&#x2009;&#x2212;0.2851); <italic>n</italic>&#x2009;=&#x2009;312.</p>
</caption>
<graphic xlink:href="fvets-11-1385681-g005.tif"/>
</fig>
<p>Furthermore, <xref ref-type="fig" rid="fig6">Figure 6</xref> depicts the discrimination index of individual MCQ items in relation to their respective character count, where a significant (<italic>p</italic>&#x2009;=&#x2009;0.0236) negative correlation (<italic>r</italic><sub>s</sub>&#x2009;=&#x2009;&#x2212;0.1496) was found for both parameters. Consequently, the discrimination index tended to decrease slightly for MCQ items with a higher character count. However, for Kprim items, no correlation (<italic>p</italic>&#x2009;=&#x2009;0.5534) between the discrimination index and character count could be verified.</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>Item discrimination indexes of MCQ items (<italic>n</italic>&#x2009;=&#x2009;231) depending on their respective character count. Spearman&#x2019;s rank correlation analysis displays a statistically significant negative correlation between discrimination index and character count (<italic>p</italic>&#x2009;=&#x2009;0.0236, <italic>r</italic><sub>s</sub>&#x2009;=&#x2009;&#x2212;0.1496); <italic>n</italic>&#x2009;=&#x2009;231.</p>
</caption>
<graphic xlink:href="fvets-11-1385681-g006.tif"/>
</fig>
<p>Regarding the evaluation of the effect of the length of the question stem, no significant correlation was found between the parameters &#x201C;relative proportion of the question stem&#x201D; and &#x201C;difficulty&#x201D; for MCQ (<italic>n</italic>&#x2009;=&#x2009;231, <italic>p</italic>&#x2009;=&#x2009;0.5627) and Kprim (<italic>n</italic>&#x2009;=&#x2009;81, <italic>p</italic>&#x2009;=&#x2009;0.619) nor for &#x201C;relative proportion of the questions stem&#x201D; and &#x201C;discrimination index&#x201D; for both MCQ (<italic>n</italic>&#x2009;=&#x2009;231, <italic>p</italic>&#x2009;=&#x2009;0.9718) and Kprim (<italic>n</italic>&#x2009;=&#x2009;81, <italic>p</italic>&#x2009;=&#x2009;0.4449).</p>
</sec>
<sec id="sec10">
<label>3.3</label>
<title>Response selection behavior</title>
<p><xref ref-type="table" rid="tab3">Table 3</xref> shows the calculated parameters concerning the relative proportions of changes made to the chosen answer option for each of the five subjects.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Evaluation of the response selection behavior of multiple choice questions (<italic>n</italic>&#x2009;=&#x2009;231) separated by examination subject (A&#x2013;E), including an overall average of all exams.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th/>
<th align="center" valign="top">Subject A<break/>(<italic>n</italic> =&#x2009;50)</th>
<th align="center" valign="top">Subject B<break/>(<italic>n</italic> =&#x2009;68)</th>
<th align="center" valign="top">Subject C<break/>(<italic>n</italic> =&#x2009;46)</th>
<th align="center" valign="top">Subject D<break/>(<italic>n</italic> =&#x2009;25)</th>
<th align="center" valign="top">Subject E<break/>(<italic>n</italic> =&#x2009;42)</th>
<th align="center" valign="top">Average<break/>(<italic>n</italic> =&#x2009;231)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Proportion of items where response selections were changed</td>
<td align="center" valign="top">89.6%</td>
<td align="center" valign="top">98.4%</td>
<td align="center" valign="top">90.2%</td>
<td align="center" valign="top">92.0%</td>
<td align="center" valign="top">100%</td>
<td align="center" valign="top">94.04%</td>
</tr>
<tr>
<td align="left" valign="top">Percentage of examinees who changed at least one of their originally selected answers to a different option</td>
<td align="center" valign="top">88.6%</td>
<td align="center" valign="top">92.7%</td>
<td align="center" valign="top">79.8%</td>
<td align="center" valign="top">67.1%</td>
<td align="center" valign="top">96.2%</td>
<td align="center" valign="top">84.88%</td>
</tr>
<tr>
<td align="left" valign="top">Average proportion of items of the exam for which an examinee changed their response</td>
<td align="center" valign="top">6.25%</td>
<td align="center" valign="top">6.45%</td>
<td align="center" valign="top">4.88%</td>
<td align="center" valign="top">4.00%</td>
<td align="center" valign="top">10.0%</td>
<td align="center" valign="top">6.32%</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The relative proportions of changes between incorrect and correct options are presented for each individual exam in <xref ref-type="table" rid="tab4">Table 4</xref>. The mean values for the changes were 42.89% (SD&#x2009;&#x00B1;&#x2009;3.81%) for incorrect to correct ones, 27.34% (SD&#x2009;&#x00B1;&#x2009;6.71%) for incorrect to incorrect ones, and 29.77% (SD&#x2009;&#x00B1;&#x2009;4.01%) for correct to incorrect ones.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>Evaluation of the quality of response selection modifications separated by corresponding examination subject.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Subject</th>
<th align="center" valign="top">Incorrect to correct (in %)</th>
<th align="center" valign="top">Incorrect to incorrect (in %)</th>
<th align="center" valign="top">Correct to incorrect (in %)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">A</td>
<td align="center" valign="top">40.42</td>
<td align="center" valign="top">28.74</td>
<td align="center" valign="top">30.84</td>
</tr>
<tr>
<td align="left" valign="top">B</td>
<td align="center" valign="top">43.55</td>
<td align="center" valign="top">25.05</td>
<td align="center" valign="top">31.40</td>
</tr>
<tr>
<td align="left" valign="top">C</td>
<td align="center" valign="top">43.64</td>
<td align="center" valign="top">21.59</td>
<td align="center" valign="top">34.77</td>
</tr>
<tr>
<td align="left" valign="top">D</td>
<td align="center" valign="top">49.12</td>
<td align="center" valign="top">21.64</td>
<td align="center" valign="top">29.24</td>
</tr>
<tr>
<td align="left" valign="top">E</td>
<td align="center" valign="top">37.73</td>
<td align="center" valign="top">39.67</td>
<td align="center" valign="top">22.60</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="sec11">
<label>4</label>
<title>Discussion</title>
<sec id="sec12">
<label>4.1</label>
<title>Time of submission of the exams</title>
<p>Within this study, log data of electronic examinations at the TiHo as well as the response selection behavior of the examination participants were evaluated for the first time in order to examine whether recommendations for the design of online examinations can be derived from them.</p>
<p>When examining the submission times of participants for different exams, a wide range of exam completion times becomes evident. Within a time frame of 58&#x2013;90% of the maximum time limit, 75% of the participants managed to complete their exams. The time range for 90% of the students was between 70 and 98% of the available time. This highlights that the time required for exam completion cannot be reliably standardized. However, trends can be derived from the statistical analyses conducted. Notable was the observation that participants tended to finish exams earlier in relation to the maximum available time when there was more time allocated to answer individual items and when the overall exam was easier.</p>
<p>In exams with the longest allocated time to complete items in subjects D (120&#x2009;s per item) and E (90&#x2009;s per item), it is apparent that students generally did not utilize this longer time frame. The majority of candidates managed to complete the exam before 70% of the exam time had elapsed. Thus, theoretically, there is room to reduce the time allocated per item from the 90&#x2009;s per item recommended in the literature (<xref ref-type="bibr" rid="ref12">12</xref>, <xref ref-type="bibr" rid="ref14">14</xref>) and used by the TiHo. However, this reduction needs to consider the correlation between the available exam time and the overall difficulty of the exams reported in this study, as students tend to answer items more accurately when they are granted more time per item, which is also described in Lovett&#x2019;s publication (<xref ref-type="bibr" rid="ref15">15</xref>). The positive impact of additional time per item on students&#x2019; performance was already demonstrated in previous studies (<xref ref-type="bibr" rid="ref16 ref17 ref18 ref19">16&#x2013;19</xref>). Here, the effects are most apparent for students with relatively weaker performance (<xref ref-type="bibr" rid="ref16">16</xref>, <xref ref-type="bibr" rid="ref17">17</xref>, <xref ref-type="bibr" rid="ref19">19</xref>), or those experiencing test anxiety (<xref ref-type="bibr" rid="ref20">20</xref>). This effect might be attributed to reduced exam stress (<xref ref-type="bibr" rid="ref21">21</xref>), more time to consider items, ample time for item completion (<xref ref-type="bibr" rid="ref22">22</xref>) as well as reduced test anxiety (<xref ref-type="bibr" rid="ref20">20</xref>, <xref ref-type="bibr" rid="ref21">21</xref>), which is said to lead to difficulties in concentration and impaired information processing skills (<xref ref-type="bibr" rid="ref20">20</xref>). In this context, a reduction in exam time should be carefully considered with regard to exam fairness. Particularly with regard to the mentioned exam fairness, it must be noted that increasing time pressure can put certain other groups of examinees at a disadvantage. For example, studies indicate that increasing time pressure has a more substantial negative impact on the performance of female examinees than male participants (<xref ref-type="bibr" rid="ref23">23</xref>, <xref ref-type="bibr" rid="ref24">24</xref>). Furthermore, students with learning, cognitive or psychiatric disabilities such as ADHD (<xref ref-type="bibr" rid="ref15">15</xref>, <xref ref-type="bibr" rid="ref25">25</xref>), Asperger&#x2019;s syndrome (<xref ref-type="bibr" rid="ref25">25</xref>) or dyslexia (<xref ref-type="bibr" rid="ref9">9</xref>) are entitled to accommodations for disadvantages, which is often in the form of an extended time limit (<xref ref-type="bibr" rid="ref15">15</xref>), since the negative impact of time pressure is especially evident for students with disabilities (<xref ref-type="bibr" rid="ref26">26</xref>).</p>
</sec>
<sec id="sec13">
<label>4.2</label>
<title>Item response time</title>
<p>Examining the actual time taken by candidates to complete the exams on the level of individual items also indicates that candidates did not utilize the time limit of 90&#x2009;s per item. The average response time ranged between 47 and 57&#x2009;s per item for all veterinary subjects. This observation aligns with the findings of other analyses, consistently reporting a time of 60&#x2009;s or less per multiple-choice item (<xref ref-type="bibr" rid="ref27 ref28 ref29 ref30">27&#x2013;30</xref>). In some cases, the required response time per item was approximately 40&#x2009;s or less (<xref ref-type="bibr" rid="ref28 ref29 ref30">28&#x2013;30</xref>). Therefore, there is an opportunity of reducing the exam time limit to 60&#x2009;s per item, a practice already standard in some other exams (<xref ref-type="bibr" rid="ref28">28</xref>, <xref ref-type="bibr" rid="ref31">31</xref>, <xref ref-type="bibr" rid="ref32">32</xref>). On the one hand, this would allow more items to be included in the same time frame, thus improving the validity and reliability of exams (<xref ref-type="bibr" rid="ref12">12</xref>, <xref ref-type="bibr" rid="ref33 ref34 ref35 ref36 ref37">33&#x2013;37</xref>), and, on the other hand, a stricter time limit can be used as a tool to reduce interchange between candidates in unproctored online examinations (<xref ref-type="bibr" rid="ref9">9</xref>, <xref ref-type="bibr" rid="ref38">38</xref>).</p>
<p>It is important to note that three additional aspects need to be considered when determining time limits for exams and individual items.</p>
<p>Firstly, exams in the medical field are intended to test students&#x2019; knowledge, understanding, and application of knowledge and should thus be conducted as power tests (<xref ref-type="bibr" rid="ref12">12</xref>). When time constraints are introduced or exams are conducted under high time pressure, factors such as candidates&#x2019; stress resistance and cognitive performance affect overall performance (<xref ref-type="bibr" rid="ref9">9</xref>, <xref ref-type="bibr" rid="ref12">12</xref>). Consequently, the difficulty of exams increases (<xref ref-type="bibr" rid="ref39">39</xref>, <xref ref-type="bibr" rid="ref40">40</xref>), and a potential time shortage can result in unanswered questions or blind guessing, raising concerns about the validity of exam results (<xref ref-type="bibr" rid="ref12">12</xref>, <xref ref-type="bibr" rid="ref22">22</xref>, <xref ref-type="bibr" rid="ref41">41</xref>). In light of the examination objective in veterinary medicine, so-called speed tests where the time factor significantly impacts candidates&#x2019; performance should be avoided.</p>
<p>Secondly, the influence of exam time limits on performance, primarily observed in weaker students, should be considered to preserve exam fairness. A shorter time frame can lead to poorer test results.</p>
<p>Lastly, the ratio of different item formats should be mentioned (<xref ref-type="bibr" rid="ref42">42</xref>) since significant differences in response time for formats were shown. Especially for answering the three formats Kprim (74&#x2009;s), picture diagnosis (77&#x2009;s), and picture mapping (87&#x2009;s), students needed on average more than 60&#x2009;s of response time, while MCQ (43&#x2009;s) and Key Feature (44&#x2009;s per subquestion) required less than 60&#x2009;s. Other studies also conclude that formats like picture diagnosis and picture mapping are more time-consuming for candidates than MCQ (<xref ref-type="bibr" rid="ref43">43</xref>, <xref ref-type="bibr" rid="ref44">44</xref>). Therefore, when determining the time limits for exams, it is essential to consider the item formats used and their ratio (<xref ref-type="bibr" rid="ref42">42</xref>). This ratio is particularly important if aiming for the aforementioned reduction of response time to 60&#x2009;s per item. Since students require significantly longer than 60&#x2009;s on average for Kprim and image-based formats, there is a risk that an exam with a high proportion of these formats becomes a speed test. Hence, when limiting it to 60&#x2009;s per item, attention should be given to predominantly select MCQ items, as less than 60&#x2009;s are sufficient for this format, thereby balancing the additional time required for Kprim and image-based formats.</p>
<p>Furthermore, such exams can be evaluated to determine whether time significantly impacts students&#x2019; performance. The Education Testing Service defines a multiple-choice exam as a power test if all participants answer at least 75% of all items, and 80% of participants respond to all of the items (<xref ref-type="bibr" rid="ref45">45</xref>). Otherwise, it is considered a speed test if these criteria are not met. However, this rule of thumb was developed based on paper-based exams and the assumption that examinees do not complete the exam if they run out of time. In the context of electronic multiple-choice tests where an incorrect answer does not result in a negative grading, it is more likely that examinees randomly select answer options for the remaining items to still have a chance to randomly choose the correct option. This is referred to as rapid guessing behavior (<xref ref-type="bibr" rid="ref22">22</xref>, <xref ref-type="bibr" rid="ref44">44</xref>). Modern analysis methods use the log data from electronic assessments to identify rapid guessing behavior. Schnipke (<xref ref-type="bibr" rid="ref22">22</xref>) graphed the standardized natural logarithm of response time to detect examinees with accelerated response times and a lower frequency of correct answers, indicating that these students might be running out of time. In addition, other complex models and methods, including those based on Item Response Theory, have been developed to assess the time influence on students&#x2019; exam performance and behavior (<xref ref-type="bibr" rid="ref41">41</xref>, <xref ref-type="bibr" rid="ref44">44</xref>).</p>
<p>Regarding the analysis of character counts, response time, and item parameters of the tasks and their correlations, it was demonstrated that students need more time for items with high difficulty, poor discrimination index, and a higher number of characters. Additionally, it was shown that MCQ items with a higher character count tend to have a poorer discrimination index, but that the character count has no effect on difficulty. These effects have been described in other studies that also concluded that examinees require more time for poorly discriminating items (<xref ref-type="bibr" rid="ref30">30</xref>, <xref ref-type="bibr" rid="ref46">46</xref>) as well as for difficult items (<xref ref-type="bibr" rid="ref46">46</xref>, <xref ref-type="bibr" rid="ref47">47</xref>). As a result, to improve discrimination and reduce required response time, items should be kept as clear and concise as possible, which aligns with the formal requirements of multiple-choice items in the literature (<xref ref-type="bibr" rid="ref12">12</xref>, <xref ref-type="bibr" rid="ref28">28</xref>, <xref ref-type="bibr" rid="ref48 ref49 ref50 ref51 ref52">48&#x2013;52</xref>). However, focusing on the brevity of items should not be at the expense of the learning content to be assessed, as it is outlined in Bloom&#x2019;s taxonomy (<xref ref-type="bibr" rid="ref53">53</xref>).</p>
<p>Bloom&#x2019;s taxonomy categorizes cognitive skills into six levels: remembering, understanding, applying, analyzing, evaluating, and creating (<xref ref-type="bibr" rid="ref53">53</xref>). Questions designed to test basic recall or understanding can indeed be kept short and precise, reducing both the time needed for students to respond to these items and the time required for question authors to create them. For example, a straightforward multiple-choice question requiring students to recall a specific fact or definition can be brief without sacrificing its effectiveness. In contrast, questions that aim to assess higher-order cognitive skills, such as applying knowledge to new situations, analyzing data, or evaluating concepts, often necessitate a more detailed question stem (<xref ref-type="bibr" rid="ref50">50</xref>, <xref ref-type="bibr" rid="ref52">52</xref>). For instance, regarding the clinical sciences of veterinary medicine, presenting a comprehensive scenario that provides sufficient context to assess students&#x2019; critical thinking and problem-solving abilities is essential. Such scenarios might include clinical findings, laboratory results or detailed case vignettes (<xref ref-type="bibr" rid="ref50">50</xref>). These elements are crucial for testing students&#x2019; abilities to integrate and apply their knowledge but inevitably lead to an increased character count of these items. Subsequently, a careful balance is necessary when creating questions that are able to assess these higher-level cognitive skills. While the objective is still to keep items as concise as possible, it is equally important to ensure that all necessary information is included to allow students to fully understand and respond to the question.</p>
<p>Consequently, an efficient question design is recommended to reduce the time needed for authors to create items and for students to answer them, which can make it feasible to reduce the time per item and thus include more items in the same timeframe as before, which positively impacts the quality criteria of exams (<xref ref-type="bibr" rid="ref52">52</xref>).</p>
<p>Applying these findings to the design of online assessments, a time limit of 60&#x2009;s can be set for this format as long as the exam is primarily composed of MCQs and Key Feature items, which are short and concise. The shorter allocated time per item can help reduce the risk of cheating. However, it is essential to ensure that students do not feel overly rushed. Some authors even discuss explicitly designing exams to create significant time pressure on participants as a means to reduce cheating attempts (<xref ref-type="bibr" rid="ref9">9</xref>, <xref ref-type="bibr" rid="ref38">38</xref>, <xref ref-type="bibr" rid="ref54">54</xref>). This statement should be viewed critically based on the aspects discussed earlier. When it comes to students with disabilities (see section 4.1), reducing the time per item must be approached with caution, as this can create additional challenges in administering exams (<xref ref-type="bibr" rid="ref15">15</xref>, <xref ref-type="bibr" rid="ref25">25</xref>). In compliance with local laws and regulations, appropriate accommodations must be provided, which typically involve extending the exam time to allow students to utilize these accommodations effectively (<xref ref-type="bibr" rid="ref15">15</xref>). Decision-making processes should be established to determine whether, and to what extent, accommodations are offered to disadvantaged students, as well as how much extra time is actually necessary (<xref ref-type="bibr" rid="ref15">15</xref>). Additionally, the technical and organizational challenges of granting extra time to only some students must be considered (<xref ref-type="bibr" rid="ref25">25</xref>).</p>
</sec>
<sec id="sec14">
<label>4.3</label>
<title>Response selection behavior</title>
<p>The analysis of the log data from the five exams indicates that for almost all of the items there was at least one student changing their answer and that the majority of students belonged to the group that revisited and modified the originally selected response option. However, it is noteworthy that each individual student changed their originally selected answers only for a very small proportion of items. These findings based on the data from veterinary students corroborate the results of other studies, which also conclude that nearly all students changed their original answers to a different one, yet each individual participant only makes corrections for a small fraction of all exam items (<xref ref-type="bibr" rid="ref55">55</xref>).</p>
<p>One potential influencing factor on this low frequency of modified answers could be the widespread belief that the initial intuition when reading the question is the correct answer. Benjamin et al. (<xref ref-type="bibr" rid="ref55">55</xref>) reported in their literature review that almost all students believe that revising their answer does not impact the exam result; 75% were even of the opinion that the result would deteriorate. The teaching staff shared a similar belief, with over half of the respondents stating that exam results would worsen due to answer modifications (<xref ref-type="bibr" rid="ref55">55</xref>). In contrast, studies demonstrate that the majority of answer revisions are from incorrect to correct responses, resulting in improved exam results (<xref ref-type="bibr" rid="ref55 ref56 ref57 ref58 ref59 ref60">55&#x2013;60</xref>). Hence, this phenomenon is termed the &#x201C;first instinct fallacy&#x201D; (<xref ref-type="bibr" rid="ref55">55</xref>, <xref ref-type="bibr" rid="ref57 ref58 ref59">57&#x2013;59</xref>). The analysis of the response selection behavior in this study reaches the same conclusion, as the majority of all corrections were either from incorrect to correct options or between two different incorrect options, resulting in a positive or neutral impact on exam scores.</p>
<p>Accordingly, the available time for exams should not be stipulated within such a short timeframe that participants lose the opportunity to adjust their response selections due to time constraints. For online examinations, one study recommends that students should only be allowed to select an answer for an item once only (<xref ref-type="bibr" rid="ref38">38</xref>). We do not follow this recommendation based on the results of our analysis, as not being able to change the originally selected answer could potentially worsen the exam results.</p>
<p>As an alternative, a so-called &#x201C;block setting&#x201D; can be used for electronic exams. In this approach, the exam is divided into equally sized groups of items, and the items within the blocks as well as the blocks themselves can be individually randomized for each examinee. Access is only granted to the items in the currently worked-on block, which can be accessed and modified multiple times. After completing the block, the following block is unlocked and the previous item block is locked. This means that students no longer have access to the previously worked-on items. This approach allows exams to be structured better, the students focusing their concentration on the items within the block. Randomizing the blocks and the included items can be used to reduce cheating (<xref ref-type="bibr" rid="ref38">38</xref>).</p>
</sec>
<sec id="sec15">
<label>4.4</label>
<title>Implications and limitations of the study</title>
<p>A limitation of this study was the small sample size for the image-based item formats, namely picture diagnosis and picture mapping. The reason for that was that for the sake of comparability exams were chosen in which all five item formats were used and had already been established for several years. However, the image-based formats were less frequently employed in these exams. In future studies, a greater emphasis should be placed on a more in-depth evaluation of the two formats picture diagnosis and picture mapping.</p>
<p>Since the focus was primarily on analyzing log data and response selection behavior of electronic examinations, more research is needed regarding the evaluation of the psychometric quality of item formats in online examinations and thereby identify suitable formats for distance exams. As both lecturers and students in veterinary medicine call for more application- and competence-based item formats (<xref ref-type="bibr" rid="ref61">61</xref>, <xref ref-type="bibr" rid="ref62">62</xref>), these would be of particular interest, for example key feature and the image-based formats referenced in this work.</p>
<p>Regarding the items included in this study and their item analysis parameters, it is noticeable that some items display a low discrimination index, and a few items exhibit a negative discrimination index (see <xref ref-type="fig" rid="fig5">Figure 5</xref>). This indicates that these items might not effectively differentiate between higher-performing and lower-performing students, according to Krebs (<xref ref-type="bibr" rid="ref12">12</xref>) as well as McCowan and McCowan (<xref ref-type="bibr" rid="ref13">13</xref>). Negative discrimination indices imply that students understood these items in the opposite way than originally intended, meaning good students answered incorrectly while lower-performing students chose the correct answer (<xref ref-type="bibr" rid="ref12">12</xref>). Therefore, such a negative discrimination index is an indicator of potential flaws in item construction or a lack of alignment with the learning objectives. To address this issue, the content of items with negative discrimination was reviewed for correctness, relevance, formal errors, and cueing, including a distractor analysis. However, no significant issues were identified. The low discrimination indices can partly be attributed to some items being relatively difficult (<italic>p</italic>&#x2009;&#x003C;&#x2009;40%) and some items being relatively easy (<italic>p</italic>&#x2009;&#x003E;&#x2009;80%) (see <xref ref-type="fig" rid="fig4">Figure 4</xref>). The best discrimination indices are found in items with medium difficulty, and deviations upwards or downwards result in significantly poorer discrimination values (<xref ref-type="bibr" rid="ref12">12</xref>). Regardless of the reason, it must be noted that such items directly impact the calculated correlations of this study.</p>
<p>Open-book exams are a format commonly used for electronic distance assessments. Over the course of the COVID-19 pandemic, they gained importance as online proctoring is not necessarily required for this format (<xref ref-type="bibr" rid="ref4">4</xref>), which resolves data protection issues associated with remote examinations. Comparative analyses already showed that the psychometric parameters of closed-book in-person exams and online open-book exams do not significantly differ (<xref ref-type="bibr" rid="ref54">54</xref>, <xref ref-type="bibr" rid="ref63 ref64 ref65">63&#x2013;65</xref>). Hence, according to the mentioned literature, this innovative format can be considered suitable for summative assessments. However, it is crucial to take into account the current developments in the field of artificial intelligence (AI). Generative AIs, particularly ChatGPT, are capable of successfully passing challenging final and licensure exams, including several law bar exams (<xref ref-type="bibr" rid="ref66">66</xref>), the United States Medical Licensing Exam (<xref ref-type="bibr" rid="ref67">67</xref>), the German medical state exam (<xref ref-type="bibr" rid="ref68">68</xref>), and other assessments (<xref ref-type="bibr" rid="ref69">69</xref>, <xref ref-type="bibr" rid="ref70">70</xref>). In Progress Tests, ChatGPT answered over 60% of the items correctly (<xref ref-type="bibr" rid="ref71">71</xref>). It can be assumed that this performance can also be reproduced in the field of veterinary medicine. Given that such AIs are readily available to students as well (<xref ref-type="bibr" rid="ref72">72</xref>), the feasibility of distance assessments needs to be critically examined and re-evaluated (<xref ref-type="bibr" rid="ref72 ref73 ref74">72&#x2013;74</xref>).</p>
</sec>
</sec>
<sec sec-type="data-availability" id="sec16">
<title>Data availability statement</title>
<p>The datasets presented in this article are not readily available because they include legally protected personal and examination information. Requests to access the datasets should be directed to RR, <email>richter.robin@tiho-hannover.de</email>.</p>
</sec>
<sec sec-type="ethics-statement" id="sec17">
<title>Ethics statement</title>
<p>This study was conducted according to the ethical standards of the University of Veterinary Medicine Hannover, Foundation. The doctoral thesis committee of the university, which acts as the university&#x2019;s ethics committee, validated the project in accordance with ethical guidelines regarding research with human participants and approved the study.</p>
</sec>
<sec sec-type="author-contributions" id="sec18">
<title>Author contributions</title>
<p>RR: Conceptualization, Investigation, Methodology, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. AT: Writing &#x2013; review &#x0026; editing, Funding acquisition, Supervision. ES: Supervision, Writing &#x2013; review &#x0026; editing, Funding acquisition, Methodology, Project administration.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="sec19">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. We acknowledge financial support by the Open Access Publication Fund of the University of Veterinary Medicine Hannover, Foundation. This publication was supported by Stiftung Innovation in der Hochschullehre within the project &#x201C;FERVET &#x2013; Digital Teaching and Review of Clinical Practical Skills in Veterinary Medicine from an Animal Welfare Perspective.&#x201D;</p>
</sec>
<sec sec-type="COI-statement" id="sec20">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec sec-type="disclaimer" id="sec21">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1">
<label>1.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Marinoni</surname> <given-names>G</given-names></name> <name><surname>Vant Land</surname> <given-names>H</given-names></name> <name><surname>Jensen</surname> <given-names>T</given-names></name></person-group>. <article-title>The impact of Covid-19 on higher education around the world</article-title>. <source>IAU Global Survey Rep</source>. (<year>2020</year>) <volume>23</volume>:<fpage>1</fpage>&#x2013;<lpage>17</lpage>.</citation>
</ref>
<ref id="ref2">
<label>2.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Seyfeli</surname> <given-names>F</given-names></name> <name><surname>Elsner</surname> <given-names>L</given-names></name> <name><surname>Wannemacher</surname> <given-names>K</given-names></name></person-group>. <source>Vom Corona-Shutdown Zur Blended University?: Expertinnenbefragung Digitales Sommersemester</source>. <edition>1st</edition> ed. <publisher-loc>Baden-Baden, Germany</publisher-loc>: <publisher-name>Tectum</publisher-name> (<year>2020</year>).</citation>
</ref>
<ref id="ref3">
<label>3.</label>
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Wissing</surname> <given-names>F</given-names></name>
</person-group>. <article-title>Digitale Lehre f&#x00FC;r alle: Voraussetzungen, Machbarkeit und Optionen im Human- und Zahnmedizinstudium</article-title>. <source>Medizinischer Fakult&#x00E4;tentag</source>. (<year>2020</year>). Available at: <ext-link xlink:href="https://medizinische-fakultaeten.de/wp-content/uploads/2020/10/MFT-und-GMA-Positionspapier-zu-digitalen-Lehr-und-Pru%CC%88fungsformaten.pdf" ext-link-type="uri">https://medizinische-fakultaeten.de/wp-content/uploads/2020/10/MFT-und-GMA-Positionspapier-zu-digitalen-Lehr-und-Pru%CC%88fungsformaten.pdf</ext-link></citation>
</ref>
<ref id="ref4">
<label>4.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Routh</surname> <given-names>J</given-names></name> <name><surname>Paramasivam</surname> <given-names>SJ</given-names></name> <name><surname>Cockcroft</surname> <given-names>P</given-names></name> <name><surname>Nadarajah</surname> <given-names>VD</given-names></name> <name><surname>Jeevaratnam</surname> <given-names>K</given-names></name></person-group>. <article-title>Veterinary education during Covid-19 and beyond-challenges and mitigating approaches</article-title>. <source>Animals</source>. (<year>2021</year>) <volume>11</volume>:<fpage>1818</fpage>. doi: <pub-id pub-id-type="doi">10.3390/ani11061818</pub-id>, PMID: <pub-id pub-id-type="pmid">34207202</pub-id></citation>
</ref>
<ref id="ref5">
<label>5.</label>
<citation citation-type="book"><person-group person-group-type="author">
<name><surname>Gnewuch</surname> <given-names>L</given-names></name>
</person-group>. <source>Digitalisierung der Lehre&#x2013; Situationsanalyse und Perspektiven in der Veterin&#x00E4;rmedizin</source>. <comment>[Dissertation]</comment>. <publisher-loc>Berlin, Germany</publisher-loc>: <publisher-name>Freie Universit&#x00E4;t Berlin</publisher-name> (<year>2023</year>).</citation>
</ref>
<ref id="ref6">
<label>6.</label>
<citation citation-type="book"><person-group person-group-type="author">
<name><surname>Naundorf</surname> <given-names>H</given-names></name>
</person-group>. <source>Untersuchung der Hybridsemester-Lehre w&#x00E4;hrend der Covid-19 Pandemie an der Stiftung Tier&#x00E4;rztliche Hochschule Hannover</source>. <comment>[Dissertation]</comment>. <publisher-loc>Hannover, Germany</publisher-loc>: <publisher-name>Stiftung Tier&#x00E4;rztliche Hochschule Hannover</publisher-name> (<year>2023</year>).</citation>
</ref>
<ref id="ref7">
<label>7.</label>
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Gatti</surname> <given-names>T</given-names></name> <name><surname>Helm</surname> <given-names>F</given-names></name> <name><surname>Huskobla</surname> <given-names>G</given-names></name> <name><surname>Maciejowska</surname> <given-names>D</given-names></name> <name><surname>McGeever</surname> <given-names>B</given-names></name> <name><surname>Pincemin</surname> <given-names>J-M</given-names></name> <etal/></person-group>. Practices at Coimbra group universities in response to the COVID-19: a collective reflection on the present and future of higher education in Europe. (<year>2020</year>). <comment>Available at:</comment> <ext-link xlink:href="https://www.coimbra-group.eu/wp-content/uploads/Final-Report-Practices-at-CG-Universities-in-response-to-the-COVID-19-3.pdf" ext-link-type="uri">https://www.coimbra-group.eu/wp-content/uploads/Final-Report-Practices-at-CG-Universities-in-response-to-the-COVID-19-3.pdf</ext-link></citation>
</ref>
<ref id="ref8">
<label>8.</label>
<citation citation-type="other"><person-group person-group-type="author">
<name><surname>Thiel</surname> <given-names>B.</given-names></name>
</person-group> Eckpunkte F&#x00FC;r Datenschutzkonforme Online-Pr&#x00FC;fungen an Nieders&#x00E4;chsischen Hochschulen. Hannover. (<year>2021</year>). <comment>Available at:</comment> <ext-link xlink:href="https://lfd.niedersachsen.de/startseite/themen/weitere_themen_von_a_z/hochschulen/eckpunkte_fur_die_datenschutzkonforme_durchfuhrung_von_online_prufungen/" ext-link-type="uri">https://lfd.niedersachsen.de/startseite/themen/weitere_themen_von_a_z/hochschulen/eckpunkte_fur_die_datenschutzkonforme_durchfuhrung_von_online_prufungen/</ext-link></citation>
</ref>
<ref id="ref9">
<label>9.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stadler</surname> <given-names>M</given-names></name> <name><surname>Kolb</surname> <given-names>N</given-names></name> <name><surname>Sailer</surname> <given-names>M</given-names></name></person-group>. <article-title>The right amount of pressure: implementing time pressure in online exams</article-title>. <source>Distance Educ</source>. (<year>2021</year>) <volume>42</volume>:<fpage>219</fpage>&#x2013;<lpage>30</lpage>. doi: <pub-id pub-id-type="doi">10.1080/01587919.2021.1911629</pub-id></citation>
</ref>
<ref id="ref10">
<label>10.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ehlers</surname> <given-names>JP</given-names></name> <name><surname>Carl</surname> <given-names>T</given-names></name> <name><surname>Windt</surname> <given-names>K-H</given-names></name> <name><surname>M&#x00F6;bs</surname> <given-names>D</given-names></name> <name><surname>Rehage</surname> <given-names>J</given-names></name> <name><surname>Tipold</surname> <given-names>A</given-names></name></person-group>. <article-title>Blended Assessment: M&#x00FC;ndliche Und Elektronische Pr&#x00FC;fungen Im Klinischen Kontext</article-title>. <source>Zeitschrift f&#x00FC;r Hochschulentwicklung</source>. (<year>2010</year>) <volume>4</volume>:<fpage>24</fpage>&#x2013;<lpage>36</lpage>. doi: <pub-id pub-id-type="doi">10.3217/zfhe-4-03/02</pub-id></citation>
</ref>
<ref id="ref11">
<label>11.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Thorndike</surname> <given-names>RM</given-names></name> <name><surname>Cunningham</surname> <given-names>GK</given-names></name> <name><surname>Thorndike</surname> <given-names>RL</given-names></name> <name><surname>Hagen</surname> <given-names>EP</given-names></name></person-group>. <source>Measurement and evaluation in psychology and education</source>. <edition>5th</edition> ed. <publisher-loc>New York, NY, England</publisher-loc>: <publisher-name>Macmillan Publishing Co, Inc</publisher-name> (<year>1991</year>). <fpage>544</fpage> p.</citation>
</ref>
<ref id="ref12">
<label>12.</label>
<citation citation-type="book"><person-group person-group-type="author">
<name><surname>Krebs</surname> <given-names>R</given-names></name>
</person-group>. <source>Pr&#x00FC;fen Mit Multiple Choice. Kompetent Planen, Entwickeln, Durchf&#x00FC;hren Und Auswerten</source>. <edition>1st</edition> ed. <publisher-loc>Bern, Austria</publisher-loc>: <publisher-name>Hogrefe</publisher-name> (<year>2019</year>).</citation>
</ref>
<ref id="ref13">
<label>13.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>McCowan</surname> <given-names>RJ</given-names></name> <name><surname>McCowan</surname> <given-names>SC</given-names></name></person-group>. <source>Item analysis for criterion-referenced tests</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Center for Development of Human Services (CDHS)</publisher-name> (<year>1999</year>).</citation>
</ref>
<ref id="ref14">
<label>14.</label>
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Harden</surname> <given-names>RM</given-names></name>
</person-group>. <article-title>Constructing multiple choice questions of the multiple true/false type</article-title>. <source>Med Educ</source>. (<year>1979</year>) <volume>13</volume>:<fpage>305</fpage>&#x2013;<lpage>12</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1365-2923.1979.tb01517.x</pub-id>, PMID: <pub-id pub-id-type="pmid">470653</pub-id></citation>
</ref>
<ref id="ref15">
<label>15.</label>
<citation citation-type="book"><person-group person-group-type="author">
<name><surname>Lovett</surname> <given-names>BJ</given-names></name>
</person-group>. <article-title>Extended time testing accommodations for students with disabilities: impact on score meaning and construct representation</article-title> In: <person-group person-group-type="editor"><name><surname>Margolis</surname> <given-names>MJ</given-names></name> <name><surname>Feinberg</surname> <given-names>RA</given-names></name></person-group>, editors. <source>Integrating timing considerations to improve testing practices</source>. <publisher-loc>Oxfordshire</publisher-loc>: <publisher-name>Routledge</publisher-name> (<year>2020</year>). <fpage>47</fpage>&#x2013;<lpage>58</lpage>.</citation>
</ref>
<ref id="ref16">
<label>16.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mitchell</surname> <given-names>G</given-names></name> <name><surname>Ford</surname> <given-names>DM</given-names></name> <name><surname>Prinz</surname> <given-names>W</given-names></name></person-group>. <article-title>Optimising marks obtained in multiple choice question examinations</article-title>. <source>Med Teach</source>. (<year>1986</year>) <volume>8</volume>:<fpage>49</fpage>&#x2013;<lpage>53</lpage>. doi: <pub-id pub-id-type="doi">10.3109/01421598609036845</pub-id>, PMID: <pub-id pub-id-type="pmid">3724402</pub-id></citation>
</ref>
<ref id="ref17">
<label>17.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bridgeman</surname> <given-names>B</given-names></name> <name><surname>Cline</surname> <given-names>F</given-names></name> <name><surname>Hessinger</surname> <given-names>J</given-names></name></person-group>. <article-title>Effect of extra time on verbal and quantitative Gre scores</article-title>. <source>Appl Meas Educ</source>. (<year>2004</year>) <volume>17</volume>:<fpage>25</fpage>&#x2013;<lpage>37</lpage>. doi: <pub-id pub-id-type="doi">10.1207/s15324818ame1701_2</pub-id></citation>
</ref>
<ref id="ref18">
<label>18.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cuddy</surname> <given-names>MM</given-names></name> <name><surname>Swanson</surname> <given-names>DB</given-names></name> <name><surname>Dillon</surname> <given-names>GF</given-names></name> <name><surname>Holtman</surname> <given-names>MC</given-names></name> <name><surname>Clauser</surname> <given-names>BE</given-names></name></person-group>. <article-title>A multilevel analysis of the relationships between selected examinee characteristics and United States medical licensing examination step 2 clinical knowledge performance: revisiting old findings and asking new questions</article-title>. <source>Acad Med</source>. (<year>2006</year>) <volume>81</volume>:<fpage>103</fpage>&#x2013;<lpage>7</lpage>. doi: <pub-id pub-id-type="doi">10.1097/00001888-200610001-00026</pub-id></citation>
</ref>
<ref id="ref19">
<label>19.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Harik</surname> <given-names>P</given-names></name> <name><surname>Clauser</surname> <given-names>BE</given-names></name> <name><surname>Grabovsky</surname> <given-names>I</given-names></name> <name><surname>Baldwin</surname> <given-names>P</given-names></name> <name><surname>Margolis</surname> <given-names>MJ</given-names></name> <name><surname>Bucak</surname> <given-names>D</given-names></name> <etal/></person-group>. <article-title>A comparison of experimental and observational approaches to assessing the effects of time constraints in a medical licensing examination</article-title>. <source>J Educ Meas</source>. (<year>2018</year>) <volume>55</volume>:<fpage>308</fpage>&#x2013;<lpage>27</lpage>. doi: <pub-id pub-id-type="doi">10.1111/jedm.12177</pub-id></citation>
</ref>
<ref id="ref20">
<label>20.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Onwuegbuzie</surname> <given-names>AJ</given-names></name> <name><surname>Seaman</surname> <given-names>MA</given-names></name></person-group>. <article-title>The effect of time constraints and statistics test anxiety on test performance in a statistics course</article-title>. <source>J Exp Educ</source>. (<year>1995</year>) <volume>63</volume>:<fpage>115</fpage>&#x2013;<lpage>24</lpage>. doi: <pub-id pub-id-type="doi">10.1080/00220973.1995.9943816</pub-id></citation>
</ref>
<ref id="ref21">
<label>21.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Portolese</surname> <given-names>L</given-names></name> <name><surname>Krause</surname> <given-names>J</given-names></name> <name><surname>Bonner</surname> <given-names>J</given-names></name></person-group>. <article-title>Timed online tests: do students perform better with more time?</article-title> <source>Am J Dist Educ</source>. (<year>2016</year>) <volume>30</volume>:<fpage>264</fpage>&#x2013;<lpage>71</lpage>. doi: <pub-id pub-id-type="doi">10.1080/08923647.2016.1234301</pub-id></citation>
</ref>
<ref id="ref22">
<label>22.</label>
<citation citation-type="book"><person-group person-group-type="author">
<name><surname>Schnipke</surname> <given-names>DL</given-names></name>
</person-group>. <source>In Paper presented at the Annual Meeting of the National Council on Measurement in Education</source>. <publisher-loc>San Francisco, CA, USA</publisher-loc>. (<year>1995</year>). <fpage>2</fpage>&#x2013;<lpage>322</lpage>.</citation>
</ref>
<ref id="ref23">
<label>23.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Steinmayr</surname> <given-names>R</given-names></name> <name><surname>Spinath</surname> <given-names>B</given-names></name></person-group>. <article-title>Why time constraints increase the gender gap in measured numerical intelligence in academically high achieving samples</article-title>. <source>Eur J Psychol Assess</source>. (<year>2019</year>) <volume>35</volume>:<fpage>392</fpage>&#x2013;<lpage>402</lpage>. doi: <pub-id pub-id-type="doi">10.1027/1015-5759/a000400</pub-id></citation>
</ref>
<ref id="ref24">
<label>24.</label>
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Voyer</surname> <given-names>D</given-names></name>
</person-group>. <article-title>Time limits and gender differences on paper-and-pencil tests of mental rotation: a meta-analysis</article-title>. <source>Psychon Bull Rev</source>. (<year>2011</year>) <volume>18</volume>:<fpage>267</fpage>&#x2013;<lpage>77</lpage>. doi: <pub-id pub-id-type="doi">10.3758/s13423-010-0042-0</pub-id>, PMID: <pub-id pub-id-type="pmid">21327340</pub-id></citation>
</ref>
<ref id="ref25">
<label>25.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Persike</surname> <given-names>M</given-names></name> <name><surname>G&#x00FC;nther</surname> <given-names>S</given-names></name> <name><surname>Dohr</surname> <given-names>J</given-names></name> <name><surname>Dorok</surname> <given-names>P</given-names></name> <name><surname>Rampelt</surname> <given-names>F</given-names></name></person-group>. <article-title>Digitale Fernpr&#x00FC;fungen / Online-Pr&#x00FC;fungen au&#x00DF;erhalb der Hochschule</article-title> In: <person-group person-group-type="editor"><name><surname>Bandtel</surname> <given-names>M</given-names></name> <name><surname>Baume</surname> <given-names>M</given-names></name> <name><surname>Brinkmann</surname> <given-names>E</given-names></name> <name><surname>Bedenlier</surname> <given-names>S</given-names></name> <name><surname>Budde</surname> <given-names>J</given-names></name> <name><surname>Eugster</surname> <given-names>B</given-names></name> <etal/></person-group>, editors. <source>Digitale Pr&#x00FC;fungen in der Hochschule. Whitepaper einer Community Working Group aus Deutschland, &#x00D6;sterreich und der Schweiz</source>. <publisher-loc>Berlin, DE</publisher-loc>: <publisher-name>Hochschulforum Digitalisierung</publisher-name> (<year>2021</year>). <fpage>81</fpage>&#x2013;<lpage>91</lpage>.</citation>
</ref>
<ref id="ref26">
<label>26.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Waterfield</surname> <given-names>J</given-names></name> <name><surname>West</surname> <given-names>B</given-names></name></person-group>. <source>Inclusive assessment in higher education: a resource for change</source>. <publisher-loc>Plymouth, UK</publisher-loc>: <publisher-name>University of Plymouth</publisher-name> (<year>2006</year>).</citation>
</ref>
<ref id="ref27">
<label>27.</label>
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Cui</surname> <given-names>Z</given-names></name>
</person-group>. <article-title>On the cover: time spent on multiple-choice items</article-title>. <source>Educ Meas Issues Pract</source>. (<year>2021</year>) <volume>40</volume>:<fpage>6</fpage>&#x2013;<lpage>7</lpage>. doi: <pub-id pub-id-type="doi">10.1111/emip.12420</pub-id></citation>
</ref>
<ref id="ref28">
<label>28.</label>
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Brothen</surname> <given-names>T</given-names></name>
</person-group>. <article-title>Time limits on tests: updating the 1-minute rule</article-title>. <source>Teach Psychol</source>. (<year>2012</year>) <volume>39</volume>:<fpage>288</fpage>&#x2013;<lpage>92</lpage>. doi: <pub-id pub-id-type="doi">10.1177/0098628312456630</pub-id>, PMID: <pub-id pub-id-type="pmid">38309959</pub-id></citation>
</ref>
<ref id="ref29">
<label>29.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schneid</surname> <given-names>SD</given-names></name> <name><surname>Armour</surname> <given-names>C</given-names></name> <name><surname>Park</surname> <given-names>YS</given-names></name> <name><surname>Yudkowsky</surname> <given-names>R</given-names></name> <name><surname>Bordage</surname> <given-names>G</given-names></name></person-group>. <article-title>Reducing the number of options on multiple-choice questions: response time, psychometrics and standard setting</article-title>. <source>Med Educ</source>. (<year>2014</year>) <volume>48</volume>:<fpage>1020</fpage>&#x2013;<lpage>7</lpage>. doi: <pub-id pub-id-type="doi">10.1111/medu.12525</pub-id>, PMID: <pub-id pub-id-type="pmid">25200022</pub-id></citation>
</ref>
<ref id="ref30">
<label>30.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chae</surname> <given-names>YM</given-names></name> <name><surname>Park</surname> <given-names>SG</given-names></name> <name><surname>Park</surname> <given-names>I</given-names></name></person-group>. <article-title>The relationship between classical item characteristics and item response time on computer-based testing</article-title>. <source>Korean J Med Educ</source>. (<year>2019</year>) <volume>31</volume>:<fpage>1</fpage>&#x2013;<lpage>9</lpage>. doi: <pub-id pub-id-type="doi">10.3946/kjme.2019.113</pub-id>, PMID: <pub-id pub-id-type="pmid">30852856</pub-id></citation>
</ref>
<ref id="ref31">
<label>31.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Renner</surname> <given-names>C</given-names></name> <name><surname>Renner</surname> <given-names>M</given-names></name></person-group>. <article-title>How to create a good exam</article-title> In: <person-group person-group-type="editor"><name><surname>Perlman</surname> <given-names>B</given-names></name> <name><surname>McCann</surname> <given-names>LI</given-names></name> <name><surname>McFadden</surname> <given-names>SH</given-names></name></person-group>, editors. <source>Lessons learned: practical advice for teaching of psychology</source>. <publisher-loc>Washington, DC</publisher-loc>: <publisher-name>American Psychological Society</publisher-name> (<year>1999</year>). <fpage>43</fpage>&#x2013;<lpage>7</lpage>.</citation>
</ref>
<ref id="ref32">
<label>32.</label>
<citation citation-type="book"><person-group person-group-type="author">
<name><surname>McKeachie</surname> <given-names>W</given-names></name>
</person-group>. <source>Teaching tips</source>. <edition>11th</edition> ed. <publisher-loc>Boston, MA</publisher-loc>: <publisher-name>Houghton Mifflin Company</publisher-name> (<year>2002</year>) <comment>ch. 8</comment>.</citation>
</ref>
<ref id="ref33">
<label>33.</label>
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Downing</surname> <given-names>SM</given-names></name>
</person-group>. <article-title>Reliability: on the reproducibility of assessment data</article-title>. <source>Med Educ</source>. (<year>2004</year>) <volume>38</volume>:<fpage>1006</fpage>&#x2013;<lpage>12</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1365-2929.2004.01932.x</pub-id>, PMID: <pub-id pub-id-type="pmid">15327684</pub-id></citation>
</ref>
<ref id="ref34">
<label>34.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>M&#x00F6;ltner</surname> <given-names>A</given-names></name> <name><surname>Schellberg</surname> <given-names>D</given-names></name> <name><surname>J&#x00FC;nger</surname> <given-names>J</given-names></name></person-group>. <article-title>Grundlegende quantitative analysen medizinischer pr&#x00FC;fungen</article-title>. <source>GMS Z Med Ausbild</source>. (<year>2006</year>) <volume>23</volume>:<fpage>11</fpage>.</citation>
</ref>
<ref id="ref35">
<label>35.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tavakol</surname> <given-names>M</given-names></name> <name><surname>Dennick</surname> <given-names>R</given-names></name></person-group>. <article-title>Making sense of Cronbach&#x2019;s alpha</article-title>. <source>Int J Med Educ</source>. (<year>2011</year>) <volume>2</volume>:<fpage>53</fpage>&#x2013;<lpage>5</lpage>. doi: <pub-id pub-id-type="doi">10.5116/ijme.4dfb.8dfd</pub-id>, PMID: <pub-id pub-id-type="pmid">28029643</pub-id></citation>
</ref>
<ref id="ref36">
<label>36.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>J&#x00FC;nger</surname> <given-names>J</given-names></name> <name><surname>Just</surname> <given-names>I</given-names></name></person-group>. <article-title>Recommendations of the German Society for Medical Education and the German Association of Medical Faculties regarding university-specific assessments during the study of human, dental and veterinary medicine</article-title>. <source>GMS Z Med Ausbild</source>. (<year>2014</year>) <volume>31</volume>:<fpage>Doc34</fpage>. doi: <pub-id pub-id-type="doi">10.3205/zma000926</pub-id>, PMID: <pub-id pub-id-type="pmid">25228936</pub-id></citation>
</ref>
<ref id="ref37">
<label>37.</label>
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Kibble</surname> <given-names>JD</given-names></name>
</person-group>. <article-title>Best practices in summative assessment</article-title>. <source>Adv Physiol Educ</source>. (<year>2017</year>) <volume>41</volume>:<fpage>110</fpage>&#x2013;<lpage>9</lpage>. doi: <pub-id pub-id-type="doi">10.1152/advan.00116.2016</pub-id>, PMID: <pub-id pub-id-type="pmid">28188198</pub-id></citation>
</ref>
<ref id="ref38">
<label>38.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cluskey</surname> <given-names>C</given-names> <suffix>Jr</suffix></name> <name><surname>Ehlen</surname> <given-names>C</given-names></name> <name><surname>Raiborn</surname> <given-names>M</given-names></name></person-group>. <article-title>Thwarting online exam cheating without proctor supervision</article-title>. <source>J Acad Bus Ethics</source>. (<year>2011</year>) <volume>4</volume>:<fpage>1</fpage>&#x2013;<lpage>7</lpage>.</citation>
</ref>
<ref id="ref39">
<label>39.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Perlini</surname> <given-names>AH</given-names></name> <name><surname>Lind</surname> <given-names>DL</given-names></name> <name><surname>Zumbo</surname> <given-names>BD</given-names></name></person-group>. <article-title>Context effects on examinations: the effects of time, item order and item difficulty</article-title>. <source>Can Psychol</source>. (<year>1998</year>) <volume>39</volume>:<fpage>299</fpage>&#x2013;<lpage>307</lpage>. doi: <pub-id pub-id-type="doi">10.1037/h0086821</pub-id></citation>
</ref>
<ref id="ref40">
<label>40.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lindner</surname> <given-names>MA</given-names></name> <name><surname>Mayntz</surname> <given-names>SM</given-names></name> <name><surname>Schult</surname> <given-names>J</given-names></name></person-group>. <article-title>Studentische Bewertung und Pr&#x00E4;ferenz von Hochschulpr&#x00FC;fungen mit Aufgaben im offenen und geschlossenen Antwortformat</article-title>. <source>Zeitschrift f&#x00FC;r P&#x00E4;dagogische Psychol</source>. (<year>2018</year>) <volume>32</volume>:<fpage>239</fpage>&#x2013;<lpage>48</lpage>. doi: <pub-id pub-id-type="doi">10.1024/1010-0652/a000229</pub-id></citation>
</ref>
<ref id="ref41">
<label>41.</label>
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Cintron</surname> <given-names>DW</given-names></name>
</person-group>. <article-title>Methods for measuring speededness: chronology, classification, and ensuing research and development</article-title>. <source>ETS Res Rep Ser</source>. (<year>2021</year>) <volume>2021</volume>:<fpage>1</fpage>&#x2013;<lpage>36</lpage>. doi: <pub-id pub-id-type="doi">10.1002/ets2.12337</pub-id></citation>
</ref>
<ref id="ref42">
<label>42.</label>
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Hsieh</surname> <given-names>C</given-names></name>
</person-group>. <article-title>Time needed for undergraduate biomechanics exams</article-title>. <source>ISBS Proc Arch</source>. (<year>2018</year>) <volume>36</volume>:<fpage>847</fpage>&#x2013;<lpage>850</lpage>.</citation>
</ref>
<ref id="ref43">
<label>43.</label>
<citation citation-type="other"><person-group person-group-type="author">
<collab id="coll1">Association of Test Publishers (ATP) and Institute for Credentialing Excellence (ICE)</collab>
</person-group>. (<year>2017</year>). Innovative item types: a white paper and portfolio. Available at: <ext-link xlink:href="https://atpu.memberclicks.net/assets/innovative%20item%20types%20w.%20appendix%20copy.pdf" ext-link-type="uri">https://atpu.memberclicks.net/assets/innovative%20item%20types%20w.%20appendix%20copy.pdf</ext-link></citation>
</ref>
<ref id="ref44">
<label>44.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sireci</surname> <given-names>SG</given-names></name> <name><surname>Botha</surname> <given-names>SM</given-names></name></person-group>. <article-title>Timing considerations in test development and administration</article-title> In: <person-group person-group-type="editor"><name><surname>Margolis</surname> <given-names>MJ</given-names></name> <name><surname>Feinberg</surname> <given-names>RA</given-names></name></person-group>, editors. <source>Integrating timing considerations to improve testing practices</source>. <publisher-loc>Oxfordshire</publisher-loc>: <publisher-name>Routledge</publisher-name> (<year>2020</year>). <fpage>32</fpage>&#x2013;<lpage>46</lpage>.</citation>
</ref>
<ref id="ref45">
<label>45.</label>
<citation citation-type="book"><person-group person-group-type="author">
<name><surname>Swineford</surname> <given-names>F</given-names></name>
</person-group>. <source>The test analysis manual (ETS SR 74-06)</source>. <publisher-loc>Princeton, NJ</publisher-loc>: <publisher-name>Educational Testing Service</publisher-name> (<year>1974</year>).</citation>
</ref>
<ref id="ref46">
<label>46.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lahza</surname> <given-names>H</given-names></name> <name><surname>Smith</surname> <given-names>TG</given-names></name> <name><surname>Khosravi</surname> <given-names>H</given-names></name></person-group>. <article-title>Beyond item analysis: connecting student behaviour and performance using E-assessment logs</article-title>. <source>Br J Educ Technol</source>. (<year>2023</year>) <volume>54</volume>:<fpage>335</fpage>&#x2013;<lpage>54</lpage>. doi: <pub-id pub-id-type="doi">10.1111/bjet.13270</pub-id></citation>
</ref>
<ref id="ref47">
<label>47.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gonz&#x00E1;lez-Espada</surname> <given-names>WJ</given-names></name> <name><surname>Bullock</surname> <given-names>DW</given-names></name></person-group>. <article-title>Innovative applications of classroom response systems: investigating students&#x2019; item response times in relation to final course grade, gender, general point average, and high school act scores</article-title>. <source>Electron J Integr Technol Educ</source>. (<year>2007</year>) <volume>6</volume>:<fpage>97</fpage>&#x2013;<lpage>108</lpage>.</citation>
</ref>
<ref id="ref48">
<label>48.</label>
<citation citation-type="book"><person-group person-group-type="author">
<name><surname>Paterson</surname> <given-names>DG</given-names></name>
</person-group>. <source>Preparation and use of new-type examinations; a manual for teachers</source>. <publisher-loc>Yonkers-on-Hudson, NY</publisher-loc>: <publisher-name>World Book Company</publisher-name> (<year>1924</year>). p. <fpage>42</fpage>&#x2013;<lpage>66</lpage></citation>
</ref>
<ref id="ref49">
<label>49.</label>
<citation citation-type="book"><person-group person-group-type="author">
<name><surname>Cronbach</surname> <given-names>LJ</given-names></name>
</person-group>. <source>Essentials of psychological testing</source>. <edition>4th</edition> ed. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Harper &#x0026; Row</publisher-name> (<year>1984</year>) <comment>ch. 4</comment>.</citation>
</ref>
<ref id="ref50">
<label>50.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Case</surname> <given-names>S</given-names></name> <name><surname>Swanson</surname> <given-names>D</given-names></name></person-group>. <article-title>Constructing written test questions for the basic and clinical sciences</article-title>. <source>Natl Board Exam</source>. (<year>2002</year>):<fpage>13</fpage>&#x2013;<lpage>104</lpage>.</citation>
</ref>
<ref id="ref51">
<label>51.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Haladyna</surname> <given-names>TM</given-names></name> <name><surname>Downing</surname> <given-names>SM</given-names></name> <name><surname>Rodriguez</surname> <given-names>MC</given-names></name></person-group>. <article-title>A review of multiple-choice item-writing guidelines for classroom assessment</article-title>. <source>Appl Meas Educ</source>. (<year>2002</year>) <volume>15</volume>:<fpage>309</fpage>&#x2013;<lpage>33</lpage>. doi: <pub-id pub-id-type="doi">10.1207/S15324818AME1503_5</pub-id></citation>
</ref>
<ref id="ref52">
<label>52.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Haladyna</surname> <given-names>TM</given-names></name> <name><surname>Rodriguez</surname> <given-names>MC</given-names></name></person-group>. <source>Developing and validating test items</source>. <publisher-loc>New York, USA</publisher-loc>: <publisher-name>Routledge</publisher-name> (<year>2013</year>). doi: <pub-id pub-id-type="doi">10.4324/9780203850381</pub-id></citation>
</ref>
<ref id="ref53">
<label>53.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Anderson</surname> <given-names>LW</given-names></name> <name><surname>Krathwohl</surname> <given-names>DR</given-names></name> <name><surname>Airasian</surname> <given-names>PW</given-names></name> <name><surname>Cruikshank</surname> <given-names>KA</given-names></name> <name><surname>Mayer</surname> <given-names>RE</given-names></name> <name><surname>Pintrich</surname> <given-names>PR</given-names></name> <etal/></person-group>. <source>A taxonomy for learning, teaching, and assessing: a revision of Bloom&#x2019;s taxonomy of educational objectives</source>. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Longman</publisher-name> (<year>2001</year>).</citation>
</ref>
<ref id="ref54">
<label>54.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Durning</surname> <given-names>SJ</given-names></name> <name><surname>Dong</surname> <given-names>T</given-names></name> <name><surname>Ratcliffe</surname> <given-names>T</given-names></name> <name><surname>Schuwirth</surname> <given-names>L</given-names></name> <name><surname>Artino</surname> <given-names>AR</given-names> <suffix>Jr</suffix></name> <name><surname>Boulet</surname> <given-names>JR</given-names></name> <etal/></person-group>. <article-title>Comparing open-book and closed-book examinations: a systematic review</article-title>. <source>Acad Med</source>. (<year>2016</year>) <volume>91</volume>:<fpage>583</fpage>&#x2013;<lpage>99</lpage>. doi: <pub-id pub-id-type="doi">10.1097/ACM.0000000000000977</pub-id>, PMID: <pub-id pub-id-type="pmid">26535862</pub-id></citation>
</ref>
<ref id="ref55">
<label>55.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Benjamin</surname> <given-names>LT</given-names></name> <name><surname>Cavell</surname> <given-names>TA</given-names></name> <name><surname>Shallenberger</surname> <given-names>WR</given-names></name></person-group>. <article-title>Staying with initial answers on objective tests: is it a myth?</article-title> <source>Teach Psychol</source>. (<year>1984</year>) <volume>11</volume>:<fpage>133</fpage>&#x2013;<lpage>41</lpage>. doi: <pub-id pub-id-type="doi">10.1177/009862838401100303</pub-id></citation>
</ref>
<ref id="ref56">
<label>56.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fischer</surname> <given-names>MR</given-names></name> <name><surname>Herrmann</surname> <given-names>S</given-names></name> <name><surname>Kopp</surname> <given-names>V</given-names></name></person-group>. <article-title>Answering multiple-choice questions in high-stakes medical examinations</article-title>. <source>Med Educ</source>. (<year>2005</year>) <volume>39</volume>:<fpage>890</fpage>&#x2013;<lpage>4</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1365-2929.2005.02243.x</pub-id>, PMID: <pub-id pub-id-type="pmid">16150028</pub-id></citation>
</ref>
<ref id="ref57">
<label>57.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kruger</surname> <given-names>J</given-names></name> <name><surname>Wirtz</surname> <given-names>D</given-names></name> <name><surname>Miller</surname> <given-names>DT</given-names></name></person-group>. <article-title>Counterfactual thinking and the first instinct fallacy</article-title>. <source>J Pers Soc Psychol</source>. (<year>2005</year>) <volume>88</volume>:<fpage>725</fpage>&#x2013;<lpage>35</lpage>. <comment>Epub 2005/05/19</comment>. doi: <pub-id pub-id-type="doi">10.1037/0022-3514.88.5.725</pub-id>, PMID: <pub-id pub-id-type="pmid">15898871</pub-id></citation>
</ref>
<ref id="ref58">
<label>58.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Couchman</surname> <given-names>JJ</given-names></name> <name><surname>Miller</surname> <given-names>NE</given-names></name> <name><surname>Zmuda</surname> <given-names>SJ</given-names></name> <name><surname>Feather</surname> <given-names>K</given-names></name> <name><surname>Schwartzmeyer</surname> <given-names>T</given-names></name></person-group>. <article-title>The instinct fallacy: the metacognition of answering and revising during college exams</article-title>. <source>Metacogn Learn</source>. (<year>2015</year>) <volume>11</volume>:<fpage>171</fpage>&#x2013;<lpage>85</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11409-015-9140-8</pub-id></citation>
</ref>
<ref id="ref59">
<label>59.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>M&#x00F6;ltner</surname> <given-names>A</given-names></name> <name><surname>Heid</surname> <given-names>J</given-names></name> <name><surname>Wagener</surname> <given-names>S</given-names></name> <name><surname>J&#x00FC;nger</surname> <given-names>J</given-names></name></person-group>. <source>Beantwortungszeiten von Fragen bei einem online durchgef&#x00FC;hrten Progresstest: Abh&#x00E4;ngigkeit von Schwierigkeit, Studienjahr und Korrektheit der Antwort und die First Instinct Fallacy</source>. <publisher-loc>Bern, D&#x00FC;sseldorf</publisher-loc>: <publisher-name>Jahrestagung der Gesellschaft f&#x00FC;r Medizinische Ausbildung (GMA)</publisher-name> (<year>2016</year>).</citation>
</ref>
<ref id="ref60">
<label>60.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>AlMahmoud</surname> <given-names>T</given-names></name> <name><surname>Regmi</surname> <given-names>D</given-names></name> <name><surname>Elzubeir</surname> <given-names>M</given-names></name> <name><surname>Howarth</surname> <given-names>FC</given-names></name> <name><surname>Shaban</surname> <given-names>S</given-names></name></person-group>. <article-title>Medical student question answering behaviour during high-stakes multiple choice examinations</article-title>. <source>Int J Technol Enhanc Learn</source>. (<year>2019</year>) <volume>11</volume>:<fpage>157</fpage>&#x2013;<lpage>71</lpage>. doi: <pub-id pub-id-type="doi">10.1504/IJTEL.2019.098777</pub-id>, PMID: <pub-id pub-id-type="pmid">21607743</pub-id></citation>
</ref>
<ref id="ref61">
<label>61.</label>
<citation citation-type="book"><person-group person-group-type="author">
<name><surname>Ehrich</surname> <given-names>F</given-names></name>
</person-group>. <source>Untersuchungen zu kompetenzorientierten Pr&#x00FC;fungen an der Stiftung Tier&#x00E4;rztliche Hochschule</source>. <comment>[Dissertation]</comment>. <publisher-loc>Hannover, Germany</publisher-loc>: <publisher-name>Tier&#x00E4;rztliche Hochschule Hannover</publisher-name> (<year>2019</year>).</citation>
</ref>
<ref id="ref62">
<label>62.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schaper</surname> <given-names>E</given-names></name> <name><surname>Tipold</surname> <given-names>A</given-names></name> <name><surname>Fischer</surname> <given-names>M</given-names></name> <name><surname>Ehlers</surname> <given-names>JP</given-names></name></person-group>. <article-title>Fallbasiertes, elektronisches Lernen und Pr&#x00FC;fen in der Tiermedizin - Auf der Suche nach einer Alternative zu Multiple-Choice Pr&#x00FC;fungen</article-title>. <source>Tierarztl Umsch</source>. (<year>2011</year>) <volume>66</volume>:<fpage>261</fpage>&#x2013;<lpage>8</lpage>.</citation>
</ref>
<ref id="ref63">
<label>63.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brightwell</surname> <given-names>R</given-names></name> <name><surname>Daniel</surname> <given-names>J-H</given-names></name> <name><surname>Stewart</surname> <given-names>A</given-names></name></person-group>. <article-title>Evaluation: is an open book examination easier?</article-title> <source>Biosci Educ</source>. (<year>2015</year>) <volume>3</volume>:<fpage>1</fpage>&#x2013;<lpage>10</lpage>. doi: <pub-id pub-id-type="doi">10.3108/beej.2004.03000004</pub-id></citation>
</ref>
<ref id="ref64">
<label>64.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Heijne-Penninga</surname> <given-names>M</given-names></name> <name><surname>Kuks</surname> <given-names>JB</given-names></name> <name><surname>Schonrock-Adema</surname> <given-names>J</given-names></name> <name><surname>Snijders</surname> <given-names>TA</given-names></name> <name><surname>Cohen-Schotanus</surname> <given-names>J</given-names></name></person-group>. <article-title>Open-book tests to complement assessment-programmes: analysis of open and closed-book tests</article-title>. <source>Adv Health Sci Educ Theory Pract</source>. (<year>2008</year>) <volume>13</volume>:<fpage>263</fpage>&#x2013;<lpage>73</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10459-006-9038-y</pub-id></citation>
</ref>
<ref id="ref65">
<label>65.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sam</surname> <given-names>AH</given-names></name> <name><surname>Reid</surname> <given-names>MD</given-names></name> <name><surname>Amin</surname> <given-names>A</given-names></name></person-group>. <article-title>High-stakes, remote-access, open-book examinations</article-title>. <source>Med Educ</source>. (<year>2020</year>) <volume>54</volume>:<fpage>767</fpage>&#x2013;<lpage>8</lpage>. doi: <pub-id pub-id-type="doi">10.1111/medu.14247</pub-id>, PMID: <pub-id pub-id-type="pmid">32421858</pub-id></citation>
</ref>
<ref id="ref66">
<label>66.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Choi</surname> <given-names>JH</given-names></name> <name><surname>Hickman</surname> <given-names>KE</given-names></name> <name><surname>Monahan</surname> <given-names>A</given-names></name> <name><surname>Schwarcz</surname> <given-names>DB</given-names></name></person-group>. <article-title>Chatgpt goes to law school</article-title>. <source>J Legal Educ</source>. (<year>2022</year>) <volume>71</volume>:<fpage>387</fpage>&#x2013;<lpage>400</lpage>. doi: <pub-id pub-id-type="doi">10.2139/ssrn.4335905</pub-id></citation>
</ref>
<ref id="ref67">
<label>67.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kung</surname> <given-names>TH</given-names></name> <name><surname>Cheatham</surname> <given-names>M</given-names></name> <name><surname>Medenilla</surname> <given-names>A</given-names></name> <name><surname>Sillos</surname> <given-names>C</given-names></name> <name><surname>De Leon</surname> <given-names>L</given-names></name> <name><surname>Elepano</surname> <given-names>C</given-names></name> <etal/></person-group>. <article-title>Performance of ChatGPT on USMLE: potential for AI-assisted medical education using large language models</article-title>. <source>PLOS Digit Health</source>. (<year>2023</year>) <volume>2</volume>:<fpage>e0000198</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pdig.0000198</pub-id>, PMID: <pub-id pub-id-type="pmid">36812645</pub-id></citation>
</ref>
<ref id="ref68">
<label>68.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jung</surname> <given-names>LB</given-names></name> <name><surname>Gudera</surname> <given-names>JA</given-names></name> <name><surname>Wiegand</surname> <given-names>TLT</given-names></name> <name><surname>Allmendinger</surname> <given-names>S</given-names></name> <name><surname>Dimitriadis</surname> <given-names>K</given-names></name> <name><surname>Koerte</surname> <given-names>IK</given-names></name></person-group>. <article-title>Chatgpt passes German state examination in medicine with picture questions omitted</article-title>. <source>Dtsch Arztebl Int</source>. (<year>2023</year>) <volume>120</volume>:<fpage>373</fpage>&#x2013;<lpage>4</lpage>. doi: <pub-id pub-id-type="doi">10.3238/arztebl.m2023.0113</pub-id>, PMID: <pub-id pub-id-type="pmid">37530052</pub-id></citation>
</ref>
<ref id="ref69">
<label>69.</label>
<citation citation-type="other"><person-group person-group-type="author">
<collab id="coll2">OpenAI</collab>
</person-group>. <source>Gpt-4 technical report</source>. (<year>2023</year>). <publisher-loc>Ithaca, NY, USA</publisher-loc>: <publisher-name>arXiv</publisher-name>.</citation>
</ref>
<ref id="ref70">
<label>70.</label>
<citation citation-type="other"><person-group person-group-type="author">
<name><surname>Terwiesch</surname> <given-names>C.</given-names></name>
</person-group> Would ChatGPT3 get a Wharton MBA? A prediction based on its performance in the operations management course. (<year>2023</year>). <comment>Available at:</comment> <ext-link xlink:href="https://mackinstitute.wharton.upenn.edu/wp-content/uploads/2023/01/Christian-Terwiesch-Chat-GTP.pdf" ext-link-type="uri">https://mackinstitute.wharton.upenn.edu/wp-content/uploads/2023/01/Christian-Terwiesch-Chat-GTP.pdf</ext-link></citation>
</ref>
<ref id="ref71">
<label>71.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Friederichs</surname> <given-names>H</given-names></name> <name><surname>Friederichs</surname> <given-names>WJ</given-names></name> <name><surname>Marz</surname> <given-names>M</given-names></name></person-group>. <article-title>Chatgpt in medical school: how successful is AI in progress testing?</article-title> <source>Med Educ</source>. (<year>2023</year>) <volume>28</volume>:<fpage>2220920</fpage>. doi: <pub-id pub-id-type="doi">10.1080/10872981.2023.2220920</pub-id>, PMID: <pub-id pub-id-type="pmid">37307503</pub-id></citation>
</ref>
<ref id="ref72">
<label>72.</label>
<citation citation-type="other"><person-group person-group-type="author">
<name><surname>Susnjak</surname> <given-names>T.</given-names></name>
</person-group> <source>ChatGPT: the end of online exam integrity?</source> (<year>2022</year>). <publisher-loc>Ithaca, NY, USA</publisher-loc>: <publisher-name>arXiv</publisher-name>.</citation>
</ref>
<ref id="ref73">
<label>73.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cotton</surname> <given-names>DRE</given-names></name> <name><surname>Cotton</surname> <given-names>PA</given-names></name> <name><surname>Shipway</surname> <given-names>JR</given-names></name></person-group>. <article-title>Chatting and cheating: ensuring academic integrity in the era of ChatGPT</article-title>. <source>Innov Educ Teach Int</source>. (<year>2023</year>) <volume>61</volume>:<fpage>228</fpage>&#x2013;<lpage>39</lpage>. doi: <pub-id pub-id-type="doi">10.1080/14703297.2023.2190148</pub-id></citation>
</ref>
<ref id="ref74">
<label>74.</label>
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Oravec</surname> <given-names>JA</given-names></name>
</person-group>. <article-title>Artificial intelligence implications for academic cheating: expanding the dimensions of responsible human-AI collaboration with ChatGPT</article-title>. <source>J Interact Learn Res</source>. (<year>2023</year>) <volume>34</volume>:<fpage>213</fpage>&#x2013;<lpage>37</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>