<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Psychiatry</journal-id>
<journal-title-group>
<journal-title>Frontiers in Psychiatry</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Psychiatry</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">1664-0640</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpsyt.2025.1643221</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>The reproducibility of structured functional assessments in a social security setting: a pre-specified explanatory analysis of the RELY-studies</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Kunz</surname><given-names>Regina</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>*</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/884052/overview"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Giezendanner</surname><given-names>Stephanie</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>von Allmen</surname><given-names>David Y.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/924464/overview"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project-administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Jeger</surname><given-names>Joerg</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3237025/overview"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Eichhorn</surname><given-names>Martin</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Hoffmann-Richter</surname><given-names>Ulrike</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="author-notes" rid="fn003"><sup>&#x2020;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Fischer</surname><given-names>Katrin</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
</contrib>
<contrib contrib-type="author">
<name><surname>de Boer</surname><given-names>Wout</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>Evidence-based Insurance Medicine (EbIM), Research &amp; Education, Division of Clinical Epidemiology, Department of Clinical Research, University of Basel</institution>, <city>Basel</city>,&#xa0;<country country="ch">Switzerland</country></aff>
<aff id="aff2"><label>2</label><institution>MEDAS Central Switzerland</institution>, <city>Luzern</city>,&#xa0;<country country="ch">Switzerland</country></aff>
<aff id="aff3"><label>3</label><institution>Mental Health Practice</institution>, <city>Basel</city>,&#xa0;<country country="ch">Switzerland</country></aff>
<aff id="aff4"><label>4</label><institution>Swiss National Accident Insurance Funds</institution>, <city>Lucerne</city>,&#xa0;<country country="ch">Switzerland</country></aff>
<aff id="aff5"><label>5</label><institution>Institute Humans in Complex Systems, School of Applied Psychology, University of Applied Sciences Northwestern Switzerland</institution>, <city>Olten</city>,&#xa0;<country country="ch">Switzerland</country></aff>
<author-notes>
<corresp id="c001"><label>*</label>Correspondence: Regina Kunz, <email xlink:href="mailto:regina.kunz@usb.ch">regina.kunz@usb.ch</email></corresp>
<fn fn-type="deceased" id="fn003">
<label>&#x2020;</label>
<p>Deceased</p></fn>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2026-01-02">
<day>02</day>
<month>01</month>
<year>2026</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1643221</elocation-id>
<history>
<date date-type="received">
<day>08</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>17</day>
<month>11</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>13</day>
<month>11</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2026 Kunz, Giezendanner, von Allmen, Jeger, Eichhorn, Hoffmann-Richter, Fischer and de Boer.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>Kunz, Giezendanner, von Allmen, Jeger, Eichhorn, Hoffmann-Richter, Fischer and de Boer</copyright-holder>
<license>
<ali:license_ref start_date="2026-01-02">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Limitations in work capacity (WC) need to be quantified in a transparent and reproducible way when insurers of social security decide whether an individual is entitled to disability benefits and to what extent. Structured assessments of work-related physical, mental and social functioning might provide an empirical basis for judgments on residual work capacity (rWC) which determines entitlement to disability benefits. This study examined the functional assessments themselves, their reliability and expert agreement when applied to claimants with mental disorders, and analyzed their relationship to rWC judgments.</p>
</sec>
<sec>
<title>Material and methods</title>
<p>We used RELY-data on the reproducibility of rWC judgments. A pool of 40 psychiatric experts interviewed 55 claimants for disability benefits. Interviews were videotaped and watched by three observing psychiatric experts, resulting in 280 individual ratings. All independently rated claimants&#x2019; impairments in work-related mental functions and capacity limitations using the Instrument for Functional Assessment in Psychiatry (IFAP-1 mental functions, IFAP-2a/-2b functional capacities related to the last job and alternative work, scaled 0=none to 4=worst) based on the Mini-ICF-APP, and judged rWC (in Switzerland, scaled 100% to 0%) for the last job and suitable alternative work. Analysis for reliability (ICC, intraclass-correlation coefficient) included a two-way random-effects and a linear mixed-effects model. Expert agreement was estimated as standard error of measurement, SEM.</p>
</sec>
<sec>
<title>Results</title>
<p>The mean score for mental functions (IFAP-1<sub>global</sub>) was 1.21 (SD 0.63) and for functional capacities in alternative work (IFAP-2b<sub>global</sub>) 0.87 (SD 0.56). Reliability of IFAP ratings was low to fair (IFAP-1<sub>global</sub>: ICC = 0.46; IFAP-2b<sub>global</sub>: ICC = 0.26), similar to the low interrater reliability of rWC. Agreement showed substantial measurement error: IFAP-1<sub>global</sub>: SEM = 0.47; IFAP-2b<sub>global</sub>: SEM = 0.49. The rWC judgments for claimants with identical ratings in functional limitations (IFAP-2b<sub>global</sub>=1) ranged from 100% to 5%.</p>
</sec>
<sec>
<title>Conclusions</title>
<p>Evidence indicates that Functional Assessment, if carried out well, may lead to more reproducibility. This explanatory analysis revealed low to fair interrater reproducibility in mental functions (IFAP-1), in functional capacities (IFAP-2a/b) which extends to rWC. Among various other explanations, we believe this to be mostly due to insufficient training in Functional Assessment and therefore reflects real-world variability in judgment. We recommend revising training format and intensity, and monitoring adherence in practice, followed by re-evaluation of reproducibility of expert judgements. As of today, the outcome is uncertain.</p>
</sec>
</abstract>
<kwd-group>
<kwd>disability insurance</kwd>
<kwd>work capacity evaluation</kwd>
<kwd>reproducibility of results</kwd>
<kwd>evidence-based medicine</kwd>
<kwd>mental disorders</kwd>
<kwd>international classification of functioning</kwd>
<kwd>disability and health</kwd>
<kwd>independent medical evaluation</kwd>
<kwd>Mini-ICF-APP</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declared that financial support was received for this work and/or its publication. FUNDING The secondary analyses reported in this publication were conducted without external funding. The main study was supported by grants from the Swiss National Science Foundation (project number 325130_144200), from the Federal Social Insurance Office, and from the Swiss National Accident Insurance. None of these organisations were involved in the design, data collection, analysis or interpretation of the data. (Kunz et&#xa0;al. BMC Psychiatry 2019).</funding-statement>
</funding-group>
<counts>
<fig-count count="2"/>
<table-count count="3"/>
<equation-count count="1"/>
<ref-count count="43"/>
<page-count count="13"/>
<word-count count="7977"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Social Psychiatry and Psychiatric Rehabilitation</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Social security systems provide disability benefits for employees whose work capacity (WC) is noticeably and permanently impaired due to illness or accident. It is common practice for employees claiming disability benefits to undergo an expert evaluation to determine their ability to work (<xref ref-type="bibr" rid="B1">1</xref>). What makes expert evaluations a sensitive and controversial issue is the established evidence that different experts reach different conclusions when assessing residual work capacity (rWC) (<xref ref-type="bibr" rid="B2">2</xref>). In many countries, including Switzerland, the expert judgment on rWC largely determines the decision on entitlement to disability benefits and their amount.</p>
<sec id="s1_1">
<label>1.1</label>
<title>What motivated our research program</title>
<p>In an illustrative study, 23 psychiatric experts assessed the WC of a hypothetical claimant suffering from recurrent moderate depression (<xref ref-type="bibr" rid="B3">3</xref>). Based on identical information (psychiatric history, medical reports, diagnoses from treating physicians, a staged video interview), eight psychiatrists found no impairment, ten concluded partial impairment, and four determined no rWC. The researchers judged such expert variation to be &#x201c;unacceptable for members of the German state pension system&#x201d;.</p>
</sec>
<sec id="s1_2">
<label>1.2</label>
<title>Methodological considerations: high expert variation signals poor reproducibility, crucial in WC assessments</title>
<p>Reproducibility is defined as the degree to which repeated measurements yield similar results (<xref ref-type="bibr" rid="B4">4</xref>) and encompasses two distinct concepts, reliability and agreement (<xref ref-type="bibr" rid="B4">4</xref>&#x2013;<xref ref-type="bibr" rid="B6">6</xref>). Reliability refers to how well patients and claimants can be distinguished from each other, despite measurement errors, when assessed by two or more raters (interrater reliability), while agreement refers to the absolute differences between expert ratings (interrater agreement) (<xref ref-type="bibr" rid="B4">4</xref>). Activities used to perform an evaluation require a high degree of agreement between raters, for example, when evaluating experts as they conduct WC assessments (<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B7">7</xref>). This requirement applies to the evaluation of work (in-)capacity. Of note, high interrater reliability does not imply high interrater agreement (<xref ref-type="bibr" rid="B4">4</xref>).</p>
</sec>
<sec id="s1_3">
<label>1.3</label>
<title>Our solution: develop a structured approach - functional Assessment of WC - to improve expert agreement</title>
<p>A systematic review of 23 studies from 12 countries revealed low to fair reproducibility of experts&#x2019; judgments on WC (<xref ref-type="bibr" rid="B2">2</xref>). However, disability evaluations that employed a structured approach in both, procedures and outcome measurements, showed higher reproducibility (<xref ref-type="bibr" rid="B8">8</xref>). These plausible findings caught our attention. We developed the functional interview, a semi-structured conversation about the claimants&#x2019; work, self-perceived work (in-)capacity, and remaining ability to perform work-related tasks and used the social functioning scale Mini-ICF-APP for experts to report their findings. This scale has repeatedly been proposed for assessing work disability in social benefit claims (<xref ref-type="bibr" rid="B9">9</xref>&#x2013;<xref ref-type="bibr" rid="B12">12</xref>), although it has not yet been tested in national medicolegal settings. We integrated a scale for mental functions and the Swiss scale for judging rWC related to the last job and alternative work and referred to as &#x201c;Instruments for Functional Assessment in Psychiatry&#x201d; (IFAP-1,-2,-3). Following their two-hourly interviews, psychiatrists were instructed to use the IFAPs to document their judgment about the claimants&#x2019; functional capacities and limitations in work-related activities (<xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B14">14</xref>). This structured approach was named &#x201c;Functional Assessment&#x201d; (<xref ref-type="bibr" rid="B15">15</xref>).</p>
</sec>
<sec id="s1_4">
<label>1.4</label>
<title>The RELY-studies showed low expert agreement despite functional assessment and training</title>
<p>Our multicenter reproducibility studies in a real-world setting (RELY-1 and -2) investigated whether the Functional Assessment would improve agreement and reliability for the main outcome, rWC (<xref ref-type="bibr" rid="B13">13</xref>). Claimants were assessed by four randomly allocated psychiatrists, all trained in Functional Assessment: one interviewer and three observers who watched the video-taped interview. All experts filled out the IFAP independently from each other. Since RELY-1 showed low agreement and reliability among experts, the study was repeated with more intensive expert training and a much shorter time span between training and application in RELY-2. We have not formally evaluated the effectiveness of the training. In a subsequent comparison, expert agreement for claimants&#x2019; rWC in RELY-2 improved by about 20% (reported as standard error of measurement, SEM). Since important decisions for claimants &#x2013; the entitlement to disability benefits and their amount &#x2013; were based on these assessments, agreement remained unacceptably low, and measurement error among experts unacceptably high. Interrater reliability of rWC judgments was fair (RELY-1: ICC 0.43; RELY-2: ICC 0.44) and did not change between studies (<xref ref-type="bibr" rid="B13">13</xref>).</p>
</sec>
<sec id="s1_5">
<label>1.5</label>
<title>More work-related conversation with the claimant showed better expert agreement</title>
<p>The content analysis of the RELY-studies investigated the coverage of work-related topics in these interviews to identify factors contributing to the poor reproducibility (<xref ref-type="bibr" rid="B16">16</xref>). Prominent finding: Experts asked very little about claimants&#x2019; self-perceived activity limitations and work (in-)capacity, which indicated insufficient compliance with the training. Interviews with higher coverage achieved significantly higher expert agreement on WC ratings than those with low coverage, suggesting that interviews conducted with sufficient focus on work may improve reproducibility. Hence, the functional interview on work (<xref ref-type="bibr" rid="B15">15</xref>) is a compulsory requirement for completing the IFAP.</p>
</sec>
<sec id="s1_6">
<label>1.6</label>
<title>Similarities and differences between IFAP scales and mini-ICF-APP scales</title>
<p>The analyses reported in this paper explore the role of the IFAP rating scales used by RELY-experts to quantify mental impairments and capacity limitations observed in medicolegal assessments. IFAP and the social functioning scale Mini-ICF-APP relate to the WHO International Classification of Functioning, ICF and its 5-item-rating scale with generic descriptions (<xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B17">17</xref>). The Mini-ICF-APP was developed and validated in occupational rehabilitation (<xref ref-type="bibr" rid="B17">17</xref>) and community mental health (<xref ref-type="bibr" rid="B18">18</xref>, <xref ref-type="bibr" rid="B19">19</xref>) to evaluate individuals with mental disorders on their functional (in-)capacities in domains of social functioning. The English translation (<xref ref-type="bibr" rid="B19">19</xref>) reported high internal consistency (Cronbach&#x2019;s &#x3b1; 0.869 &#x2013; 0.912) and good test-retest reliability (ICC 0.832) when applied by two raters who were not described any further (<xref ref-type="bibr" rid="B5">5</xref>). The original German study to the Mini-ICF-APP (<xref ref-type="bibr" rid="B17">17</xref>, <xref ref-type="bibr" rid="B20">20</xref>) did not publish any psychometric properties. In contrast, the IFAPs are part of the Functional Assessment designed specifically for psychiatric experts who evaluate work (in-)capacity in employees with mental disorders on behalf of social insurers. IFAP takes into account the Swiss medicolegal context, which requires capacity assessment related to the claimants&#x2019; last job and to alternative work adjusted to the claimants&#x2019; limitations and requires a final judgment on the claimants&#x2019; rWC on a scale from 100% to 0%. This judgment should reflect the path from impairments in mental functions (IFAP-1) to limitations in functional capacity (IFAP-2a/-b) to rWC (IFAP-3). Since instruments (here: the Functional Assessment) require validation in the context in which they are being used, IFAP requires validation in the medicolegal context. The &#x2018;reliability studies&#x2019; of the Mini-ICF-APP (<xref ref-type="bibr" rid="B17">17</xref>&#x2013;<xref ref-type="bibr" rid="B19">19</xref>) include data from controlled research settings of social or rehabilitation context without the purpose to capture real-world heterogeneity. They were reported as classical Spearman rank correlation coefficient (<xref ref-type="bibr" rid="B17">17</xref>) which, however, is not a reliability measure (<xref ref-type="bibr" rid="B21">21</xref>), or as intraclass correlation coefficients ICCs (<xref ref-type="bibr" rid="B18">18</xref>, <xref ref-type="bibr" rid="B19">19</xref>). Classical correlations (e.g., Pearson or Spearman) measure the relationship between two different variables, while the ICC, a true reliability measure, describes the ability to distinguish between subjects within a group (same variable = intra-&#x2019;class&#x2019;) (<xref ref-type="bibr" rid="B22">22</xref>, <xref ref-type="bibr" rid="B23">23</xref>). It is known that highly correlated observations may have poor agreement (<xref ref-type="bibr" rid="B24">24</xref>). Of note, none of the Mini-ICF-APP studies investigated (interrater) agreement (<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B5">5</xref>). The specific procedures (if any), in which the Mini-ICF-APP was used, were not reported in these studies which compromises repeatability. It is legitimate to question whether work disability evaluations under so many &#x201c;real-world&#x201d; conditions in a medicolegal context could achieve similarly high interrater reliability if investigated using rigorous methodology (<xref ref-type="bibr" rid="B5">5</xref>, <xref ref-type="bibr" rid="B20">20</xref>). This needs to be tested.</p>
</sec>
<sec id="s1_7">
<label>1.7</label>
<title>Psychiatrists are not work experts</title>
<p>The low agreement in judging rWC observed in RELY possibly reflects psychiatrists&#x2019; limited understanding of job demands (<xref ref-type="bibr" rid="B2">2</xref>), as psychiatrists are experts in mental health, not work. Hence, mental health professionals might show better agreement on functions which are closer to their mental health expertise, such as the ones reviewed in IFAP-1.</p>
</sec>
<sec id="s1_8">
<label>1.8</label>
<title>Research objectives</title>
<p>In the RELY-studies, the main outcome was agreement between experts and their reliability when using the medicolegal construct &#x2018;rWC in adjusted work&#x2019; (<xref ref-type="bibr" rid="B13">13</xref>), derived from the functional assessment (i.e., the IFAP-1 and -2a/b-instruments). Agreement and reliability turned out to be low.</p>
<p>In the current analysis, we examined the functional assessment itself (IFAP 1 and IFAP 2a/b). We hypothesized that the functional assessment should yield better agreement and reliability, given the explicit elaboration of the IFAP 1 and 2a/b items in the manual (<xref ref-type="bibr" rid="B15">15</xref>) and their ratings against the specific requirements of adjusted work. Furthermore, we examined the value of the global instruments IFAP-1<sub>global</sub> and IFAP-2a/b<sub>global</sub> in predicting rWC in alternative work.</p>
</sec>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Methods and material</title>
<p>Reporting of our reliability and agreement studies followed the Guidelines for Reporting Reliability and Agreement Studies, GRRAS-Guidelines (<xref ref-type="bibr" rid="B5">5</xref>). The protocol paper describes the key features of the design (<xref ref-type="bibr" rid="B14">14</xref>), the main publication reports the findings on the main endpoint rWC (<xref ref-type="bibr" rid="B13">13</xref>).</p>
<sec id="s2_1">
<label>2.1</label>
<title>The RELY-studies</title>
<p>We conducted two multicenter reproducibility studies, RELY-1 and -2, with a partial crossover design. Four expert psychiatrists (one interviewer, three video raters) independently rated the rWC of real patients who had claimed disability benefits, four ratings per patient. Protocol (<xref ref-type="bibr" rid="B14">14</xref>) and main publication (<xref ref-type="bibr" rid="B13">13</xref>) report the details. We recruited 30 claimants for RELY-1 (resulting in 30*4 = 120 ratings) and increased this to 40 claimants for RELY-2. Of these, we recruited 25 new applicants (resulting in 25*4 = 100 ratings) and re-used 15 videos from RELY-1 (resulting in 15*4 = 60 ratings). In total, there were 280 ratings. In the current analysis, we merged the data from RELY-1 and -2 given their high correlations (r=0.88) and did not differentiate between the two different training regimes.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Procedures</title>
<p>The interviews of the claimants conducted in real-world settings were videotaped. Claimants were assessed by four randomly assigned psychiatrists trained in Functional Assessment: one interviewer and three psychiatrists who rated independently the video-taped interview. Pseudo-randomization was used to assign claimants to the interviewer, true randomization to assign them to the observing psychiatrists.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Functional assessments and judgments of work capacity</title>
<p>The three-part IFAP was developed for the RELY-studies (<xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B14">14</xref>) to quantify and document impairments in mental functions and limitations in functional domains for assessing rWC, whereby structure and content of IFAP-2 is identical to that of Mini-ICF-APP. IFAP-1 rates twelve mental functions: temperament and personality functions, agreeableness, mental stability, openness to experience, confidence, energy and drive functions, attention functions, memory functions, emotional functions, thought functions, higher-level cognitive functions, experience of self and time functions. IFAP-2 rates functional limitations in thirteen domains: adherence to regulations, planning and structuring of tasks, flexibility, applying expertise, competence to judge and decide, endurance, assertiveness, contact with others, group integration, intimate relationships, non-work activities, self-care, and mobility (<xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B25">25</xref>). Both, IFAP-1 and -2, employ the 5-item scale of the ICF rating system: &#x201c;0&#x201d;=none, &#x201c;1&#x201d;=mild, &#x201c;2&#x201d;=moderate, &#x201c;3&#x201d;=severe impairment/limitation and &#x201c;4&#x201d;=complete disability (<xref ref-type="bibr" rid="B26">26</xref>). A rating of &#x201c;mild&#x201d; means that the restrictions do not affect WC, while a rating of &#x201c;moderate&#x201d; implies that WC is affected (<xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B26">26</xref>). IFAP-2a is rated against the profile of the claimants&#x2019; last job, IFAP-2b against a profile for alternative work adjusted to the claimants&#x2019; limitations. Experts were expected to use the ratings of IFAP-2a/b to judge the claimants&#x2019; rWC in the last job (rWC<sub>last</sub>) and in alternative work (rWC<sub>alt</sub>) on a scale from 100% to 0% (IFAP-3a/b). When the experts in RELY-1 had judged a claimant&#x2019;s rWC<sub>last</sub> as 100%, they omitted to fill in the IFAP-2b form, which related to alternative work and was initially considered as redundant. This procedure reduced the number of IFAP-2b cases for calculating the ICCs. In RELY-2, this detail was modified and IFAP-2b forms had to be completed regardless of the claimant&#x2019;s rWC<sub>last</sub>. To facilitate quantitative analyses of the IFAP instruments, we calculated IFAP-1<sub>global</sub>, -2<sub>global,</sub> -3<sub>global</sub> scores (&#x201c;global score&#x201d;) by taking the mean sum score (= mean across the sum) of the 12 respectively 13 items of the IFAP-1 and -2 scales (<xref ref-type="bibr" rid="B17">17</xref>), and the value of the assigned rWC as IFAP-3<sub>global</sub> score. In this way, we report the summary of the IFAP scales in the same way as the individual domain scales, which facilitates comparison.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Expert certainty in judgments</title>
<p>Expert certainty in their judgment was frequently raised as a potentially strong variable impacting on the variation of rWC. Furthermore, it was postulated that interviewing versus observing experts might vary substantially in their certainty of judgment. To this end, experts were asked to express their certainty in their own rWC judgment (Certainty_rWC<sub>last</sub> and Certainty_rWC<sub>alt</sub>) on an 11-point Likert scale from 0 (very uncertain) to 10 (very certain). Findings will be reported in <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary 8</bold></xref> only.</p>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Statistics</title>
<p>For question 1, we estimated the interrater reliability of the IFAP ratings using the intraclass correlation coefficient ICC, and the interrater agreement using standard error of measurement SEM and percentage of agreement. We calculated the ICC for the individual IFAP domains (<xref ref-type="bibr" rid="B23">23</xref>) and for IFAP<sub>global</sub> (IFAP-1<sub>global</sub>, -2<sub>global</sub>, -3<sub>global</sub>) taking into account the two-way incomplete crossover design. We fitted a linear mixed-effects model with the IFAP<sub>global</sub> scores as dependent variable, the intercept as the only fixed effect and psychiatrists and claimants as random effects (using the function &#x201c;lmer&#x201d; of the R package &#x201c;lme4&#x201d;) (<xref ref-type="bibr" rid="B27">27</xref>, <xref ref-type="bibr" rid="B28">28</xref>). Subsequently, we extracted the variance components of the psychiatrists and claimants from the fitted model to calculate the ICC (A, 1). As the judgment from a single rater will be the basis of the measurement, we therefore considered &#x2018;single rater&#x2019; type of agreement even though the reliability experiment involved four raters. In summary, the ICC was estimated based on a two-way random-effects model with a single-rating (k = 1), and absolute agreement, using the formula (<xref ref-type="bibr" rid="B21">21</xref>):</p>
<disp-formula>
<mml:math display="block" id="M1"><mml:mrow><mml:mi>I</mml:mi><mml:mi>C</mml:mi><mml:mi>C</mml:mi><mml:mtext>&#xa0;</mml:mtext><mml:mrow><mml:mo mathvariant="italic">(</mml:mo><mml:mrow><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#xa0;</mml:mtext><mml:mn>1</mml:mn></mml:mrow><mml:mo mathvariant="italic">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mtext>&#xa0;</mml:mtext><mml:mfrac><mml:mrow><mml:msubsup><mml:mi>&#x3c3;</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msubsup><mml:mtext>&#xa0;</mml:mtext></mml:mrow><mml:mrow><mml:msubsup><mml:mi>&#x3c3;</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>&#x3c3;</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mi>s</mml:mi><mml:mi>y</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>&#x3c3;</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mo>*</mml:mo><mml:mi>p</mml:mi><mml:mi>s</mml:mi><mml:mi>y</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msubsup></mml:mrow></mml:mfrac></mml:mrow></mml:math>
</disp-formula>
<p>where cl=claimants, psy=psychiatrists, cl*psy=residuals [claimants*psychiatrist interaction effect and the random measurement error (<xref ref-type="bibr" rid="B29">29</xref>)]</p>
<p>The 95% confidence intervals (CI) were obtained by applying a model-based parametric bootstrap for the mixed-effects models with R = 9999 repetitions and reporting the 2.5th and the 97.5th percentiles using the function &#x201c;bootMer&#x201d; of the package &#x201c;lmer&#x201d; and the function &#x201c;boot.ci&#x201d; from the package &#x201c;boot&#x201d;. Thus, the calculated ICCs reflect reliability of absolute agreement between raters. We interpreted the ICC as poor (ICC&lt; 0.40), fair (0.40&#x2013;0.59), good (0.60&#x2013;0.74) and excellent (&gt; 0.75) (<xref ref-type="bibr" rid="B30">30</xref>).</p>
<p>Measurement error for continuous variables is represented by SEM and equals the square root of the error variance. SEM equals the square root of the error variance and is a suitable measure of agreement for continuous variables reported in natural units. It is calculated as <inline-formula>
<mml:math display="inline" id="im1"><mml:mrow><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>g</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>e</mml:mi><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:msubsup><mml:mi>&#x3c3;</mml:mi><mml:mrow><mml:mi>P</mml:mi><mml:mi>s</mml:mi><mml:mi>y</mml:mi><mml:mi>c</mml:mi><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>&#x3c3;</mml:mi><mml:mrow><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>d</mml:mi><mml:mi>u</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msubsup><mml:mo>&#xa0;</mml:mo><mml:mo>&#xa0;</mml:mo></mml:mrow></mml:msqrt></mml:mrow></mml:math></inline-formula> (<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B13">13</xref>). For rWC, SEM is reported as % rWC, while for the unitless IFAP-scale, SEM is a unitless number. Lower SEM values indicate higher agreement because the measurement error is low. Expected value of &#x2018;standard error of measurement&#x2019; is defined as SEM<sub>expected</sub> = <inline-formula>
<mml:math display="inline" id="im2"><mml:mrow><mml:mfrac><mml:mrow><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:mn>1.96</mml:mn><mml:mo>*</mml:mo><mml:mo>&#xa0;</mml:mo><mml:mo>&#x221a;</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:mfrac></mml:mrow></mml:math></inline-formula> (<xref ref-type="bibr" rid="B4">4</xref>). If the maximum acceptable difference (MAD) of rWC judgements between raters (psychiatrists and experts) was 25%, the observed &#x2018;standard error of measurement&#x2019; had to be smaller than 9.0 percentage points WC [Table&#xa0;4 in (<xref ref-type="bibr" rid="B13">13</xref>)]. Similarly, if we think the maximum acceptable difference of IFAP ratings between raters on a 5-point Likert scale is 1 point, the observed &#x2018;standard error of measurement&#x2019; had to be smaller than 0.36 points. For the ordinal data of individual IFAP items (scale from 0 to 4), we focused on clinically relevant disagreement, i.e., impacting rWC. Thus, we created three levels out of the five levels of the IFAP scale: ratings of &#x201c;no and mild limitations&#x201d; (mild limitations had been defined as limitations with no impact on WC (<xref ref-type="bibr" rid="B15">15</xref>)), &#x201c;moderate limitations&#x201d;, and &#x201c;severe limitations and total disability&#x201d;. For the 13 single IFAP items regarding rWC<sub>last/alt</sub>, we reported ICC and percentage of agreement based on these 3 levels (<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B5">5</xref>).</p>
<p>For question 2, we analyzed the association between IFAP ratings and judgments on rWC<sub>alt</sub> (<xref ref-type="bibr" rid="B31">31</xref>). First, a scatter plot visualized the relationship between IFAP-2b<sub>global</sub> versus rWC<sub>alt</sub>. To describe the relationship between the dimensions mental functions, functional capacity and rWC, we divided the range of rWC into 10%-intervals from 100% to 0% and calculated mean and standard deviation of the IFAP-1 and -2a/b values for these intervals separately for rWC estimates in the last job and in alternative work. Each claimant&#x2013;expert pairing was treated as independent observation (i.e., IFAP ratings were not averaged across the four rating experts). To enable a comparison with previous publications (<xref ref-type="bibr" rid="B31">31</xref>, <xref ref-type="bibr" rid="B32">32</xref>), we additional grouped the rWC as high (100% to 70%), medium (69% to 31%) and low (30% to 0%). Second, we performed three univariable linear mixed-effects model (LMM) analyses to examine the effect of IFAP-1<sub>global</sub>, -2a<sub>global</sub> and -2b<sub>global</sub> on rWC<sub>alt</sub>, accounting for the potential variability between random effect of claimants and psychiatrist raters. LMMs are particularly suitable for data with hierarchical or nested structures, allowing us to model both fixed effects and random effects. We estimated the model parameters using R lmer (linear mixed-effect model regression) function of the lme4 package, with restricted maximum likelihood (REML) estimation to ensure unbiased estimates of the variance components. Furthermore, we calculated univariable LMM analyses for the 38 individual domains of the three IFAP instruments, which are reported in the supplement.</p>
<p>To test whether the level of expert certainty in his own rWC judgment contributed to the variability (&#x2018;low agreement&#x2019;) of the four experts&#x2019; rWC judgments on the same claimant, a linear regression was performed using the absolute deviation of the rWC<sub>last/alt</sub> of raters from the mean rWC<sub>last/alt</sub> per claimant as dependent variable and the raters&#x2019; certainty in their own rWC<sub>last/alt</sub> as independent variable. Further, we tested for group differences in certainty in rWC<sub>alt</sub>/rWC<sub>last</sub> estimation across observer and interviewing raters using independent Welch two sample t-test. Findings will be reported in <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary 8</bold></xref>.</p>
<p>Calculations were carried out in Software R 4.3.2 (2023&#x2013;10&#x2013;31) (<xref ref-type="bibr" rid="B33">33</xref>). To illustrate expert variation in the interpretation of an IFAP-2b-score of 1 (=mild) and 2 (=moderate) - we fitted a regression line using ordinary least squares, and computed 95% confidence bands using the standard error of the predicted mean response (conducted in R using the ggplot2 and stats packages).</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<sec id="s3_1">
<label>3.1</label>
<title>Participants</title>
<p>Claimants for disability benefits undergoing a psychiatric work disability evaluation at one of four medical assessment centers participated in the study. The mean age of the claimants was 47.8 (SD 9.3) years. All claimants were German-speaking. For claimant characteristics (marital status, nationality, main ICD-10 F-codes) and expert characteristics (age, gender, years of experience, number of work disability evaluations in the year before study participation) see main publication (<xref ref-type="bibr" rid="B13">13</xref>).</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Reliability and agreement of expert psychiatrists rating mental functions (IFAP-1) and functional capacities (IFAP-2a/b)</title>
<sec id="s3_2_1">
<label>3.2.1</label>
<title>Mental functions, IFAP-1, 280 ratings</title>
<p>(<xref ref-type="fig" rid="f1"><bold>Figure&#xa0;1</bold></xref>, <xref ref-type="table" rid="T1"><bold>Table&#xa0;1A</bold></xref>, <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table S1A, B</bold></xref>). The mean rating of IFAP-1<sub>global</sub> was 1.21 (SD 0.63). The most impaired mental functions were <italic>Mental Stability</italic> (mean 1.71, SD 0.89), <italic>Self-Confidence (</italic>mean 1.69, SD 0.95), and <italic>Energy</italic> (mean 1.59, SD 0.85). Less than 10% of the individual IFAP-1 ratings indicated severe impairment or complete disability (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table S2</bold></xref>). <xref ref-type="fig" rid="f1"><bold>Figure&#xa0;1</bold></xref> illustrates the variation in expert ratings on the same claimant. The reliability of IFAP-1<sub>global</sub> was 0.46 (ICC, 95% CI 0.32; 0.58) with mainly poor ICC values for the 12 domains ranging from 0.25 to 0.42. Agreement on IFAP-1<sub>global</sub> among experts measured as SEM was 0.47 (95% CI 0.41; 0.52). Agreement on individual domains, measured in percentage of agreement, was ranging from 54.4% (<italic>Thought Functions</italic>) to 16.4% (<italic>Self Confidence</italic>).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Heterogeneity of the claimant population with regards to impairment in mental functions and variability of the experts rating the impairment of the same claimant. The graph shows to what extent the 55 claimants were impaired in their mental functions (IFAP &#x2013; 1<sub>global</sub>). Each claimant was rated by 4 experts. 15 claimants from the RELY-1 study were re-rated by different raters in the RELY-2 study (e.g., claimant 5 or 13 or 29 or 39 or 52), resulting in 8 ratings per claimant. The red dots show the mean across the ratings of the 4 (<xref ref-type="bibr" rid="B8">8</xref>) experts and are aligned in ascending order. The vertical spread of black dots (= four (eight) individual ratings) illustrates the differences between when rating the same claimant based on the same information. Differences between raters on the same claimant can be substantial. Abbreviations: Instruments of Functional Assessment in Psychiatry, IFAP, with IFAP-1 = Mental Functions; IFAP-2: Functional Capacities; IFAP-3: Swiss scale for rWC. IFAP-2b<sub>global</sub>: mean sum score related to alternative work on a scale from 0 (= no impairment) to 4 (= complete disability).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1643221-g001.tif">
<alt-text content-type="machine-generated">Scatter plot with 55 applicants on the x-axis and their mental functions (IFAP-1 values) on the y-axis. The mental functions of each applicant were assessed by 4 (or 8) experts and are depicted as black dots on a vertical line for each claimant. A red dot marks the mean value of the mental functions as judged by the 4 (8) experts.</alt-text>
</graphic></fig>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Relationship between the claimants&#x2019; functional capacity (IFAP-2b<sub>global</sub>) and their residual work capacity. Each dot represents an expert&#xb4;s global IFAP-2b rating (scale from 0 to 4) and residual WC<sub>alt</sub> judgment (alternative work, scale from 0% to 100%; 260 observations). The vertical scatter of dots along the blue line illustrates the variability in WC judgments for the same degree of functional impairment (&#x2018;mild&#x2019; corresponds to IFAP score of 1.0): Experts judged the claimants&#x2019; rWC with mild functional impairment to be between 5% and 100%. For a mean functional impairment of 2.0 (orange line), experts judged the rWC to be between 0% and 60%. The regression line (in black) was fitted using simple univariable regression and is accompanied by its 95% confidence band (in grey). Abbreviations: Instruments of Functional Assessment in Psychiatry, IFAP, with IFAP-1 = Mental Functions; IFAP-2: Functional Capacities; IFAP-3: Swiss scale for rWC. IFAP-2b<sub>global</sub>: mean sum score related to alternative work on a scale from 0 (= no impairment) to 4 (= complete disability).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1643221-g002.tif">
<alt-text content-type="machine-generated">Scatter plot showing a negative correlation between functional impairment (IFAP-2b global on the x-axis, categorized from none to complete) and work capacity (y-axis). A trend line with a shaded confidence band indicates decreasing work capacity as functional impairment increases.</alt-text>
</graphic></fig>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Reliability (ICC) and agreement (SEM) of the three IFAP instruments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">The IFAP Instruments</th>
<th valign="middle" align="center">Reliability ICC [95% CI]</th>
<th valign="middle" align="center">Agreement SEM [95% CI]</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="middle" colspan="3" align="left">a) Mental functions</th>
</tr>
<tr>
<td valign="middle" align="left">IFAP-1<sub>global</sub><break/>(5-item scale)</td>
<td valign="middle" align="center">0.46<break/>[0.32; 0.58]</td>
<td valign="middle" align="center">0.47<break/>[0.41; 0.52]</td>
</tr>
<tr>
<th valign="middle" colspan="3" align="left">b) Functional capacities related to the last job</th>
</tr>
<tr>
<td valign="middle" align="left">IFAP-2a<sub>global</sub><break/>(5-item scale)</td>
<td valign="middle" align="center">0.41<break/>[0.28; 0.53]</td>
<td valign="middle" align="center">0.49<break/>[0.43; 0.54]</td>
</tr>
<tr>
<td valign="middle" align="left">Residual Work Capacity<sub>last</sub><break/>(scale: 0% - 100% rWC)</td>
<td valign="middle" align="center">0.44<break/>[0.3; 0.55]</td>
<td valign="middle" align="center">24.6% rWC<break/>[21.9; 27.5]</td>
</tr>
<tr>
<th valign="middle" colspan="3" align="left">c) Functional capacities related to alternative work</th>
</tr>
<tr>
<td valign="middle" align="left">IFAP-2b<sub>global</sub><break/>(5-item scale)</td>
<td valign="middle" align="center">0.26<break/>[0.15; 0.38]</td>
<td valign="middle" align="center">0.49<break/>[0.45; 0.52]</td>
</tr>
<tr>
<td valign="middle" align="left">Residual Work Capacity<sub>alt</sub><break/>(scale: 0% - 100% rWC)</td>
<td valign="middle" align="center">0.45<break/>[0.31; 0.57]</td>
<td valign="middle" align="center">21.49% rWC<break/>[19.1; 24.1]</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The table reveals the experts' inability to discriminate between claimants based on their functional capacity, IFAP-1<sub>global</sub>, -2a/b<sub>global</sub> (low ICC), and the low agreement among experts (SEM) in these judgments. Similarly weak findings were observed for the claimants' rWC (IFAP-3a/b) as judged by experts using the functional capacity.</p></fn>
<fn>
<p>Abbreviations: IFAP, Instrument of Functional Assessment in Psychiatry; IFAP-1, mental functions; IFAP-2a/2b, functional capacities related to last job / alternative work; IFAP-3a/3b, residual work capacity related to last job / alternative work; ICC, Intra-class correlation coefficient; SEM, standard error of measurement.  </p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_2_2">
<label>3.2.2</label>
<title>Functional capacity related to the last job, IFAP-2a, 280 ratings</title>
<p>(<xref ref-type="table" rid="T1"><bold>Table&#xa0;1B</bold></xref> and <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table S3A, B</bold></xref>). The mean rating IFAP-2a<sub>global</sub> was 1.11 (SD 0.64). The domains <italic>Endurance</italic> (mean 2.07, SD 0.81), <italic>Flexibility</italic> (mean 1.56, SD 0.97) and <italic>Assertiveness</italic> (mean 1.44, SD 1.00) revealed the most severe limitations. For all functional domains but <italic>Endurance</italic>, less than 20% of the IFAP-2a ratings indicated severe limitations or complete disability (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table S4</bold></xref>). Reliability of functional capacity ratings for the last job (IFAP-2a<sub>global</sub>) was 0.41 (ICC, 95% CI 0.28; 0.53), with mainly poor ICC values for the 13 domains ranging from 0.20 to 0.43. Agreement among experts on IFAP-2a<sub>global</sub> was 0.49 (SEM, 95% CI 0.43; 0.54). Percentage of agreement on individual domains was ranging from 82.4% (<italic>Selfcare</italic>) to 15.2% (<italic>Endurance</italic>).</p>
</sec>
<sec id="s3_2_3">
<label>3.2.3</label>
<title>Functional capacity related to alternative work, IFAP-2b, 260 ratings</title>
<p>(<xref ref-type="table" rid="T1"><bold>Table&#xa0;1C</bold></xref> and <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Tables S5A, B</bold></xref>). The mean rating IFAP-2b<sub>global</sub> was 0.87 (SD 0.56). Again, the domains <italic>Endurance</italic> (mean 1.68, SD 0.82), <italic>Flexibility</italic> (1.15, SD 0.87), and <italic>Assertiveness</italic> (1.11, SD 0.91) revealed the most severe limitations, although they were rated as less severe compared to IFAP-2a, where the reference was the last job. At all functional domains but <italic>Endurance</italic>, less than 10% of the IFAP-2b ratings indicated severe limitations or complete disability (<xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary Table S6</bold></xref>). Reliability of ratings on functional capacity for alternative work, IFAP-2b<sub>global</sub>, was poor (ICC 0.26, 95% CI 0.15; 0.38) as were all ratings on individual IFAP-2b domains. Agreement among experts on IFAP-2b<sub>global</sub> was 0.49 (SEM, 95% CI 0.45; 0.52). Percentage of agreement on individual domains was ranging from 89.3% (<italic>Selfcare</italic>) to 26.4% (<italic>Endurance</italic>).</p>
<p>In summary, the experts&#x2019; interrater reliability for mental functions and functional capacities, both for last job and alternative work (IFAP-1<sub>global</sub> and -2<sub>global</sub>) was poor to fair. Likewise, the corresponding agreement was poor, too.</p>
</sec>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Relationship between mental functions, functional capacities, and residual work capacity</title>
<p>Question 2 addressed the association between IFAP-ratings and judgments on rWC in alternative work, using IFAP-2b<sub>global</sub> and judgments on rWC<sub>alt</sub> (<xref ref-type="bibr" rid="B31">31</xref>) as an example (<xref ref-type="table" rid="T2"><bold>Table&#xa0;2</bold></xref>). The linear mixed-effect regression results showed that the fixed effect of IFAP-2b<sub>global</sub> explained 38% of the variance in the expert judgment of rWC<sub>alt</sub>.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Values of the global instruments IFAP-1, IFAP-2a and IFAP-2b in predicting rWC in alternative work.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">FAP instruments</th>
<th valign="middle" align="center">Estimate (in % rWC)</th>
<th valign="middle" align="center">Lower 95% CI</th>
<th valign="middle" align="center">Upper 95% CI</th>
<th valign="middle" align="center">Marginal R<sup>2</sup></th>
<th valign="middle" align="center">n</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="middle" colspan="6" align="left">IFAP-1<sub>global</sub></th>
</tr>
<tr>
<td valign="middle" align="left">(Intercept)</td>
<td valign="middle" align="center">95.87</td>
<td valign="middle" align="center">88.80</td>
<td valign="middle" align="center">102.92</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
</tr>
<tr>
<td valign="middle" align="left">IFAP-1<sub>global</sub></td>
<td valign="middle" align="center">-30.38</td>
<td valign="middle" align="center">-35.04</td>
<td valign="middle" align="center">-25.73</td>
<td valign="middle" align="center">0.45</td>
<td valign="middle" align="center">278</td>
</tr>
<tr>
<th valign="middle" colspan="6" align="left">IFAP-2a<sub>global</sub></th>
</tr>
<tr>
<td valign="middle" align="left">(Intercept)</td>
<td valign="middle" align="center">89.09</td>
<td valign="middle" align="center">82.06</td>
<td valign="middle" align="center">96.10</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
</tr>
<tr>
<td valign="middle" align="left">IFAP-2a<sub>global</sub></td>
<td valign="middle" align="center">-27.16</td>
<td valign="middle" align="center">-31.99</td>
<td valign="middle" align="center">-22.38</td>
<td valign="middle" align="center">0.37</td>
<td valign="middle" align="center">260</td>
</tr>
<tr>
<th valign="middle" colspan="6" align="left">IFAP-2b<sub>global</sub></th>
</tr>
<tr>
<td valign="middle" align="left">(Intercept)</td>
<td valign="middle" align="center">86.09</td>
<td valign="middle" align="center">80.06</td>
<td valign="middle" align="center">92.14</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
</tr>
<tr>
<td valign="middle" align="left">IFAP-2b<sub>global</sub></td>
<td valign="middle" align="center">-27.32</td>
<td valign="middle" align="center">-32.39</td>
<td valign="middle" align="center">-22.44</td>
<td valign="middle" align="center">0.38</td>
<td valign="middle" align="center">259</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Results of the three univariable linear mixed effect regression model in which IFAP-1<sub>global</sub> and IFAP-2b<sub>global</sub> were entered separately into the models of rWC<sub>alt</sub>. Marginal R<sup>2</sup> explains the variance due to fixed effects.</p></fn>
</table-wrap-foot>
</table-wrap>
<p><xref ref-type="fig" rid="f2"><bold>Figure&#xa0;2</bold></xref> visualizes the dispersion of rWC judgments among claimants with similar degree of functional limitations: claimants with mild global functional limitations (i.e., an IFAP-2b<sub>global</sub> value of 1, blue line) were attributed rWCs to be between 5% and 100% in the observed data, between 20% and 97% in the 95% prediction interval (PI) and between 56% and 61% in the 95% confidence interval (CI). Individuals with moderate global functional impairments (i.e., an IFAP-2b value of 2, orange line) were assigned rWCs between 0% and 60% in observed data, between -11% and 67% in the 95% PI and between 23% and 34% in the 95% CI.</p>
<p>The explanatory pathway from &#x2018;mental functions&#x2019; to &#x2018;functional abilities&#x2019; to &#x2018;rWC&#x2019; provides a different perspective. <xref ref-type="table" rid="T3"><bold>Table&#xa0;3</bold></xref> shows the distribution of the IFAP-ratings in 10%-steps rWC for the last job (<xref ref-type="table" rid="T3"><bold>Table&#xa0;3A</bold></xref>) and for alternative work (<xref ref-type="table" rid="T3"><bold>Table&#xa0;3B</bold></xref>). As intended by law, claimants with low and moderate rWC for the last job were attested higher levels of rWC when referred to alternative work adjusted for their functional limitations. The IFAP instruments, however, did not discriminate well between different levels of rWC. This was particularly relevant in the category &#x201c;moderate rWC&#x201d; of <xref ref-type="table" rid="T3"><bold>Table&#xa0;3B</bold></xref>. In this category, the law is designed in such a way that even a slight change in rWC impacts the disability benefits.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Mental functions and functional capacity analyzed by 10% rWC levels.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">rWC</th>
<th valign="middle" align="center">100%</th>
<th valign="middle" align="center">90%</th>
<th valign="middle" align="center">80%</th>
<th valign="middle" align="center">70%</th>
<th valign="middle" align="center">60%</th>
<th valign="middle" align="center">50%</th>
<th valign="middle" align="center">40%</th>
<th valign="middle" align="center">30%</th>
<th valign="middle" align="center">20%</th>
<th valign="middle" align="center">10%</th>
<th valign="middle" align="center">0%</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">a) Last job</td>
<td valign="middle" colspan="4" align="center">High rWC<sub>last</sub><break/>(n=83)</td>
<td valign="middle" colspan="3" align="center">Moderate rWC<sub>last</sub><break/>(n=85)</td>
<td valign="middle" colspan="4" align="center">Low rWC<sub>last</sub><break/>(n=112)</td>
</tr>
<tr>
<td valign="middle" align="left">IFAP-1<sub>global</sub><break/>(SD) n=280</td>
<td valign="middle" align="center">0.31<break/>(0.36)</td>
<td valign="middle" align="center">0.70<break/>(0.69)</td>
<td valign="middle" align="center">0.74<break/>(0.45)</td>
<td valign="middle" align="center">0.82<break/>(0.35)</td>
<td valign="middle" align="center">1.04<break/>(0.44)</td>
<td valign="middle" align="center">1.34<break/>(0.42)</td>
<td valign="middle" align="center">1.61<break/>(0.40)</td>
<td valign="middle" align="center">1.51<break/>(0.51)</td>
<td valign="middle" align="center">1.46<break/>(0.57)</td>
<td valign="middle" align="center">1.84<break/>(0.42)</td>
<td valign="middle" align="center">1.69<break/>(0.43)</td>
</tr>
<tr>
<td valign="middle" align="left">IFAP-2a<sub>global</sub><break/>(SD) n=280</td>
<td valign="middle" align="center">0.24<break/>(0.20)</td>
<td valign="middle" align="center">0.64<break/>(0.44)</td>
<td valign="middle" align="center">0.63<break/>(0.40)</td>
<td valign="middle" align="center">0.67<break/>(0.37)</td>
<td valign="middle" align="center">0.94<break/>(0.39)</td>
<td valign="middle" align="center">1.16<break/>(0.40)</td>
<td valign="middle" align="center">1.28<break/>(0.40)</td>
<td valign="middle" align="center">1.45<break/>(0.54)</td>
<td valign="middle" align="center">1.29<break/>(0.60)</td>
<td valign="middle" align="center">1.76<break/>(0.44)</td>
<td valign="middle" align="center">1.68<break/>(0.50)</td>
</tr>
<tr>
<td valign="middle" align="left">N</td>
<td valign="middle" align="center">30</td>
<td valign="middle" align="center">5</td>
<td valign="middle" align="center">26</td>
<td valign="middle" align="center">22</td>
<td valign="middle" align="center">32</td>
<td valign="middle" align="center">40</td>
<td valign="middle" align="center">13</td>
<td valign="middle" align="center">24</td>
<td valign="middle" align="center">22</td>
<td valign="middle" align="center">12</td>
<td valign="middle" align="center">54</td>
</tr>
<tr>
<td valign="middle" align="left">b) Alternative work</td>
<td valign="middle" colspan="4" align="center">High rWC<sub>alt</sub><break/>(n=128)</td>
<td valign="middle" colspan="3" align="center">Moderate rWC<sub>alt</sub><break/>(n=99)</td>
<td valign="middle" colspan="4" align="center">Low rWC<sub>alt</sub><break/>(n=52)</td>
</tr>
<tr>
<td valign="middle" align="left">IFAP-1<sub>global</sub><break/>(SD) n=279</td>
<td valign="middle" align="center">0.45<break/>(0.42)</td>
<td valign="middle" align="center">0.70<break/>(0.36)</td>
<td valign="middle" align="center">0.93<break/>(0.43)</td>
<td valign="middle" align="center">1.17<break/>(0.54)</td>
<td valign="middle" align="center">1.29<break/>(0.57)</td>
<td valign="middle" align="center">1.47<break/>(0.42)</td>
<td valign="middle" align="center">1.57<break/>(0.38)</td>
<td valign="middle" align="center">1.78<break/>(0.54)</td>
<td valign="middle" align="center">1.81<break/>(0.61)</td>
<td valign="middle" align="center">1.63<break/>(0.38)</td>
<td valign="middle" align="center">1.82<break/>(0.36)</td>
</tr>
<tr>
<td valign="middle" align="left">IFAP-2b<sub>global</sub><break/>(SD) n=260</td>
<td valign="middle" align="center">0.28<break/>(0.30)</td>
<td valign="middle" align="center">0.47<break/>(0.22)</td>
<td valign="middle" align="center">0.60<break/>(0.33)</td>
<td valign="middle" align="center">0.78<break/>(0.36)</td>
<td valign="middle" align="center">1.02<break/>(0.51)</td>
<td valign="middle" align="center">1.14<break/>(0.38)</td>
<td valign="middle" align="center">1.19<break/>(0.48)</td>
<td valign="middle" align="center">1.41<break/>(0.66)</td>
<td valign="middle" align="center">1.53<break/>(0.72)</td>
<td valign="middle" align="center">1.43<break/>(0.74)</td>
<td valign="middle" align="center">1.56<break/>(0.42)</td>
</tr>
<tr>
<td valign="middle" align="left">N</td>
<td valign="middle" align="center">44/42*</td>
<td valign="middle" align="center">14</td>
<td valign="middle" align="center">42</td>
<td valign="middle" align="center">28</td>
<td valign="middle" align="center">21</td>
<td valign="middle" align="center">54</td>
<td valign="middle" align="center">24</td>
<td valign="middle" align="center">15</td>
<td valign="middle" align="center">10</td>
<td valign="middle" align="center">7</td>
<td valign="middle" align="center">20/3*</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The table shows the mean global score (+ standard deviation) for IFAP-1 and IFAP-2 for each 10% rWC level, separately for rWC judgment in the last job (IFAP-2a top rows) and in alternative work (IFAP-2b bottom rows).</p></fn>
<fn>
<p>Abbreviations: rWC <sub>last/alt:</sub> residual work capacity in the last job/in alternative work; IFAP = Instrument of Functional Assessment in Psychiatry. IFAP-1<sub>global</sub> = global score of impairments in mental functions;IFAP-2a/-2b: global score of functional limitations related to the last job/- to adjusted alternative work; SD: Standard Deviation.</p></fn>
<fn>
<p>* N sometimes varied due to missing IFAP-2b-ratings. The first value refers to IFAP-1 ratings, the second one to IFAP-2b ratings.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>The decline in rWC (from 60% to 50% to 40%) can only marginally be explained by the small decline with large overlapping standard variation in functional abilities of IFAP-2b<sub>alt</sub> [from 1.02 (mean, SD 0.51) to 1.14 (mean, SD 0.38) to 1.19 (mean, SD 0.48)]. These claimants were quite similar in their level of functioning, and the values do not allow a valid discrimination between adjacent levels of rWC.</p>
<p>In summary, using univariable analysis, IFAP-2b<sub>global</sub> explains only to a moderate extent the variance in expert judgments of rWC<sub>alt</sub>, the 95% prediction intervals for single patients are quite large, and these ratings provide only limited guidance to expert judgment on rWC<sub>alt</sub> in individual patients.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<sec id="s4_1">
<label>4.1</label>
<title>Principal findings</title>
<p>This secondary analysis about mental functions (IFAP-1) and functional capacities (IFAP-2) in claimants for work disability was pre-specified to explain the findings of the RELY-studies (<xref ref-type="bibr" rid="B14">14</xref>). The assessments with IFAP-1 and IFAP-2 showed low reliability to discriminate between claimants&#x2019; functional capacity, i.e. between those with high, moderate, fair, or low capacity. The application of both instruments showed low expert agreement which means that a large &#x2018;measurement error&#x2019; in the experts&#x2019; assessment led to low agreement between experts when evaluating functions and capacities. These findings resemble the poor to fair reproducibility of expert judgments about rWC in the RELY-studies. Poor reproducibility has already been noted in the experts&#x2019; inconsistent assessments of mental functions. It continued in the ratings of functional capacities, and it manifested itself in a wide range of judgments regarding rWC.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Strengths and weaknesses</title>
<p>Strengths: GRRAS guidance states that the interpretation of the results of reliability and agreement studies requires sufficient information on study design and conduct and a good description of the measurement setting and the method of calculation (<xref ref-type="bibr" rid="B5">5</xref>). Our design has not been set up to get optimal levels of agreement, rather to reflect real-world performance with all its heterogeneity, and inform insurers, medical and legal professionals and the public. A pre-specified question guided the explanatory analysis of the secondary outcomes (IFAP-1 and -2); the rigorous design of the RELY-studies (<xref ref-type="bibr" rid="B13">13</xref>) ensuring trustworthiness in the findings and applicability to the Swiss setting: real WC assessments commissioned by insurers; recruitment of &#x2018;typical&#x2019; claimants with a representative spectrum of mental disorders; a large mixed group of psychiatrists; randomly assigned groups of four experts to prevent a rater-group effect, and more (<xref ref-type="bibr" rid="B34">34</xref>). Finally, we provide a comprehensive supplement on our data to facilitate comparisons with other studies.</p>
<p>Weaknesses and Limitations: The manual-based training in functional interviewing and defining work demands proved insufficient as was the monitoring of expert compliance to the rating rules prior and during the study. Therefore, we cannot say whether only the training for using the instrument was insufficient or whether the instrument did not work as expected. RELY-1 suffered a serious setback when changes in the governmental administration led to a one-year disruption resulting in a change of the research design (<xref ref-type="bibr" rid="B13">13</xref>).</p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Methodological considerations related to the low reproducibility</title>
<sec id="s4_3_1">
<label>4.3.1</label>
<title>Rating the IFAP - User training and compliance</title>
<p>The innovative component of RELY is the Functional Assessment as described in the introduction (<xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B14">14</xref>, <xref ref-type="bibr" rid="B16">16</xref>) and in <xref ref-type="supplementary-material" rid="SM1"><bold>Supplementary S7</bold></xref>. Manual-based training should ensure that experts stick to the semi-structured interview and all apply the same criteria in their rating judgments. Psychiatrists were told to rate the claimants&#x2019; limitations to a reference, &#x201c;the claimant&#x2019;s last job&#x201d; (IFAP-2a) or to &#x201c;suitable alternative work adjusted to the limitations&#x201d; (IFAP-2b). Informative job descriptions for suitable alternative work (&#x2018;hotel jobs&#x2019;) were provided as part of our study. While claimants have varying capacity profiles with regards to the limitations, the reference was always the &#x201c;job requirements&#x201d; and their match with the &#x201c;claimants&#x2019; (in-) capacities&#x201d;. If a claimant had a severe agoraphobia, but was only working in home office, this severe limitation had no impact on his job.</p>
<p>The content analysis of RELY-1 documents poor compliance with the two most important steps of the functional interview &#x2013; the enquiry of self-perceived work limitations and of work-related health complaints [median number of enquiries: 0 to 1.5 coding units (= smallest meaningful unit of a text)] (<xref ref-type="bibr" rid="B16">16</xref>): The relationship between the claimants&#x2019; functional capacity in adjusted work (IFAP-2b<sub>global</sub>) and the rWC assigned by the expert in <xref ref-type="fig" rid="f2"><bold>Figure&#xa0;2</bold></xref> highlights an example how experts did not follow guidance in translating capacity limitations into the final judgment of WC: The rule for &#x201c;mild impairment&#x201d; stipulated that the claimants&#x2019; limitations do not affect WC, while functional limitations that affect WC need to be rated as &#x201c;moderate limitations&#x201d;. Nevertheless, experts assigned claimants with mild functional limitations a rWC of between 5% and 100%. If they had adhered to the RELY-framework, most dots to the left of the blue line would have to lie at 100% WC. Dots below 100% indicate user errors. When experts assign work incapacities if the mean global rating of IFAP-2b is 1 or below, they failed to understand the IFAP rating.</p>
<p>Others argued that - for instance - a rating of 3 (severe impairment) in two domains and all other domains being 0, would have resulted in a mean sum score of (6/13 == 0.46). Nevertheless, such a person might experience severe limitations to work. This theoretical constellation, however, was very rarely seen in practice, if at all. If claimants had substantial impairments (e.g., score of 3) in one or two domains, almost all mild to moderate impairments in others which shifted their total score above a mean sum score of 1. [Personal communication with co-author J. Jeger who collected mini-ICF-ratings from more than 1000 claimants over a 10-year period (<xref ref-type="bibr" rid="B31">31</xref>)].</p>
<p><xref ref-type="fig" rid="f2"><bold>Figure&#xa0;2</bold></xref> shows, above all, that we did not succeed in enforcing our own guidance in the study. We recognise insufficient user training as one of the biggest problems with both, RELY-1 and -2.</p>
</sec>
<sec id="s4_3_2">
<label>4.3.2</label>
<title>Reliability and agreement in IFAP-ratings</title>
<p>High interrater reliability indicates that two or more experts can well distinguish between claimants with high, moderate, low and very low functional capacity. Low reliability means that experts are unable to discriminate claimants. Apart from insufficient training (misclassification, inconsistent application of rules, disagreement in judgments), poor interrater reliability can occur when instruments, like the 5-point IFAP scale, have only few levels to describe claimants&#x2019; functioning when their level of functional limitations varies considerably. Some suggest that scales with 7 to 10 points are best for achieving valid reliability (<xref ref-type="bibr" rid="B35">35</xref>, <xref ref-type="bibr" rid="B36">36</xref>). This, however, presupposes that the specific scale values can be precisely defined and operationalized, and that psychiatrists can differentiate the functional limitations accordingly. The authors of this paper expressed strong doubts that these preconditions can be met.</p>
</sec>
<sec id="s4_3_3">
<label>4.3.3</label>
<title>Agreement</title>
<p>Agreement informs about the &#x201c;measurement error&#x201d; of an instrument. It becomes low when the measurement error exceeds patient variance. &#x201c;Our instrument&#x201d; for the Functional Assessment was an expert with a range of competencies: trained in collecting relevant information from claimants, experienced in using appropriate instruments (e.g., tests validated in similar settings to the one in which they were used), with a good understanding of work requirements, and the ability to transform the information collected into reasoned judgment about functional capacities for work. These were high expectations. The co-authors of this paper concluded that inadequate training and a lack of supervising the use of the instrument were the most probable reasons for the disagreement (<xref ref-type="bibr" rid="B5">5</xref>).</p>
</sec>
<sec id="s4_3_4">
<label>4.3.4</label>
<title>Claimants and settings</title>
<p>Since a person&#x2019;s level of functioning is an interaction between her or his health conditions and environmental factors (ICF) (<xref ref-type="bibr" rid="B26">26</xref>), claimants need to be evaluated against an explicit reference setting (e.g., last vocational setting, general working life, general labor market), which will affect the level of disability in that specific setting (<xref ref-type="bibr" rid="B26">26</xref>). This requirement is also justified from a methodological perspective, as reliability and agreement coefficients are population- and context-specific (<xref ref-type="bibr" rid="B5">5</xref>). To this end, our study referenced the evaluation to the claimants&#x2019; last job and an alternative work with explicit description of the main functional demands. Such information is crucial to allow comparison of findings across studies, but it is often not or not adequately reported [e.g. &#x2018;uniform standard environment&#x2019; (<xref ref-type="bibr" rid="B18">18</xref>)]. Taken alone, numerical values of ratings or reproducibility measures without context have limited meaning and hinder cross-study comparisons.</p>
</sec>
<sec id="s4_3_5">
<label>4.3.5</label>
<title>The impact of real-world raters</title>
<p>The RELY-studies were designed to mimic the diversity of real-world assessments (<xref ref-type="bibr" rid="B2">2</xref>): Experts vary in their professional approach (e.g., behavioral therapy, systemic therapy, psychoanalysis), in their setting (hospital, community centers, individual practice, rehabilitation, forensic), in their experience in performing medical evaluations. The setting determines the kind of patients they see, experience determines judgments (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B21">21</xref>, <xref ref-type="bibr" rid="B37">37</xref>). These features strengthen our conclusion that the observed low interrater reliability and agreement on functioning and WC reflects the real world of the Swiss setting.</p>
</sec>
<sec id="s4_3_6">
<label>4.3.6</label>
<title>Talking about work works</title>
<p>To align professional heterogeneity, we had trained the experts in collecting work-related information from claimants using a semi-standardized five-step interview about claimants&#x2019; perceptions of their work and functional limitations (<xref ref-type="bibr" rid="B38">38</xref>). While our content analysis of the RELY-1 interviews revealed that compliance with the training had been low (<xref ref-type="bibr" rid="B16">16</xref>), groups with interviewers who did comply, achieved significantly higher agreement in their rWC judgments. This confirms the need for more training, for checking the learning success and for monitoring its use in practice.</p>
</sec>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Comparison to other studies</title>
<p>Overall, we noticed a lack of research for comparison. We identified three studies in patients with mental disorders (<xref ref-type="bibr" rid="B17">17</xref>&#x2013;<xref ref-type="bibr" rid="B19">19</xref>) on the reproducibility of the global score of the Mini-ICF-APP (the instrument underlying IFAP-2). None of them was carried out as part of a medicolegal assessment with the aim of determining the applicants&#x2019; ability to work and to serve as a basis for a decision on a disability pension. One study investigated inpatients in a psychosomatic rehabilitation clinic (<xref ref-type="bibr" rid="B17">17</xref>), two other studies took place in community mental health centers in Italy (<xref ref-type="bibr" rid="B18">18</xref>) and the UK (<xref ref-type="bibr" rid="B19">19</xref>). Reliability and agreement are not fixed properties of instruments, rather, they are the product of interactions between purpose, subjects or objects, instruments, setting, conduct and analysis (<xref ref-type="bibr" rid="B5">5</xref>). Since these studies differ in important ways from our medicolegal study, a direct comparison is not informative.</p>
</sec>
<sec id="s4_5">
<label>4.5</label>
<title>Comparing the RELY-data with findings from routine care</title>
<p>Important insights can be gained from a comparison with a recent single-center study (ScS) about Mini-ICF-APP assessments on more than 900 consecutive claimants (<xref ref-type="bibr" rid="B31">31</xref>). Similarities between the two studies include the medicolegal context, claimants randomly commissioned from the same national insurer, and rWC<sub>alt</sub> as outcome. Studies differed in that RELY-claimants participated voluntarily while ScS-claimants underwent a routine assessment with Mini-ICF-APP ratings. RELY included 40 distinct psychiatrists, while three freelance ScS-psychiatrists assessed 84% of claimants over 10 years. The ScS did not investigate reproducibility. Finally, the ScS seems to allow very different procedures of assessment in which the Mini-ICF is applied.</p>
<p>The ScS found higher functional limitations [ScS: Mini-ICF<sub>global</sub> 1.39 (mean, SD 0.60) vs. RELY: IFAP-2b<sub>global</sub> 0.87 (mean; SD 0.56)] and lower rWC<sub>alt</sub> [ScS: 50.6% (mean, 95% CI 48.7; 52.5) vs. RELY: 59.2% (mean, 95% CI 55.7; 62.6)]. Apart from differences in design, the recruitment of ScS-claimants included those with more functional limitations, while the voluntary participation in RELY may have attracted claimants with less severe limitations. Alternatively, ScS-experts may have been more lenient and attributed higher levels of limitations and consequently lower rWC than the more representative mix of RELY-experts. Both explanations suggest possible bias highlighting the need for integrating methodological procedures to protect against biased selection of claimants and experts in future studies.</p>
</sec>
<sec id="s4_6">
<label>4.6</label>
<title>Implications for the practice of work disability assessments and further research</title>
<p>WC evaluations require an in-depth exchange about work between claimant and expert. Our content analysis of RELY-1 showed that this in-depth exchange did not take place (<xref ref-type="bibr" rid="B16">16</xref>). However, many psychiatrists do not see themselves as experts on work and work demands. In a representative survey, experts from various disciplines expressed their need for tools when assessing WC. Their expectations: high predictiveness, high interrater agreement, and comprehensive for laypeople (<xref ref-type="bibr" rid="B39">39</xref>). Functional interviewing (<xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B16">16</xref>) has a strong face validity and could serve as a framework. Regrettably, the RELY-studies did not deliver &#x201c;proof of concept&#x201d;. The main reasons identified were internal factors (e.g., training insufficient to change behavior) and experts&#x2019; non-compliance with the procedure. Both are modifiable. Increased training in RELY-2 led to better coverage of the topics of the functional interview (manuscript finalized). This justifies a second effort to validate the concept: Adjust training, ensure that raters know the rating rules and apply them accurately, monitor learning progress and compliance in practice. This validation approach should determine the impact of functional interviewing on expert agreement and reliability about work (in-)capacity (<xref ref-type="bibr" rid="B16">16</xref>). As of today, the outcome is uncertain. If positive, follow-up studies could investigate the impact of innovative schemes like online training programs or calibration sessions on reducing variability and improving the practical use of IFAP in social security contexts.</p>
<p>The framework could be complemented by additional psychometric tools on functional diagnostics and prognostics: Digitally Assisted Standard Diagnostics in Insurance Medicine (DASDIM) for mental disorders (<xref ref-type="bibr" rid="B40">40</xref>) or the Work Disability&#x2013;Functional Assessment Battery (WD-FAB) for physical and behavioral functions (<xref ref-type="bibr" rid="B41">41</xref>), which is currently translated into German (<xref ref-type="bibr" rid="B42">42</xref>) and French (<xref ref-type="bibr" rid="B43">43</xref>). Such instruments, validated in the medicolegal context can facilitate consistency checks about the claimants&#x2019; self-perceived capacity limitations and thereby contribute to evidence-based decisions.</p>
<p>Some may argue, why bother with tools that do not live up to expectations? First, the most plausible factors as to why the IFAP rating did not work as expected can be modified. Second, the Mini-ICF-APP (underlying the IFAP) is currently in place in multiple settings, such as expert training (SIM, <uri xlink:href="https://www.swiss-insurance-medicine.ch">www.swiss-insurance-medicine.ch</uri>), psychiatric assessments and as guidance for psychiatric assessments (<xref ref-type="bibr" rid="B11">11</xref>). Third, the Functional Assessment and IFAP-instruments are the only fully evaluated instruments developed in the national setting. If further research shows that they do not work as required, they should no longer be used, and their flawed results should not be employed to determine disability benefits. The alternative, starting from scratch in search of a better tool, is time-consuming and resource-intensive with an uncertain ending.</p>
<p>Not acting is not an option. Work disability assessments for decision-making on granting benefits are subject to the societal legal principle &#x201c;Equality before the law&#x201d;: People with similar level of limitations in similar work settings should be treated equally. Fulfilling this principle expects experts to reliably distinguish between people with high, moderate, fair and low ability to work (&#x201c;reliability&#x201d;), and to achieve a higher level of agreement with other experts in their decisions. This principle is currently under scrutiny. If no progress can be made in the current allocation of disability benefits based on functional impairments of WC, policymakers may need to consider changes to the framework to ensure equal treatment. This may include a change in the law.</p>
</sec>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusions</title>
<p>Integrating the findings of the IFAP-analyses with the findings of other RELY-analyses, we conclude that Functional Assessment if carried out well, can lead to more reproducibility (<xref ref-type="bibr" rid="B16">16</xref>). This explanatory analysis of the RELY-data revealed low to fair interrater reproducibility for mental functions (IFAP-1) for functional capacities (IFAP-2a/b) and finally for rWC. Among various other explanations, we think this to be mostly due to insufficient training in Functional Assessment. Conducting work disability assessments as currently taught and practiced is not likely to improve the poor reliability and the poor agreement, regardless of the instrument used. Rather than starting from scratch in search of a better tool, we recommend revising training format, delivery and intensity, and monitor adherence in routine practice, followed by re-evaluation of reproducibility of expert judgments. As of today, the outcome is uncertain.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The data analyzed in this study is subject to the following licenses/restrictions: The raw data supporting the conclusions of this article will be made available quickly and easily by the authors, on condition that the researchers are at a reputable academic institution and that they accept the conditions of use. Requests to access these datasets should be directed to Regina Kunz <email xlink:href="mailto:regina.kunz@usb.ch">regina.kunz@usb.ch</email>.</p></sec>
<sec id="s7" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The studies involving humans were approved by cantonal ethics committees of Basel, Bern, Lucern, Zurich; the data protection officers of Basel-Stadt, Swiss National Science Foundation, Federal Social Insurance Office, Swiss National Accident Insurance Suva, Disability Insurance Office Zurich. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p></sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>RK: Resources, Validation, Writing &#x2013; review &amp; editing, Methodology, Conceptualization, Writing &#x2013; original draft, Investigation. SG: Methodology, Formal Analysis, Validation, Writing &#x2013; original draft, Software, Writing &#x2013; review &amp; editing. DA: Investigation, Methodology, Writing &#x2013; review &amp; editing, Project administration. JJ: Validation, Supervision, Writing &#x2013; review &amp; editing, Investigation. ME: Supervision, Writing &#x2013; review &amp; editing, Investigation. UH-R: Validation, Methodology, Supervision, Writing &#x2013; original draft. KF: Validation, Methodology, Investigation, Writing &#x2013; review &amp; editing, Supervision. WB:&#xa0;Conceptualization, Investigation, Validation, Writing &#x2013; review &amp; editing, Writing &#x2013; original draft, Supervision.</p></sec>
<ack>
<title>Acknowledgments</title>
<p>We thank Dr. Renato Marelli, longtime president of the Swiss Society of Insurance Medicine, for his continuous advice and support. We thank all participating claimants and expert psychiatrists, as well as the Zurich disability office for its assistance in recruiting claimants, and everyone for their commitment to this research effort.</p>
</ack>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p></sec>
<sec id="s11" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec id="s12" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p></sec>
<sec id="s13" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpsyt.2025.1643221/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpsyt.2025.1643221/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="DataSheet1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/></sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Baumberg Geiger</surname> <given-names>B</given-names></name>
<name><surname>Garthwaite</surname> <given-names>K</given-names></name>
<name><surname>Warren</surname> <given-names>J</given-names></name>
<name><surname>Bambra</surname> <given-names>C</given-names></name>
</person-group>. 
<article-title>Assessing Work Disability for Social Security Benefits: International Models for the Direct Assessment of Work Capacity</article-title>. <source>Disabil Rehabil</source>. (<year>2018</year>) <volume>40</volume>(<issue>24</issue>):<page-range>2962&#x2013;70</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/09638288.2017.1366556</pub-id>, PMID: <pub-id pub-id-type="pmid">28841811</pub-id>
</mixed-citation>
</ref>
<ref id="B2">
<label>2</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Barth</surname> <given-names>J</given-names></name>
<name><surname>de Boer</surname> <given-names>WE</given-names></name>
<name><surname>Busse</surname> <given-names>JW</given-names></name>
<name><surname>Hoving</surname> <given-names>JL</given-names></name>
<name><surname>Kedzia</surname> <given-names>S</given-names></name>
<name><surname>Couban</surname> <given-names>R</given-names></name>
<etal/>
</person-group>. 
<article-title>Inter-rater agreement in evaluation of disability: systematic review of reproducibility studies</article-title>. <source>BMJ</source>. (<year>2017</year>) <volume>356</volume>:<fpage>j14</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1136/bmj.j14</pub-id>, PMID: <pub-id pub-id-type="pmid">28122727</pub-id>
</mixed-citation>
</ref>
<ref id="B3">
<label>3</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Dickmann</surname> <given-names>JR</given-names></name>
<name><surname>Broocks</surname> <given-names>A</given-names></name>
</person-group>. 
<article-title>Psychiatric expert opinion in case of early retirement&#x2013;how reliable]</article-title>? <source>Fortschr Neurol Psychiatr</source>. (<year>2007</year>) <volume>75</volume>:<fpage>397</fpage>&#x2013;<lpage>401</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1055/s-2006-944303</pub-id>, PMID: <pub-id pub-id-type="pmid">17031778</pub-id>
</mixed-citation>
</ref>
<ref id="B4">
<label>4</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>de Vet</surname> <given-names>HC</given-names></name>
<name><surname>Terwee</surname> <given-names>CB</given-names></name>
<name><surname>Knol</surname> <given-names>DL</given-names></name>
<name><surname>Bouter</surname> <given-names>LM</given-names></name>
</person-group>. 
<article-title>When to use agreement versus reliability measures</article-title>. <source>J Clin Epidemiol</source>. (<year>2006</year>) <volume>59</volume>:<page-range>1033&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jclinepi.2005.10.015</pub-id>, PMID: <pub-id pub-id-type="pmid">16980142</pub-id>
</mixed-citation>
</ref>
<ref id="B5">
<label>5</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Kottner</surname> <given-names>J</given-names></name>
<name><surname>Audige</surname> <given-names>L</given-names></name>
<name><surname>Brorson</surname> <given-names>S</given-names></name>
<name><surname>Donner</surname> <given-names>A</given-names></name>
<name><surname>Gajewski</surname> <given-names>BJ</given-names></name>
<name><surname>Hrobjartsson</surname> <given-names>A</given-names></name>
<etal/>
</person-group>. 
<article-title>Guidelines for reporting reliability and agreement studies (GRRAS) were proposed</article-title>. <source>J Clin Epidemiol</source>. (<year>2011</year>) <volume>64</volume>:<fpage>96</fpage>&#x2013;<lpage>106</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jclinepi.2010.03.002</pub-id>, PMID: <pub-id pub-id-type="pmid">21130355</pub-id>
</mixed-citation>
</ref>
<ref id="B6">
<label>6</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Tinsley</surname> <given-names>HEA</given-names></name>
<name><surname>Weiss</surname> <given-names>DJ</given-names></name>
</person-group>. 
<article-title>Interrater reliability and agreement</article-title>. In: <source>Handbook of applied multivariate statistics and mathematical modeling</source>. 
<publisher-name>Academic Press</publisher-name>, <publisher-loc>San Diego</publisher-loc> (<year>2000</year>). p. <fpage>95</fpage>&#x2013;<lpage>124</lpage>.
</mixed-citation>
</ref>
<ref id="B7">
<label>7</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Guyatt</surname> <given-names>G</given-names></name>
<name><surname>Walter</surname> <given-names>S</given-names></name>
<name><surname>Norman</surname> <given-names>G</given-names></name>
</person-group>. 
<article-title>Measuring change over time: assessing the usefulness of evaluative instruments</article-title>. <source>J Chronic Dis</source>. (<year>1987</year>) <volume>40</volume>:<page-range>171&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/0021-9681(87)90069-5</pub-id>, PMID: <pub-id pub-id-type="pmid">3818871</pub-id>
</mixed-citation>
</ref>
<ref id="B8">
<label>8</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Spanjer</surname> <given-names>J</given-names></name>
<name><surname>Krol</surname> <given-names>B</given-names></name>
<name><surname>Brouwer</surname> <given-names>S</given-names></name>
<name><surname>Groothoff</surname> <given-names>JW</given-names></name>
</person-group>. 
<article-title>Inter-rater reliability in disability assessment based on a semi-structured interview report</article-title>. <source>Disabil Rehabil</source>. (<year>2008</year>) <volume>30</volume>:<page-range>1885&#x2013;90</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/09638280701688185</pub-id>, PMID: <pub-id pub-id-type="pmid">19037781</pub-id>
</mixed-citation>
</ref>
<ref id="B9">
<label>9</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Ebner</surname> <given-names>G</given-names></name>
<name><surname>Dittmann</surname> <given-names>V</given-names></name>
<name><surname>Mager</surname> <given-names>R</given-names></name>
<name><surname>Stieglitz</surname> <given-names>R-D</given-names></name>
<name><surname>Tr&#xe4;bert</surname> <given-names>S</given-names></name>
<name><surname>B&#xfc;hrlen</surname> <given-names>B</given-names></name>
<etal/>
</person-group>. 
<article-title>Final report: Development of guidelines for the assessment of mental disabilities</article-title>. In: <source>The formal quality of psychiatric assessment reports</source>. 
<publisher-name>Basel Universit&#xe4;re Psychiatrische Kliniken</publisher-name>, <publisher-loc>UPK</publisher-loc> (<year>2011</year>).
</mixed-citation>
</ref>
<ref id="B10">
<label>10</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Colomb</surname> <given-names>E</given-names></name>
<name><surname>Dittmann</surname> <given-names>V</given-names></name>
<name><surname>Ebner</surname> <given-names>G</given-names></name>
<name><surname>Hermelink</surname> <given-names>U</given-names></name>
<name><surname>Hoffmann-Richter</surname> <given-names>U</given-names></name>
<name><surname>Kopp</surname> <given-names>HG</given-names></name>
<etal/>
</person-group>. <source>Qualit&#xe4;tsleitlinien f&#xfc;r psychiatrische Gutachten in der Eidgen&#xf6;ssischen Invalidenversicherung</source>. <publisher-loc>Steinhausen</publisher-loc>: 
<publisher-name>Swiss Society of Psychiatry and Psychotherapy</publisher-name> (<year>2012</year>).
</mixed-citation>
</ref>
<ref id="B11">
<label>11</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Ebner</surname> <given-names>G</given-names></name>
<name><surname>Colomb</surname> <given-names>E</given-names></name>
<name><surname>Mager</surname> <given-names>R</given-names></name>
<name><surname>Marelli</surname> <given-names>R</given-names></name>
<name><surname>Rota</surname> <given-names>F</given-names></name>
</person-group>. <source>Quality guidelines for insurer reports on psychiatric assessments</source> (<year>2016</year>). <publisher-loc>Bern</publisher-loc>: 
<publisher-name>Swiss Society of Psychiatry and Psychotherapy [SGPP]</publisher-name>. Available online at: <uri xlink:href="https://www.psychiatrie.ch/sgpp/fachleute-und-kommissionen/leitlinien">https://www.psychiatrie.ch/sgpp/fachleute-und-kommissionen/leitlinien</uri> (Accessed <date-in-citation content-type="access-date">November 4, 2025</date-in-citation>).
</mixed-citation>
</ref>
<ref id="B12">
<label>12</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Riemer-Kafka</surname> <given-names>G</given-names></name>
</person-group>. <source>Medical expertises for insurers. An interdisciplinary guidance on medical and legal issues</source>. <edition>2nd ed</edition>. <publisher-loc>AG Bern</publisher-loc>: 
<publisher-name>Universit&#xe4;t Luzern: St&#xe4;mpfli Verlag</publisher-name> (<year>2012</year>).
</mixed-citation>
</ref>
<ref id="B13">
<label>13</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Kunz</surname> <given-names>R</given-names></name>
<name><surname>von Allmen</surname> <given-names>DY</given-names></name>
<name><surname>Marelli</surname> <given-names>R</given-names></name>
<name><surname>Hoffmann-Richter</surname> <given-names>U</given-names></name>
<name><surname>Jeger</surname> <given-names>J</given-names></name>
<name><surname>Mager</surname> <given-names>R</given-names></name>
<etal/>
</person-group>. 
<article-title>The reproducibility of psychiatric evaluations of work disability: two reliability and agreement studies</article-title>. <source>BMC Psychiatry</source>. (<year>2019</year>) <volume>19</volume>:<fpage>205</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12888-019-2171-y</pub-id>, PMID: <pub-id pub-id-type="pmid">31266488</pub-id>
</mixed-citation>
</ref>
<ref id="B14">
<label>14</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Bachmann</surname> <given-names>M</given-names></name>
<name><surname>de Boer</surname> <given-names>W</given-names></name>
<name><surname>Schandelmaier</surname> <given-names>S</given-names></name>
<name><surname>Leibold</surname> <given-names>A</given-names></name>
<name><surname>Marelli</surname> <given-names>R</given-names></name>
<name><surname>Jeger</surname> <given-names>J</given-names></name>
<etal/>
</person-group>. 
<article-title>Use of a structured functional evaluation process for independent medical evaluations of claimants presenting with disabling mental illness: rationale and design for a multi-center reliability study</article-title>. <source>BMC Psychiatry</source>. (<year>2016</year>) <volume>16</volume>:<fpage>271</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12888-016-0967-6</pub-id>, PMID: <pub-id pub-id-type="pmid">27474008</pub-id>
</mixed-citation>
</ref>
<ref id="B15">
<label>15</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>de Boer</surname> <given-names>W</given-names></name>
<name><surname>Marelli</surname> <given-names>R</given-names></name>
<name><surname>Hoffmann-Richter</surname> <given-names>U</given-names></name>
<name><surname>Eichhorn</surname> <given-names>M</given-names></name>
<name><surname>Jeger</surname> <given-names>J</given-names></name>
<name><surname>Colomb</surname> <given-names>E</given-names></name>
<etal/>
</person-group>. <source>Functional assessment in psychiatry. A manual. [Die funktionsorientierte begutachtung in der psychiatrie</source>. <publisher-loc>Basel</publisher-loc>: 
<publisher-name>Research &amp; Education, University of Basel</publisher-name> (<year>2015</year>).
</mixed-citation>
</ref>
<ref id="B16">
<label>16</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>von Allmen</surname> <given-names>DY</given-names></name>
<name><surname>Kedzia</surname> <given-names>S</given-names></name>
<name><surname>Dettwiler</surname> <given-names>R</given-names></name>
<name><surname>Vogel</surname> <given-names>N</given-names></name>
<name><surname>Kunz</surname> <given-names>R</given-names></name>
<name><surname>de Boer</surname> <given-names>WEL</given-names></name>
</person-group>. 
<article-title>Functional interviewing was associated with improved agreement among expert psychiatrists in estimating claimant work capacity: A secondary data analysis of real-life work disability evaluations</article-title>. <source>Front Psychiatry</source>. (<year>2020</year>) <volume>11</volume>:<elocation-id>621</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpsyt.2020.00621</pub-id>, PMID: <pub-id pub-id-type="pmid">32719624</pub-id>
</mixed-citation>
</ref>
<ref id="B17">
<label>17</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Linden</surname> <given-names>M</given-names></name>
<name><surname>Baron</surname> <given-names>S</given-names></name>
<name><surname>Muschalla</surname> <given-names>B</given-names></name>
</person-group>. <source>Mini-ICF-APP. Mini-ICF-Rating f&#xfc;r Aktivit&#xe4;ts- und Partizipationsbeeintr&#xe4;chtigungen bei psychischen Erkrankungen</source>. <edition>2 ed</edition>. <publisher-loc>Bern</publisher-loc>: 
<publisher-name>Hogrefe</publisher-name> (<year>2015</year>).
</mixed-citation>
</ref>
<ref id="B18">
<label>18</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Balestrieri</surname> <given-names>M</given-names></name>
<name><surname>Isola</surname> <given-names>M</given-names></name>
<name><surname>Bonn</surname> <given-names>R</given-names></name>
<name><surname>Tam</surname> <given-names>T</given-names></name>
<name><surname>Vio</surname> <given-names>A</given-names></name>
<name><surname>Linden</surname> <given-names>M</given-names></name>
<etal/>
</person-group>. 
<article-title>Validation of the Italian version of Mini-ICF-APP, a short instrument for rating activity and participation restrictions in psychiatric disorders</article-title>. <source>Epidemiol Psychiatr Sci</source>. (<year>2013</year>) <volume>22</volume>:<fpage>81</fpage>&#x2013;<lpage>91</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1017/S2045796012000480</pub-id>, PMID: <pub-id pub-id-type="pmid">22989494</pub-id>
</mixed-citation>
</ref>
<ref id="B19">
<label>19</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Molodynski</surname> <given-names>A</given-names></name>
<name><surname>Linden</surname> <given-names>M</given-names></name>
<name><surname>Juckel</surname> <given-names>G</given-names></name>
<name><surname>Yeeles</surname> <given-names>K</given-names></name>
<name><surname>Anderson</surname> <given-names>C</given-names></name>
<name><surname>Vazquez-Montes</surname> <given-names>M</given-names></name>
<etal/>
</person-group>. 
<article-title>The reliability, validity, and applicability of an English language version of the Mini-ICF-APP</article-title>. <source>Soc Psychiatry Psychiatr Epidemiol</source>. (<year>2013</year>) <volume>48</volume>:<page-range>1347&#x2013;54</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00127-012-0604-8</pub-id>, PMID: <pub-id pub-id-type="pmid">23080483</pub-id>
</mixed-citation>
</ref>
<ref id="B20">
<label>20</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Baron</surname> <given-names>S</given-names></name>
<name><surname>Linden</surname> <given-names>M</given-names></name>
</person-group>. 
<article-title>Disorders of functions and disorders of capacity in relation to sick leave in mental disorders</article-title>. <source>Int J Soc Psychiatry</source>. (<year>2009</year>) <volume>55</volume>:<fpage>57</fpage>&#x2013;<lpage>63</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1177/0020764008091660</pub-id>, PMID: <pub-id pub-id-type="pmid">19129326</pub-id>
</mixed-citation>
</ref>
<ref id="B21">
<label>21</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Streiner</surname> <given-names>DL</given-names></name>
<name><surname>Norman</surname> <given-names>GR</given-names></name>
<name><surname>Cairney</surname> <given-names>J</given-names></name>
</person-group>. <source>Health measurement scales. A practical guide to their development and use</source>. <edition>5th ed</edition>. <publisher-loc>Oxford</publisher-loc>: 
<publisher-name>Oxford University Press</publisher-name> (<year>2015</year>).
</mixed-citation>
</ref>
<ref id="B22">
<label>22</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Karanicolas</surname> <given-names>PJ</given-names></name>
<name><surname>Bhandari</surname> <given-names>M</given-names></name>
<name><surname>Kreder</surname> <given-names>H</given-names></name>
<name><surname>Moroni</surname> <given-names>A</given-names></name>
<name><surname>Richardson</surname> <given-names>M</given-names></name>
<name><surname>Walter</surname> <given-names>SD</given-names></name>
<etal/>
</person-group>. 
<article-title>Evaluating agreement: conducting a reliability study</article-title>. <source>J Bone Joint Surg Am volume</source>. (<year>2009</year>) <volume>3</volume>:<fpage>99</fpage>&#x2013;<lpage>106</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2106/JBJS.H.01624</pub-id>, PMID: <pub-id pub-id-type="pmid">19411507</pub-id>
</mixed-citation>
</ref>
<ref id="B23">
<label>23</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Hernaez</surname> <given-names>R</given-names></name>
</person-group>. 
<article-title>Reliability and agreement studies: a guide for clinical investigators</article-title>. <source>Gut</source>. (<year>2015</year>) <volume>64</volume>:<page-range>1018&#x2013;27</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1136/gutjnl-2014-308619</pub-id>, PMID: <pub-id pub-id-type="pmid">25873640</pub-id>
</mixed-citation>
</ref>
<ref id="B24">
<label>24</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Ranganathan</surname> <given-names>P</given-names></name>
<name><surname>Pramesh</surname> <given-names>CS</given-names></name>
<name><surname>Aggarwal</surname> <given-names>R</given-names></name>
</person-group>. 
<article-title>Common pitfalls in statistical analysis: Measures of agreement</article-title>. <source>Perspect Clin Res</source>. (<year>2017</year>) <volume>8</volume>:<page-range>187&#x2013;91</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4103/picr.PICR_123_17</pub-id>, PMID: <pub-id pub-id-type="pmid">29109937</pub-id>
</mixed-citation>
</ref>
<ref id="B25">
<label>25</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Linden</surname> <given-names>M</given-names></name>
<name><surname>Baron</surname> <given-names>S</given-names></name>
<name><surname>Muschalla</surname> <given-names>B</given-names></name>
<name><surname>Ostholt-Corsten</surname> <given-names>M</given-names></name>
</person-group>. <source>F&#xe4;higkeitsbeeintr&#xe4;chtigungen bei psychischen Erkrankungen. Diagnostik, Therapie und sozialmedizinische Beurteilung in Anlehnung an das Mini-ICF-APP</source>. <publisher-loc>Bern</publisher-loc>: 
<publisher-name>Huber</publisher-name> (<year>2015</year>).
</mixed-citation>
</ref>
<ref id="B26">
<label>26</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author"><collab>World Health Organisation</collab>
</person-group>. <source>International classification of functioning</source> (<year>2001</year>). 
<publisher-name>Disability and Health</publisher-name>. Available online at: <uri xlink:href="http://www.who.int/classifications/icf/en/">http://www.who.int/classifications/icf/en/</uri> (Accessed <date-in-citation content-type="access-date">February 12, 2025</date-in-citation>).
</mixed-citation>
</ref>
<ref id="B27">
<label>27</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>McGraw</surname> <given-names>K</given-names></name>
<name><surname>Wong</surname> <given-names>S</given-names></name>
</person-group>. 
<article-title>Forming inferences about some intraclass correlation coefficient</article-title>. <source>psychol Methods</source>. (<year>1996</year>) <volume>1</volume>:<fpage>30</fpage>&#x2013;<lpage>46</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1037/1082-989X.1.1.30</pub-id>
</mixed-citation>
</ref>
<ref id="B28">
<label>28</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Koo</surname> <given-names>TK</given-names></name>
<name><surname>Li</surname> <given-names>MY</given-names></name>
</person-group>. 
<article-title>A guideline of selecting and reporting intraclass correlation coefficients for reliability research</article-title>. <source>J chiropractic Med</source>. (<year>2016</year>) <volume>15</volume>:<page-range>155&#x2013;63</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jcm.2016.02.012</pub-id>, PMID: <pub-id pub-id-type="pmid">27330520</pub-id>
</mixed-citation>
</ref>
<ref id="B29">
<label>29</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Shrout</surname> <given-names>PE</given-names></name>
<name><surname>Fleiss</surname> <given-names>JL</given-names></name>
</person-group>. 
<article-title>Intraclass correlations: uses in assessing rater reliability</article-title>. <source>psychol bulletin</source>. (<year>1979</year>) <volume>86</volume>:<page-range>420&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1037/0033-2909.86.2.420</pub-id>, PMID: <pub-id pub-id-type="pmid">18839484</pub-id>
</mixed-citation>
</ref>
<ref id="B30">
<label>30</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Fleiss</surname> <given-names>JL</given-names></name>
</person-group>. <source>Statistical methods for rates and proportions</source>. <edition>2 ed</edition>. <publisher-loc>New York</publisher-loc>: 
<publisher-name>Wiley</publisher-name> (<year>1981</year>).
</mixed-citation>
</ref>
<ref id="B31">
<label>31</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Rosburg</surname> <given-names>T</given-names></name>
<name><surname>Kunz</surname> <given-names>R</given-names></name>
<name><surname>Trezzini</surname> <given-names>B</given-names></name>
<name><surname>Schwegler</surname> <given-names>U</given-names></name>
<name><surname>Jeger</surname> <given-names>J</given-names></name>
</person-group>. 
<article-title>The assessment of capacity limitations in psychiatric work disability evaluations by the social functioning scale Mini-ICF-APP</article-title>. <source>BMC Psychiatry</source>. (<year>2021</year>) <volume>21</volume>:<fpage>480</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12888-021-03467-w</pub-id>, PMID: <pub-id pub-id-type="pmid">34592979</pub-id>
</mixed-citation>
</ref>
<ref id="B32">
<label>32</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Jeger</surname> <given-names>J</given-names></name>
<name><surname>Trezzini</surname> <given-names>B</given-names></name>
<name><surname>Schwegler</surname> <given-names>U</given-names></name>
</person-group>. 
<article-title>Applying the ICF in disability evaluation: a report based on clinical experience</article-title>. In: 
<person-group person-group-type="editor">
<name><surname>Escorpizo</surname> <given-names>R</given-names></name>
<name><surname>Brage</surname> <given-names>S</given-names></name>
<name><surname>Homa</surname> <given-names>D</given-names></name>
<name><surname>Stucki</surname> <given-names>G</given-names></name>
</person-group>, editors. <source>Handbook of vocational rehabilitation and disability evaluation: Application and implementation of the ICF</source>. 
<publisher-name>Springer</publisher-name>, <publisher-loc>Cham</publisher-loc> (<year>2015</year>). p. <fpage>397</fpage>&#x2013;<lpage>410</lpage>.
</mixed-citation>
</ref>
<ref id="B33">
<label>33</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author"><collab>R Core Team</collab>
</person-group>. <source>R: A language and environment for statistical computing</source>. 
<publisher-name>R Foundation for Statistical Computing</publisher-name> (<year>2018</year>). Available online at: <uri xlink:href="https://www.r-project.org/">https://www.r-project.org/</uri> (Accessed <date-in-citation content-type="access-date">December 04, 2025</date-in-citation>).
</mixed-citation>
</ref>
<ref id="B34">
<label>34</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Crits-Christoph</surname> <given-names>P</given-names></name>
<name><surname>Johnson</surname> <given-names>J</given-names></name>
<name><surname>Gallop</surname> <given-names>R</given-names></name>
<name><surname>Gibbons</surname> <given-names>MB</given-names></name>
<name><surname>Ring-Kurtz</surname> <given-names>S</given-names></name>
<name><surname>Hamilton</surname> <given-names>JL</given-names></name>
<etal/>
</person-group>. 
<article-title>A generalizability theory analysis of group process ratings in the treatment of cocaine dependence</article-title>. <source>Psychother research: J Soc Psychother Res</source>. (<year>2011</year>) <volume>21</volume>:<page-range>252&#x2013;66</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/10503307.2010.551429</pub-id>, PMID: <pub-id pub-id-type="pmid">21409739</pub-id>
</mixed-citation>
</ref>
<ref id="B35">
<label>35</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Scherpenzeel</surname> <given-names>A</given-names></name>
</person-group>. <source>Why use 11-point scales</source>. <publisher-loc>Lausanne</publisher-loc>: 
<publisher-name>Swiss Center of Social Sciences in Lausanne</publisher-name> (<year>2018</year>). Available online at: <uri xlink:href="https://forscenter.ch/wp-content/uploads/2018/10/varia_11pointscales.pdf">https://forscenter.ch/wp-content/uploads/2018/10/varia_11pointscales.pdf</uri> (Accessed <date-in-citation content-type="access-date">December 04, 2025</date-in-citation>).
</mixed-citation>
</ref>
<ref id="B36">
<label>36</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Scherpenzeel</surname> <given-names>AC</given-names></name>
</person-group>. <source>A question of quality: evaluating survey questions by multi trait - multi method studies</source>. <publisher-loc>Amsterdam</publisher-loc>: 
<publisher-name>University of Amsterdam</publisher-name> (<year>1995</year>).
</mixed-citation>
</ref>
<ref id="B37">
<label>37</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Kobak</surname> <given-names>KA</given-names></name>
<name><surname>Brown</surname> <given-names>B</given-names></name>
<name><surname>Sharp</surname> <given-names>I</given-names></name>
<name><surname>Levy-Mack</surname> <given-names>H</given-names></name>
<name><surname>Wells</surname> <given-names>K</given-names></name>
<name><surname>Ockun</surname> <given-names>F</given-names></name>
<etal/>
</person-group>. 
<article-title>Sources of unreliability in depression ratings</article-title>. <source>J Clin psychopharmacology</source>. (<year>2009</year>) <volume>29</volume>:<page-range>82&#x2013;5</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1097/JCP.0b013e318192e4d7</pub-id>, PMID: <pub-id pub-id-type="pmid">19142114</pub-id>
</mixed-citation>
</ref>
<ref id="B38">
<label>38</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Anner</surname> <given-names>J</given-names></name>
<name><surname>Kunz</surname> <given-names>R</given-names></name>
<name><surname>de Boer</surname> <given-names>W</given-names></name>
</person-group>. 
<article-title>Reporting about disability evaluation in European countries</article-title>. <source>Disabil Rehabil</source>. (<year>2014</year>) <volume>36</volume>:<page-range>848&#x2013;54</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3109/09638288.2013.821180</pub-id>, PMID: <pub-id pub-id-type="pmid">23919642</pub-id>
</mixed-citation>
</ref>
<ref id="B39">
<label>39</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Schleifer</surname> <given-names>R</given-names></name>
<name><surname>Gamma</surname> <given-names>A</given-names></name>
<name><surname>Warnke</surname> <given-names>I</given-names></name>
<name><surname>Jabat</surname> <given-names>M</given-names></name>
<name><surname>Rossler</surname> <given-names>W</given-names></name>
<name><surname>Liebrenz</surname> <given-names>M</given-names></name>
</person-group>. 
<article-title>Online survey of medical and psychological professionals on structured instruments for the assessment of work ability in psychiatric patients</article-title>. <source>Front Psychiatry</source>. (<year>2018</year>) <volume>9</volume>:<elocation-id>453</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpsyt.2018.00453</pub-id>, PMID: <pub-id pub-id-type="pmid">30319460</pub-id>
</mixed-citation>
</ref>
<ref id="B40">
<label>40</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Rosburg</surname> <given-names>T</given-names></name>
<name><surname>Deuring</surname> <given-names>G</given-names></name>
<name><surname>Ebner</surname> <given-names>G</given-names></name>
<name><surname>Hauch</surname> <given-names>V</given-names></name>
<name><surname>Pflueger</surname> <given-names>MO</given-names></name>
<name><surname>Stieglitz</surname> <given-names>RD</given-names></name>
<etal/>
</person-group>. 
<article-title>Digitally Assisted Standard Diagnostics in Insurance Medicine (DASDIM): psychometric data in psychiatric work disability evaluations</article-title>. <source>Disabil Rehabil</source>. (<year>2023</year>) <volume>45</volume>:<page-range>4457&#x2013;70</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/09638288.2022.2151655</pub-id>, PMID: <pub-id pub-id-type="pmid">36523117</pub-id>
</mixed-citation>
</ref>
<ref id="B41">
<label>41</label>
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name><surname>Marfeo</surname> <given-names>EE</given-names></name>
<name><surname>Ni</surname> <given-names>P</given-names></name>
<name><surname>McDonough</surname> <given-names>C</given-names></name>
<name><surname>Peterik</surname> <given-names>K</given-names></name>
<name><surname>Marino</surname> <given-names>M</given-names></name>
<name><surname>Meterko</surname> <given-names>M</given-names></name>
<etal/>
</person-group>. 
<article-title>Improving assessment of work related mental health function using the work disability functional assessment battery (WD-FAB)</article-title>. <source>J Occup rehabilitation</source>. (<year>2018</year>) <volume>28</volume>:<page-range>190&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10926-017-9710-5</pub-id>, PMID: <pub-id pub-id-type="pmid">28477069</pub-id>
</mixed-citation>
</ref>
<ref id="B42">
<label>42</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Weinbrenner</surname> <given-names>S</given-names></name>
</person-group>. <source>Adaptation and validation of the Work Disability Functional Assessment Battery (WD-FAB) to German. German Pension Fund. EUMASS congress: Insurance Medicine 2.0 in a Changing World</source>. <publisher-loc>Strasbourg</publisher-loc>. (<year>2023</year>).
</mixed-citation>
</ref>
<ref id="B43">
<label>43</label>
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name><surname>Vermein</surname> <given-names>E</given-names></name>
<name><surname>van Damme</surname> <given-names>S</given-names></name>
</person-group>. <source>Psychometric validation of the work disability - Functional Assessment Battery (WD-FAB) for Belgium. 2023-2026. INAMI Institut national d&#x2019;assurance maladie-invalidit&#xe9; Ghent University</source> . 
<publisher-name>Ghent Health Psychology Lab</publisher-name>. Available online at: <uri xlink:href="https://research.ugent.be/web/result/project/1a573cf3-fbe6-4584-9a18-a69859e78f5d/details/160a00123-psychometric-validation-of-the-work-disability&#x2014;functional-assessment-battery-wd-fab-for-belgium/en">https://research.ugent.be/web/result/project/1a573cf3-fbe6-4584-9a18-a69859e78f5d/details/160a00123-psychometric-validation-of-the-work-disability&#x2014;functional-assessment-battery-wd-fab-for-belgium/en</uri> (Accessed <date-in-citation content-type="access-date">December 4, 2025</date-in-citation>).
</mixed-citation>
</ref>
</ref-list>
<fn-group>
<fn id="n1" fn-type="custom" custom-type="edited-by">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1727368">Ole Steen Mortensen</ext-link>, University of Copenhagen, Denmark</p></fn>
<fn id="n2" fn-type="custom" custom-type="reviewed-by">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3123418">&#x15e;eyda &#xd6;zal</ext-link>, Ankara Medipol University, T&#xfc;rkiye</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3195089">Ben Baumberg Geiger</ext-link>, King&#x2019;s College London, United Kingdom</p></fn>
</fn-group>
</back>
</article>