<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Comput. Sci.</journal-id>
<journal-title>Frontiers in Computer Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Comput. Sci.</abbrev-journal-title>
<issn pub-type="epub">2624-9898</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fcomp.2024.1371181</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Computer Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Evaluating the robustness of multimodal task load estimation models</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Foltyn</surname> <given-names>Andreas</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1827250/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Deuschel</surname> <given-names>Jessica</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lang-Richter</surname> <given-names>Nadine R.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Holzer</surname> <given-names>Nina</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Oppelt</surname> <given-names>Maximilian P.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2631647/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department Digital Health and Analytics, Fraunhofer IIS, Fraunhofer Institute for Integrated Circuits IIS</institution>, <addr-line>Erlangen</addr-line>, <country>Germany</country></aff>
<aff id="aff2"><sup>2</sup><institution>Machine Learning and Data Analytics Lab (MaD Lab), Department Artificial Intelligence in Biomedical Engineering, Friedrich-Alexander-University Erlangen Nuremberg</institution>, <addr-line>Erlangen</addr-line>, <country>Germany</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Luca Longo, Technological University Dublin, Ireland</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Giulia Vilone, Technological University Dublin, Ireland</p>
<p>Martin Gjoreski, University of Italian Switzerland, Switzerland</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Andreas Foltyn <email>andreas.foltyn&#x00040;iis.fraunhofer.de</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>10</day>
<month>04</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>6</volume>
<elocation-id>1371181</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>22</day>
<month>03</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Foltyn, Deuschel, Lang-Richter, Holzer and Oppelt.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Foltyn, Deuschel, Lang-Richter, Holzer and Oppelt</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Numerous studies have focused on constructing multimodal machine learning models for estimating a person&#x00027;s cognitive load. However, a prevalent limitation is that these models are typically evaluated on data from the same scenario they were trained on. Little attention has been given to their robustness against data distribution shifts, which may occur during deployment. The aim of this paper is to investigate the performance of these models when confronted with a scenario different from the one on which they were trained. For this evaluation, we utilized a dataset encompassing two distinct scenarios: an <italic>n</italic>-Back test and a driving simulation. We selected a variety of classic machine learning and deep learning architectures, which were further complemented by various fusion techniques. The models were trained on the data from the <italic>n</italic>-Back task and tested on both scenarios to evaluate their predictive performance. However, the predictive performance alone may not lead to a trustworthy model. Therefore, we looked at the uncertainty estimates of these models. By leveraging these estimates, we can reduce misclassification by resorting to alternative measures in situations of high uncertainty. The findings indicate that late fusion produces stable classification results across the examined models for both scenarios, enhancing robustness compared to feature-based fusion methods. Although a simple logistic regression tends to provide the best predictive performance for <italic>n</italic>-Back, this is not always the case if the data distribution is shifted. Finally, the predictive performance of individual modalities differs significantly between the two scenarios. This research provides insights into the capabilities and limitations of multimodal machine learning models in handling distribution shifts and identifies which approaches may potentially be suitable for achieving robust results.</p></abstract>
<kwd-group>
<kwd>cognitive load</kwd>
<kwd>task load</kwd>
<kwd>multimodal</kwd>
<kwd>robustness</kwd>
<kwd>machine learning</kwd>
<kwd>deep learning</kwd>
<kwd>uncertainty quantification</kwd>
</kwd-group>
<counts>
<fig-count count="6"/>
<table-count count="7"/>
<equation-count count="2"/>
<ref-count count="49"/>
<page-count count="14"/>
<word-count count="10153"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Human-Media Interaction</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Cognitive load refers to the subjective, physiological state of mental effort that results from the dynamic interplay between an individual&#x00027;s finite cognitive resources and the demands placed upon them by a task. Task load represents the objective assortment of demands that a task inherently imposes on an individual. Together, these concepts form the cornerstone of our understanding of how tasks affect performance and well-being. Therefore, the recognition of cognitive overload could improve various working environments, such as education (Antonenko et al., <xref ref-type="bibr" rid="B4">2010</xref>), public transportation (Fridman et al., <xref ref-type="bibr" rid="B21">2018</xref>; Wilson et al., <xref ref-type="bibr" rid="B48">2021</xref>) and situations that require high attention (Abrantes et al., <xref ref-type="bibr" rid="B1">2017</xref>). Consequently, there is a need for robust cognitive load estimation models that perform well in different environments.</p>
<p>In recent years, many studies have been conducted for cognitive load estimation. Some studies have encompassed the data collection of a range of scenarios, including standardized tests (Beh et al., <xref ref-type="bibr" rid="B10">2021</xref>), driving situations (Oppelt et al., <xref ref-type="bibr" rid="B37">2023</xref>), and aviation scenarios (Wilson et al., <xref ref-type="bibr" rid="B48">2021</xref>). Based on these data sets, approaches have been evaluated for their ability to infer cognitive load, utilizing different modalities (as outlined in <xref ref-type="table" rid="T1">Table 1</xref>). This body of research has explored classical machine learning strategies, advanced deep learning techniques, fusion methods and the impact of varying data processing approaches. Many of these papers primarily emphasize predictive performance while often overlooking two equally crucial evaluation aspects of learning systems: <italic>robustness</italic> and <italic>uncertainty estimation</italic>.</p>


<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Overview of recent publications evaluating cognitive load estimation models.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Reference</bold></th>
<th valign="top" align="left"><bold>Modalities</bold></th>
<th valign="top" align="left"><bold>Setup</bold></th>
<th valign="top" align="left"><bold>Stimulus</bold></th>
<th valign="top" align="left"><bold>Evaluation</bold></th>
<th valign="top" align="left"><bold>Window size</bold></th>
<th valign="top" align="left"><bold>Accuracy (<italic>%</italic>)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Meteier et al. (<xref ref-type="bibr" rid="B36">2021</xref>)</td>
<td valign="top" align="left"><bold>ECG</bold>, EDA, RESP</td>
<td valign="top" align="left">Driving simulation (90 subjects)</td>
<td valign="top" align="left">Oral backward counting</td>
<td valign="top" align="left">10-fold CV</td>
<td valign="top" align="left">4 minutes</td>
<td valign="top" align="left">95.0</td>
</tr> <tr>
<td valign="top" align="left">Aygun et al. (<xref ref-type="bibr" rid="B6">2022</xref>)</td>
<td valign="top" align="left">EEG, <bold>EYE</bold>, BP</td>
<td valign="top" align="left">Driving simulation (80 subjects)</td>
<td valign="top" align="left">Questions and braking events</td>
<td valign="top" align="left">1-fold CV</td>
<td valign="top" align="left">-</td>
<td valign="top" align="left">80.4</td>
</tr> <tr>
<td valign="top" align="left">Gjoreski et al. (<xref ref-type="bibr" rid="B23">2020b</xref>)</td>
<td valign="top" align="left">ACC, EDA, TEMP, PPG</td>
<td valign="top" align="left">Lab (23 subjects)</td>
<td valign="top" align="left">N-Back, standardized tests</td>
<td valign="top" align="left">LOSO nested-CV</td>
<td valign="top" align="left">30 seconds</td>
<td valign="top" align="left">68.2</td>
</tr> <tr>
<td valign="top" align="left">Kumar (<xref ref-type="bibr" rid="B31">2022</xref>)</td>
<td valign="top" align="left">EEG, ECG, EDA</td>
<td valign="top" align="left">Driving simulation (33 subjects)</td>
<td valign="top" align="left">Driving tasks</td>
<td valign="top" align="left">5-fold CV</td>
<td valign="top" align="left">25 seconds</td>
<td valign="top" align="left">85.6</td>
</tr> <tr>
<td valign="top" align="left">Beh et al. (<xref ref-type="bibr" rid="B10">2021</xref>)</td>
<td valign="top" align="left"><bold>ECG</bold>, EDA, PPG (Fingertip), PPG (Wrist)</td>
<td valign="top" align="left">Lab (22 subjects)</td>
<td valign="top" align="left">N-Back</td>
<td valign="top" align="left">LOSO</td>
<td valign="top" align="left">2 mins</td>
<td valign="top" align="left">71.6</td>
</tr> <tr>
<td valign="top" align="left">Kesed&#x0017E;i&#x00107; et al. (<xref ref-type="bibr" rid="B28">2021</xref>)</td>
<td valign="top" align="left">ECG, <bold>fNIRS</bold></td>
<td valign="top" align="left">Lab (32 subjects)</td>
<td valign="top" align="left">N-Back</td>
<td valign="top" align="left">LOSO</td>
<td valign="top" align="left">75 seconds</td>
<td valign="top" align="left">84.3</td>
</tr> <tr>
<td valign="top" align="left">Oppelt et al. (<xref ref-type="bibr" rid="B37">2023</xref>)</td>
<td valign="top" align="left">ECG, EDA, EMG, <bold>EYE</bold>, PPG, RESP, TEMP, AU</td>
<td valign="top" align="left">Lab and driving simulator (51 subjects)</td>
<td valign="top" align="left">N-back and multi-tasking</td>
<td valign="top" align="left">10x10 nested-CV</td>
<td valign="top" align="left">2 min</td>
<td valign="top" align="left">-</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>All results are based on a binary classification task to discriminate between low and high cognitive load. We show the best results reported in each respective publication and mark the best performing modality bold if the information is available.</p>
</table-wrap-foot>
</table-wrap>


<p>Robustness has different definitions in the literature. In this paper we refer to the ability of a model to maintain a relatively stable performance in spite of changes in the data distribution (Freiesleben and Grote, <xref ref-type="bibr" rid="B20">2023</xref>). Such changes may arise from slight variations in the scenario, such as increased movement of a person during model deployment in comparison to training data, or sudden corruptions of individual modalities. Despite the challenges inherent in maintaining accuracy under these circumstances, it is vital that the model possesses the capacity to know what it does not know. This enables reliable predictions that can be trusted, which can be achieved through the use of well-calibrated uncertainty estimates. Furthermore, it opens up possibilities to extend the functionality of cognitive load estimation systems. For example, in instances where a prediction is deemed excessively uncertain, it can be discarded outright, or alternatively, the user can be prompted to input their current cognitive load level, thus enabling the system to adapt itself accordingly. This leads to the the following research gaps we address in this paper:</p>
<p><bold>RQ1</bold> How do machine learning, fusion, and data processing methods influence predictive performance and uncertainty estimation in estimating task load?</p>
<p>In this paper, we use data from participants exposed to tasks of varying difficulty and use these task assignments as labels to directly address task load estimation. Section 3.1.3 goes into more detail on how this relates to the definition of cognitive load. As a first step, we aim to investigate a range of classical machine learning, deep learning and fusion methods within a standard evaluation process. This means training and test data will come from participants who have performed the same task, ensuring minimal distribution shift. This will serve as the basis for comparing our methods to address our main research question:</p>
<p><bold>RQ2</bold> How does a data distribution shift influence the classification accuracy and uncertainty estimation in estimating task load?</p>
<p>To address this question, we subject the trained models to a different scenario where the user had to perform a different task, and evaluate the models from RQ1 for robustness and quality of uncertainty estimation. Consequently, we make the following contributions in this paper. We conduct a systematic investigation of unimodal and multimodal approaches, examining their impact on in-distribution classification performance and uncertainty estimates. We examine these methods in terms of distribution shifts and demonstrate their robustness. Our general desiderata for the task load estimation system are presented in <xref ref-type="fig" rid="F1">Figure 1</xref>. Although this study does not directly measure cognitive load, it is important to note that task load, which we focus on, is related to cognitive load and can potentially serve as a noisy proxy for it. This potential is contingent upon verification, as we discuss in Section 3.1.3. Thus, our results can indeed be relevant to the assessment of cognitive load to a certain extent. However, it&#x00027;s crucial to underline that future research should aim to validate these findings with more precise cognitive load labels to ensure the applicability and accuracy of the assessment.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Overview of the experimental setup and desiderata of the model predictions. Data is collected from two domains: <italic>n</italic>-Back and <italic>k</italic>-Drive. The model is trained using the <italic>n</italic>-Back data and subsequently evaluated on both domains. Our main objective is to create a model that is highly accurate within its domain and capable of making robust predictions across other domains. It should also exhibit a high degree of uncertainty when faced with potential misclassifications.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-06-1371181-g0001.tif"/>
</fig>

</sec>
<sec id="s2">
<title>2 Related work</title>
<sec>
<title>2.1 Cognitive load</title>
<p>Cognitive load, intricately linked to the notion of mental workload, is a complex concept and has been defined in various ways throughout the literature (Paas and Van Merri&#x000EB;nboer, <xref ref-type="bibr" rid="B40">1994</xref>; Haapalainen et al., <xref ref-type="bibr" rid="B25">2010</xref>; Orru and Longo, <xref ref-type="bibr" rid="B38">2019</xref>; Longo et al., <xref ref-type="bibr" rid="B34">2022</xref>). At its core, cognitive load encompasses the subjective physiological state of mental effort that emerges from the dynamic interplay between an individual&#x00027;s finite cognitive resources and the demands placed on them by a task. This understanding acknowledges both the subject involved and the task at hand as fundamental components in the conceptualization of cognitive load. Notably, Longo et al. (<xref ref-type="bibr" rid="B34">2022</xref>) offer a precise definition of cognitive load, describing it as &#x0201C;the degree of activation of a finite pool of resources, limited in capacity, while cognitively processing a primary task over time, mediated by external dynamic environmental and situational factors, as well as affected by static definite internal characteristics of a human operator, for coping with static task demands, by devoted effort and attention.&#x0201D; This definition highlights the interaction between external factors and an individual&#x00027;s inherent capabilities. In this context, task load represents the objective measure of task demands that directly influence this interplay.</p></sec>
<sec>
<title>2.2 Cognitive load measurement</title>
<p>Cognitive overload can be measured using subjective and objective approaches. Self-assessments can be used to get the subjective measure of workload, typically using standardized questionnaires such as the NASA Task Load Index (NASA-TLX) (Hart and Staveland, <xref ref-type="bibr" rid="B26">1988</xref>). However, self-assessment questionnaires are limited in that they are usually completed after a task is performed, and they rely on individual perceptions, which can vary across people. Objective measures of cognitive load can be obtained through performance evaluation on a task and physiological measures. Performance-based measures can be categorized into two types. The first is based solely on the primary task performance, while the second considers both the primary and a secondary task performance as an indicator of workload. The evaluation of workload focuses on the spare mental capacity for the secondary task given the primary task demands (Paas et al., <xref ref-type="bibr" rid="B39">2003</xref>). However, performance can also be influenced by other factors, such as the strategy a subject uses to solve the tasks, or the duration of the stimulus, whereby fatique causes a reduction in performance (Cain, <xref ref-type="bibr" rid="B14">2007</xref>).</p>
<p>Another objective measure of cognitive load is physiological signals, which can be measured using various modalities. Eye tracking, for example, is an important indicator for detecting cognitive overload. Pupil dilation, saccades, blinks and fixations are the most important features that can be extracted from eye tracking data. In particular, pupil dilation is considered a good indicator of cognitive overload in the literature (Beatty, <xref ref-type="bibr" rid="B8">1982</xref>; Palinko and Kun, <xref ref-type="bibr" rid="B41">2012</xref>; Ayres et al., <xref ref-type="bibr" rid="B7">2021</xref>; Rahman et al., <xref ref-type="bibr" rid="B44">2021</xref>), as the variation can be influenced by emotional and cognitive processes (Bradley et al., <xref ref-type="bibr" rid="B13">2008</xref>). However, pupil size also changes with illumination, which is a challenge to consider when analysing eye tracking data (Beatty and Lucero-Wagoner, <xref ref-type="bibr" rid="B9">2000</xref>). Electroencephalography (EEG) can measure brain activity using electrodes placed on the head. These can be used to draw conclusions about cognitive processes. In principle, brain activity is a good indicator of cognitive processes (Ayres et al., <xref ref-type="bibr" rid="B7">2021</xref>; Zhou et al., <xref ref-type="bibr" rid="B49">2022</xref>). However, it is susceptible to noise artifacts caused by movements and the exact placement of the electrodes influences the correct measurement of the signals.</p>
<p>Cardiovascular activities can also provide insight into variations in cognitive overload. These include, for example, heart rate or heart rate variability (Ayres et al., <xref ref-type="bibr" rid="B7">2021</xref>). However, there are some psychological and physical factors that can also influence these variables, such as activity and affective states.</p></sec>
<sec>
<title>2.3 Machine learning and cognitive load estimation</title>
<p>As discussed in Section 2.1, cognitive load is not a simple construct. When training machine learning models, the data must be annotated accordingly. This is generally a challenge in the field of affective computing when inferring psychological constructs (Booth et al., <xref ref-type="bibr" rid="B12">2018</xref>). In the case of cognitive overload, various options can be considered for annotating the data, each with its own challenges. On one hand, the difficulty level of designated phases can be uniformly labeled across all individuals, regardless of their subjective experience, which is done by many approaches in the literature (Gjoreski et al., <xref ref-type="bibr" rid="B22">2020a</xref>; Oppelt et al., <xref ref-type="bibr" rid="B37">2023</xref>). While most papers refer to cognitive load or mental load, they are actually inferring task load. It&#x00027;s important to clarify that cognitive load can vary due to different influencing factors, even when the difficulty of the task remains constant. Alternatively, subjective ratings of participants can be used, such as the NASA-TLX. Although perceived cognitive load can be depicted, the issue lies in the fact that self-assessments are often challenging to compare across individuals. Finally, while performance can also serve as an indicator, its reliability can be compromised by factors unrelated to cognitive load. These include solution strategies, attention or fatigue. Seitz and Maedche (<xref ref-type="bibr" rid="B45">2022</xref>) provide a thorough overview of many cognitive load datasets and their respective types of annotation. There are also approaches that combine the mentioned methods for annotation (Dolmans et al., <xref ref-type="bibr" rid="B19">2021</xref>).</p>
<p>Additionally, there is the question of how to formulate the target. In most publications, classification is used. Binary classification is the most commonly used method. A three-class division often proves to be quite difficult, presumably because it reflects the fuzziness of the construct (Gjoreski et al., <xref ref-type="bibr" rid="B23">2020b</xref>). Alternatively, the problem can also be formulated as regression (Oppelt et al., <xref ref-type="bibr" rid="B37">2023</xref>).</p>
<p>Models have been trained and evaluated for various scenarios in the literature. Several studies have been conducted using different modalities to infer cognitive load in the <italic>n</italic>-Back test (Beh et al., <xref ref-type="bibr" rid="B10">2021</xref>; Kesed&#x0017E;i&#x00107; et al., <xref ref-type="bibr" rid="B28">2021</xref>; Oppelt et al., <xref ref-type="bibr" rid="B37">2023</xref>). Machine learning approaches have also been evaluated for application-oriented scenarios. Wilson et al. (<xref ref-type="bibr" rid="B48">2021</xref>) used an aviation simulation to induce overload through context-specific tasks. A number of papers have conducted driving simulations under laboratory conditions (Meteier et al., <xref ref-type="bibr" rid="B36">2021</xref>; Oppelt et al., <xref ref-type="bibr" rid="B37">2023</xref>) or in real-world settings (Fridman et al., <xref ref-type="bibr" rid="B21">2018</xref>). Most datasets have been recorded under optimal conditions, making it unclear how robust models are to factors that could influence the ability to detect cognitive load. Albuquerque et al. (<xref ref-type="bibr" rid="B3">2020</xref>) used physical exercise to create an additional factor that influences the expression of some modalities, e.g. heart rate or skin conductance.</p>
<p><xref ref-type="table" rid="T1">Table 1</xref> provides an overview of relevant papers with important parameters of the experimental design and their results for binary classification. It encompasses data from EEG, electrocardiogram (ECG), photoplethysmography (PPG), blood pressure (BP), electromyogram (EMG), electrodermal activity (EDA), respiration rate (RESP), eye tracker (EYE), skin temperature (TEMP), accelaration data (ACC), and action units (AU). The selected results all come from experiments in which subject wise splitting was used. However, the exact evaluation protocols differ. The data splitting ranges from a 1-fold cross validation (CV) to a leave-one-subject-out (LOSO). It is important to note that comparing these results in a meaningful way is challenging, as many factors can impact the performance.</p>
<p>In other domains, deep learning approaches have already replaced classical machine learning methods that require expert features. An important prerequisite for this is having enough data to extract generalizable features. Aygun et al. (<xref ref-type="bibr" rid="B6">2022</xref>) provide a comparison of classical machine learning approaches with Convolutional Neural Networks (CNNs) and Recurrent Neural Networks (RNNs) that extract relevant features from raw physiological signals. In this case, methods that rely on expert features perform better. However, since deep learning approaches often provide good results in general time series literature, a possible reason for the difference is the small number of data points. Self-supervised approaches with EEG have been investigated for cognitive load estimation in this regard (Longo, <xref ref-type="bibr" rid="B32">2022</xref>).</p></sec></sec>
<sec sec-type="methods" id="s3">
<title>3 Methodology</title>
<sec>
<title>3.1 Data</title>
<sec>
<title>3.1.1 Data description</title>
<p>To answer our research questions, we need a cognitive load dataset that includes different modalities and also contains more than one stimulus in order to investigate robustness with respect to a data shift. Therefore, we use <italic>ADA</italic>Base (Oppelt et al., <xref ref-type="bibr" rid="B37">2023</xref>) for our experiments. In this dataset, two stimuli were utilized to induce cognitive overload: the <italic>n</italic>-Back test (Kirchner, <xref ref-type="bibr" rid="B30">1958</xref>) and the simulated driving situation <italic>k</italic>-Drive. Both scenarios begin with baselines, in which no overload stimulus is provided. Subsequently, the subjects have to pass through different levels of difficulty, where they are increasingly overloaded. The <italic>n</italic>-Back test (Kirchner, <xref ref-type="bibr" rid="B30">1958</xref>) is a standardized test for measuring the working memory capacity, where participants must remember the position of elements on a grid <italic>n</italic> steps before. By increasing <italic>n</italic>, the load on the working memory increases. In this dataset, participants were required to complete the <italic>n</italic>-Back test with varying difficulty levels, ranging from 1 to 3 steps. In addition to the single-task test described above, participants also completed a dual-task variant, as described by Jaeggi et al. (<xref ref-type="bibr" rid="B27">2003</xref>). This version introduces an auditory memory component that has to be performed concurrently with the visual task. In this setup, consonants are spoken by a computerized voice and have to be memorized using the <italic>n</italic>-Back approach analogous to that used in the visual task. This variant also included three levels of difficulty to further challenge participants. However, we do not use the dual-task data, as this would give us considerably more data in which the participants are overloaded. This would lead to class imbalance. Therefore, in the following we will only use data from the single-task variant. For the simulated driving situation <italic>k</italic>-Drive, the subjects had to react to various events, such as braking, in a driving scene. In the first level, subjects had to respond to only a few simple events during the driving scene. In levels 2 and 3, the complexity increased and a secondary task involved creating a playlist on a tablet. The dataset includes the following modalities: eye tracking, ECG, PPG, EDA, EMG, respiration, skin temperature, and action units. However, we exclude action units from the experiments due to the low predictive performance in Oppelt et al. (<xref ref-type="bibr" rid="B37">2023</xref>). The full dataset contains a total of 51 subjects. Individual erroneous or missing modalities were found in 5 subjects. These subjects were removed, leaving data from 46 subjects available for the experiments. All these subjects completed both <italic>n</italic>-Back and <italic>k</italic>-Drive.</p></sec>
<sec>
<title>3.1.2 Preprocessing</title>
<p>In the following section, we outline the preprocessing steps applied to the raw signals to make them suitable for training machine learning models. Initially, modality-specific preprocessing is conducted to eliminate potential artifacts, as outlined in Oppelt et al. (<xref ref-type="bibr" rid="B37">2023</xref>). This primarily involves removing outliers from the eye tracker data and detrending the ECG signal. Subsequently, the entire dataset is segmented into individual time frames using a rolling window approach. For our main experiments, we opt for a window size of 60 seconds with a stride of 10 seconds. Windows are selected only if at least the first 80% of the window matches a labeled segment, discarding windows where the label is more ambiguous. These extracted windows serve as the fundamental input for both the deep learning models and further feature extraction for the classical models. For all modalities we employ the same feature set utilized in the <italic>ADA</italic>Base publication (Oppelt et al., <xref ref-type="bibr" rid="B37">2023</xref>). The type of normalization plays an important role in affective computing. Features often have an individual-specific range, which means that subject normalization can lead to improved performance. This involves extracting the normalization parameters from the subject&#x00027;s data which is therefore only possible <italic>post hoc</italic>. For our main experiments, we use a subject wise z-score normalization using the mean and standard deviation per modality and subject. However, we also investigate the influence of normalization on in-domain and especially out-domain performance. For this purpose, an ablation study is also used to apply no normalization and a global normalization whose parameters are calculated from all training data points.</p></sec>
<sec>
<title>3.1.3 Annotation</title>
<p><italic>ADA</italic>Base provides three types of annotations that can potentially be used for machine learning tasks related to cognitive load: self-assessment using the NASA-TLX questionnaire, performance metrics, and information about the stimuli that mark the difficulty levels.</p>
<p>To understand whether these annotation strategies have the potential to be quantified, we align them with the components outlined in the Longo et al. (<xref ref-type="bibr" rid="B34">2022</xref>) definition of cognitive load. This definition states that the degree of activation of the finite pool of cognitive resources is influenced by various factors, including environmental and situational contexts, subject-specific internal characteristics, task demand, and the amount of effort and attention dedicated to a particular task. The NASA-TLX self-assessment method has the potential to represent perceived cognitive load as it can account for the impact of factors such as effort and task demand. However, self-assessments may be subject to biases and may not be consistently comparable across participants. Variations in the performance of primary tasks, such as the <italic>n</italic>-Back task, can be indicative of cognitive overload. Nonetheless, performance fluctuations can also be attributed to other factors like effort or attention, which may affect performance independently of the actual degree of activation of the finite cognitive resources.</p>
<p>In this study, we use the difficulty level of each task to create a binary classification. This allows us to operationalize the task demand as defined in the definition, as the annotation remains constant regardless of individual experience. Consequently, other factors, such as &#x0201C;effort and attention&#x0201D; or &#x0201C;internal characteristics,&#x0201D; are not taken into consideration. However, as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, an increase in task difficulty is associated with an increased perception of cognitive load, as evidenced by the NASA-TLX self-assessments. Since we are making a rough binary distinction between a scenario with very little to no task demand and a demanding task, and there is a clear trend in perceived cognitive load when distinguishing between these two conditions, the label can be considered a noisy proxy for cognitive load, even though it only represents the task load component.</p>


<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>The NASA-TLX raw ratings of unweighted mental demand for each scenario in the <italic>ADA</italic>Base dataset. <bold>(A)</bold> <italic>k</italic>-Drive. <bold>(B)</bold> <italic>n</italic>-Back.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-06-1371181-g0002.tif"/>
</fig>


<p>Next, we show the precise methodology employed in generating labels based on task difficulty. We differentiate between low load and high task load by assigning each data point <italic>x</italic><sub><italic>i</italic></sub> to a label <italic>y</italic><sub><italic>i</italic></sub>&#x02208;{<italic>Low, High</italic>}. Each of the two stimuli contains different levels of difficulty. The <italic>n</italic>-Back test includes two baselines and a total of six difficulty levels, while the <italic>k</italic>-Drive test has three baselines and three difficulty levels. In the first baseline in both scenarios, the subjects are not exposed to any stimuli. The monitor is turned off and the subject is asked to sit quietly in the chair so that a baseline measurement of the biosignals can be made. In the <italic>n</italic>-Back, for the second baseline measurement, the subject is now exposed to the same visual stimulus as during the actual test and has to randomly press buttons. They are instructed not to make any mental effort. This ensures that the same movement patterns are present as in the actual test, as well as the same lighting conditions. In the case of the first baseline, you could tell whether the subject is in the baseline or the actual test by the light-induced pupil dilation variation. This could potentially lead to a spurious correlation in the ML models and impair the reliability of the evaluation. In the driving scenario, there are 2 baselines in addition to the first one. In both baselines there is the visual stimulus as in the real driving task. In the first baseline, the subject had to watch the driving scene and randomly click on buttons. In the second baseline, they were asked to perform a behavioral pattern similar to the secondary task by looking at the tablet and clicking randomly on it to simulate creating a playlist.</p>
<p>For answering RQ2, we also need to evaluate the transfer between <italic>n</italic>-Back and <italic>k</italic>-Drive. However, this is not trivial because the stimuli differ and consequently the strength of the exposed loads. Based on the self-assessment results presented in <xref ref-type="fig" rid="F2">Figure 2</xref>, it is evident that the median ratings for levels 2 and 3 in both tests are above the midpoint of the rating scale. Conversely, level 1 has a low median rating. For the <italic>n</italic>-Back task, we define <italic>Low</italic> as {baseline<sub>2</sub>, level<sub>1</sub>}, and High as {level<sub>2</sub>, level<sub>3</sub>}. For <italic>k</italic>-Drive, we define <italic>Low</italic> as {baseline<sub>2</sub>, baseline<sub>3</sub>}, and <italic>High</italic> as {level<sub>2</sub>, level<sub>3</sub>}. Since we have two usable baselines available at <italic>k</italic>-Drive that closely resemble the actual test, we decided to use these for <italic>Low</italic> and not include level 1 to ensure a balanced class distribution.</p></sec></sec>
<sec>
<title>3.2 Models and fusion</title>
<p>Below we describe the machine learning approaches used in the experiments. In general, the methods can be divided into classical machine learning and deep learning methods. Furthermore, the models can be applied to different numbers of modalities. The experiments investigate unimodal models as well as multimodal models, for which we use different fusion approaches. These are also described in this section.</p>
<sec>
<title>3.2.1 Machine learning models</title>
<p>For our experiments, we use both classical machine learning (ML) and deep learning (DL) architectures. Classical machine learning refers to methods that rely on expertly crafted features, while deep neural networks, or deep learning, are capable of autonomously extracting features directly from raw data. As classical ML methods we use <italic>logistic regression</italic>, support vector machine (<italic>SVM</italic>) (Cortes and Vapnik, <xref ref-type="bibr" rid="B17">1995</xref>) and <italic>XGBoost</italic> (Chen and Guestrin, <xref ref-type="bibr" rid="B15">2016</xref>).</p>
<p>For the classification of time series data several deep learning architectures have been presented. These have often been evaluated on broad benchmark datasets such as UCR/UEA archive (Dau et al., <xref ref-type="bibr" rid="B18">2019</xref>). Since the domain of affective computing differs from many in the benchmark datasets, we do not limit our evaluation to the architecture that has performed best on this benchmark but examine a small selection. We use the Fully Convolutional Network (<italic>FCN</italic>) (Wang et al., <xref ref-type="bibr" rid="B47">2017</xref>), which is a simple architecture consisting of three convolutional layers with batch normalization. All layers have zero padding so that the length of the time series remains the same. Global average pooling is used to aggregate the features over the temporal dimension. As the FCN model is relatively simple, we additionally use a <italic>ResNet-1D</italic> adapted for time series data (Wang et al., <xref ref-type="bibr" rid="B47">2017</xref>). These two architectures contain global pooling, which may cause a loss in temporal patterns. Therefore, we extend the ResNet-1D by a sequence model that is applied on the latent features instead of global pooling. Since the length of the time series is not reduced by the architecture, we first use a max pooling with pooling size of 1 second and stride of 0.5 seconds to reduce the length. A Gated Recurrent Unit (GRU) (Cho et al., <xref ref-type="bibr" rid="B16">2014</xref>) is applied to this representation to aggregate the information over the temporal dimension. In the following, this architecture is referred to as <italic>ResNet1D-GRU</italic>. <xref ref-type="fig" rid="F3">Figure 3</xref> shows an overview of the used architectures.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Overview of the three deep learning architectures employed in our experiments: <bold>(A)</bold> FCN, <bold>(B)</bold> ResNet-1D, and <bold>(C)</bold> ResNet1D-GRU. For each layer, the figure details the number of channels, the activation function used, and the application of batch normalization (BN).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-06-1371181-g0003.tif"/>
</fig>

</sec>
<sec>
<title>3.2.2 Fusion methods</title>
<p>Multimodal approaches require the integration of information derived from individual modalities, and the choice of fusion method can significantly influence both accuracy and robustness. One simple approach that can be used on top of any classifier is <italic>late fusion</italic>. This involves training models for each modality independently and averaging the predictions from these models. Another simple approach is the concatenation of features of different modalities. For classical machine learning methods, expert features are concatenated prior to being fed into the classifier. In deep learning approaches, the learned latent features are concatenated. We refer to this fusion method as <italic>concat</italic> in our experiments In addition to these simple fusion methods, multi-modal gated units (Arevalo et al., <xref ref-type="bibr" rid="B5">2020</xref>) for deep learning methods are also being investigated, which we refer to as <italic>gated fusion</italic>. These units use an input dependent gating mechanism that assigns weights to individual modalities. This dynamic fusion may allow for a more robust fusion when modalities are unreliable in a distribution shift.</p></sec></sec>
<sec>
<title>3.3 Experimental setup</title>
<p>To conduct our experiments, we divide the data into training, validation and test sets using a subject-wise split, which ensures that no data points from the same subject appear in different subsets. For models that can monitor their overfitting behavior we use the validation set for early stopping. We use the model parameters from the epoch with the lowest validation loss. However, we allow the model to be trained until the end of the predefined epochs. Given the relatively small dataset, it is essential to use multiple splits to mitigate the possibility of an unfavorable split leading to biased evaluation. While a leave-one-subject-out approach is commonly used for subject-dependent data, it is not feasible for our experiments due to the computational complexity. Instead, we use a 4x4 nested cross-validation in the experiments. This approach helps us identify good hyperparameter settings and create a less biased estimate of the true error (Varma and Simon, <xref ref-type="bibr" rid="B46">2006</xref>). The test sets of four outer folds are used for calculating the final reported results. The four inner folds are used for the hyperparameter optimization (HPO) for each outer fold. For the hyperparameter search we perform 25 trials per inner fold. Finally, using the best hyperparameter setting the model is trained ten times on the inner fold, if the model does not have a deterministic inference. Consequently, up to 440 training runs are performed per model. We use the Tree-structured Parzen Estimator (Bergstra et al., <xref ref-type="bibr" rid="B11">2011</xref>) for the HPO using the library Optuna (Akiba et al., <xref ref-type="bibr" rid="B2">2019</xref>). <xref ref-type="table" rid="T2">Table 2</xref> shows the used hyperparameter search space. <xref ref-type="fig" rid="F4">Figure 4</xref> shows the overall experimental setup. For optimizing the deep learning models, we employ the ADAM optimizer (Kingma and Ba, <xref ref-type="bibr" rid="B29">2015</xref>). Note that all experiments are performed in Python. The deep learning models are implemented using PyTorch (Paszke et al., <xref ref-type="bibr" rid="B42">2019</xref>), while scitkit-learn (Pedregosa et al., <xref ref-type="bibr" rid="B43">2011</xref>) is used for the other classical machine learning methods.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Default hyper-parameters and random search grids for all algorithms.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Algorithm</bold></th>
<th valign="top" align="left"><bold>Hyper-parameter</bold></th>
<th valign="top" align="left"><bold>Search distribution</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">FCN, ResNet-1D, Resnet1D-GRU</td>
<td valign="top" align="left">Learning rate</td>
<td valign="top" align="left">RandFloat (0.0001, 0.01)</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Weight decay</td>
<td valign="top" align="left">RandFloat (0.0001, 0.03)</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Dropout</td>
<td valign="top" align="left">Choice ([0, 0.1, 0.2, 0.3, 0.5])</td>
</tr> <tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">Learning rate</td>
<td valign="top" align="left">RandFloat (0.01, 1)</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Number estimators</td>
<td valign="top" align="left">Choice ([100, 150, 200, 400])</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Max. depth</td>
<td valign="top" align="left">RandInt (4, 20)</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Subsample</td>
<td valign="top" align="left">RandFloat (0.7, 1)</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">l1</td>
<td valign="top" align="left">RandFloat (0, 1)</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">l2</td>
<td valign="top" align="left">RandFloat (0, 1)</td>
</tr> <tr>
<td valign="top" align="left">Logistic regression</td>
<td valign="top" align="left">C</td>
<td valign="top" align="left">2<sup><italic>Uniform</italic> (&#x02212;5, 4)</sup></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Iterations</td>
<td valign="top" align="left">Choice ([50, 100, 150, 200])</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Penality</td>
<td valign="top" align="left">Choice ([<italic>l</italic>1, <italic>l</italic>2])</td>
</tr> <tr>
<td valign="top" align="left">SVM</td>
<td valign="top" align="left">C</td>
<td valign="top" align="left">2<sup><italic>Uniform</italic> (&#x02212;5, 8)</sup></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Max. iterations</td>
<td valign="top" align="left">Choice ([50, 100, 150, 200, 250])</td>
</tr>
<tr>
<td/>
<td valign="top" align="left">Kernel</td>
<td valign="top" align="left">Choice ([<italic>linear, poly, rbf</italic>])</td>
</tr></tbody>
</table>
</table-wrap>

<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Training and evaluation setup using 4 &#x000D7; 4 nested cross-validation with an inner loop for optimal hyperparameter tuning. It is trained on <italic>n</italic>-Back and evaluated on <italic>n</italic>-Back and <italic>k</italic>-Drive test datasets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-06-1371181-g0004.tif"/>
</fig></sec>



<sec>
<title>3.4 Evaluation metrics</title>
<p>In this section, we describe the metrics used to evaluate classification performance and uncertainty estimates. Since our goal is to investigate robustness, we can look at the change in these performance metrics between <italic>n</italic>-Back and <italic>k</italic>-Drive. If the performance of a model decreases significantly, we can conclude that it lacks robustness with respect to this specific change in the data distribution. To evaluate the classification performance we use the F1-score because of a slight imbalance in the <italic>k</italic>-Drive dataset. For some analyses we also use the area under the receiver operating characteristic curve (AUROC). To assess the uncertainty estimation, we employ two metrics. Firstly, we utilize the Expected Calibration Error (ECE) (Guo et al., <xref ref-type="bibr" rid="B24">2017</xref>). This metric provides a measure of confidence calibration, essentially quantifying the disparity between predicted probabilities and observed outcomes. The ECE is calculated by dividing the predicted probabilities into bins and contrasting them with the actual accuracy within those bins. In turn, this error diminishes when the confidence and accuracy for a specific bin come into alignment. The ECE is calculated using <xref ref-type="disp-formula" rid="E1">Equation 1</xref>:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>E</mml:mi><mml:mi>C</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x000B7;</mml:mo><mml:mo>|</mml:mo><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:msub><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Here, <italic>M</italic> represents the number of bins, <italic>N</italic><sub><italic>i</italic></sub> is the number of samples in the <italic>i</italic>-th bin, <italic>N</italic> is the total number of samples, <italic>acc</italic><sub><italic>i</italic></sub> denotes the accuracy of the <italic>i</italic>-th bin, and <italic>conf</italic><sub><italic>i</italic></sub> signifies the confidence of the <italic>i</italic>-th bin. The ECE is reported on a scale of 0 to 1, with 0 being the optimal value.</p>
<p>However, sometimes it is not necessary for this to match exactly. In the case of a human-in-the-loop system, for example, where a subject can self-correct certain mispredictions, it would be important that the misclassifications just have a higher uncertainty than correctly classified data points. For evaluating this, rejection curves (Malinin, <xref ref-type="bibr" rid="B35">2019</xref>) can be used. The data points are sorted in descending order of uncertainty and sequentially replaced by ground truth labels. If the uncertainty correlates strongly with the misclassifications, then these misclassifications are quickly replaced with correct labels, causing the error to drop rapidly. If the uncertainties were absolutely uncorrelated with the misclassifications, then the error would fall linearly until the entire data set was replaced with ground truth labels. Whether the curve drops quickly or slowly can be calculated by the area under the curve. After normalizing this value with the random curve we get <italic>AR</italic><sub><italic>uns</italic></sub>. We contrast this curve with the normalized area under the ideal curve <italic>AR</italic><sub><italic>orc</italic></sub>, where the uncertainty perfectly correlates with the error. Our metric is the Rejection Ratio (RR) as shown in <xref ref-type="disp-formula" rid="E2">Equation 2</xref>:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>R</mml:mi><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>A</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mi>n</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>A</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>The metric&#x00027;s range is between &#x02013;1 and &#x0002B;1, with 1 indicating a perfect positive correlation between missclassification and uncertainty. If there is a negative correlation, meaning all high uncertainty data points are correctly classified, then the value approaches &#x02013;1.</p></sec></sec>
<sec sec-type="results" id="s4">
<title>4 Results</title>
<p>This section presents the results of the experiments. Section 4.1 discusses the results of evaluating the unimodal and multimodal models on the <italic>n</italic>-Back data, addressing RQ1. Section 4.2 focuses on the evaluation of these models on the drive data, allowing us to investigate the robustness of the models and address RQ2. For all questions, both the classification accuracy and the quality of the uncertainty estimates are considered.</p>
<sec>
<title>4.1 <italic>n</italic>-Back performance</title>
<p>In this section, we analyze the performance of models trained and evaluated on <italic>n</italic>-Back data. The classification performance for the modalities is detailed in <xref ref-type="table" rid="T3">Table 3</xref>. This table shows that eye tracking consistently achieves the highest F1-score across all models. Skin temperature, on the other hand, tends to yield the poorest classification results across most models. Notably, traditional machine learning methods generally outperform deep learning approaches, except in cases involving EMG data and skin temperature. Logistic regression, in particular, seems to deliver the best performance across most modalities.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>This table shows the F1-score for models trained and evaluated on <italic>n</italic>-Back.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>ECG</bold></th>
<th valign="top" align="left"><bold>EDA</bold></th>
<th valign="top" align="left"><bold>EMG</bold></th>
<th valign="top" align="left"><bold>EYE</bold></th>
<th valign="top" align="left"><bold>PPG</bold></th>
<th valign="top" align="left"><bold>RESP</bold></th>
<th valign="top" align="left"><bold>SKIN</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Logistic regression</td>
<td valign="top" align="left"><bold>0.70</bold>&#x000B1;0.06</td>
<td valign="top" align="left"><bold>0.69</bold>&#x000B1;0.05</td>
<td valign="top" align="left"><bold>0.67</bold>&#x000B1;0.01</td>
<td valign="top" align="left"><bold>0.85</bold>&#x000B1;0.04</td>
<td valign="top" align="left"><bold>0.66</bold>&#x000B1;0.03</td>
<td valign="top" align="left"><bold>0.65</bold>&#x000B1;0.03</td>
<td valign="top" align="left">0.59 &#x000B1; 0.05</td>
</tr> <tr>
<td valign="top" align="left">SVM</td>
<td valign="top" align="left">0.65 &#x000B1; 0.04</td>
<td valign="top" align="left">0.62 &#x000B1; 0.06</td>
<td valign="top" align="left">0.33 &#x000B1; 0.13</td>
<td valign="top" align="left">0.81 &#x000B1; 0.03</td>
<td valign="top" align="left">0.64 &#x000B1; 0.04</td>
<td valign="top" align="left">0.59 &#x000B1; 0.07</td>
<td valign="top" align="left">0.67 &#x000B1; 0.01</td>
</tr> <tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">0.68 &#x000B1; 0.03</td>
<td valign="top" align="left">0.63 &#x000B1; 0.05</td>
<td valign="top" align="left">0.56 &#x000B1; 0.07</td>
<td valign="top" align="left">0.83 &#x000B1; 0.04</td>
<td valign="top" align="left">0.64 &#x000B1; 0.03</td>
<td valign="top" align="left">0.61 &#x000B1; 0.03</td>
<td valign="top" align="left">0.68 &#x000B1; 0.03</td>
</tr> <tr>
<td valign="top" align="left">FCN</td>
<td valign="top" align="left">0.60 &#x000B1; 0.09</td>
<td valign="top" align="left">0.59 &#x000B1; 0.08</td>
<td valign="top" align="left">0.64 &#x000B1; 0.03</td>
<td valign="top" align="left"><bold>0.85</bold>&#x000B1;0.03</td>
<td valign="top" align="left">0.57 &#x000B1; 0.06</td>
<td valign="top" align="left">0.62 &#x000B1; 0.03</td>
<td valign="top" align="left">0.61 &#x000B1; 0.16</td>
</tr> <tr>
<td valign="top" align="left">ResNet1D-GRU</td>
<td valign="top" align="left">0.64 &#x000B1; 0.01</td>
<td valign="top" align="left">0.59 &#x000B1; 0.04</td>
<td valign="top" align="left">0.66 &#x000B1; 0.04</td>
<td valign="top" align="left"><bold>0.85</bold>&#x000B1;0.03</td>
<td valign="top" align="left">0.57 &#x000B1; 0.05</td>
<td valign="top" align="left">0.58 &#x000B1; 0.03</td>
<td valign="top" align="left"><bold>0.73</bold>&#x000B1;0.04</td>
</tr> <tr>
<td valign="top" align="left">ResNet1D</td>
<td valign="top" align="left">0.62 &#x000B1; 0.04</td>
<td valign="top" align="left">0.57 &#x000B1; 0.03</td>
<td valign="top" align="left">0.64 &#x000B1; 0.05</td>
<td valign="top" align="left"><bold>0.85</bold>&#x000B1;0.03</td>
<td valign="top" align="left">0.57 &#x000B1; 0.01</td>
<td valign="top" align="left">0.59 &#x000B1; 0.05</td>
<td valign="top" align="left">0.50 &#x000B1; 0.16</td>
</tr></tbody>
</table>
</table-wrap>


<p>The results of multimodal fusion are presented in <xref ref-type="table" rid="T4">Table 4</xref>. Early fusion, using logistic regression, is shown to provide the best classification performance. Feature fusion generally leads to better classification results compared to late fusion, a trend that is also evident in the calibration error. However, this is not the case for the rejection rate, which varies between different models. Fusion approaches tend to produce classification performance that is either equal to or slightly better than the best unimodal performance for each model. In particular, among deep learning methods, gated fusion shows a tendency toward better classification accuracy and rejection rate. The <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref> includes detailed tables for the expected calibration error and rejection scores for all modalities and models. Although the multimodal models did not achieve an overall improvement in the uncertainty estimation compared to the best performing unimodal model, they did show a slight improvement in the rejection ratio.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p><italic>n</italic>-Back fusion results.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>Fusion</bold></th>
<th valign="top" align="left"><bold>F1-Score (&#x02191;)</bold></th>
<th valign="top" align="left"><bold>Calibration error (&#x02193;)</bold></th>
<th valign="top" align="left"><bold>Rejection ratio (&#x02191;)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Logistic regression</td>
<td valign="top" align="left">Late</td>
<td valign="top" align="left">0.85 &#x000B1; 0.03</td>
<td valign="top" align="left">24.58 &#x000B1; 3.41</td>
<td valign="top" align="left">64.33 &#x000B1; 5.63</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Concat</td>
<td valign="top" align="left"><bold>0.86</bold>&#x000B1;0.02</td>
<td valign="top" align="left"><bold>7.92</bold>&#x000B1;2.01</td>
<td valign="top" align="left">58.83 &#x000B1; 20.61</td>
</tr> <tr>
<td valign="top" align="left">SVM</td>
<td valign="top" align="left">Late</td>
<td valign="top" align="left">0.79 &#x000B1; 0.03</td>
<td valign="top" align="left">22.92 &#x000B1; 3.32</td>
<td valign="top" align="left">50.17 &#x000B1; 8.18</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Concat</td>
<td valign="top" align="left">0.84 &#x000B1; 0.04</td>
<td valign="top" align="left">9.09 &#x000B1; 2.77</td>
<td valign="top" align="left">65.10 &#x000B1; 10.71</td>
</tr> <tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">Late</td>
<td valign="top" align="left">0.82 &#x000B1; 0.01</td>
<td valign="top" align="left">21.12 &#x000B1; 1.94</td>
<td valign="top" align="left">52.92 &#x000B1; 7.15</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Concat</td>
<td valign="top" align="left">0.83 &#x000B1; 0.03</td>
<td valign="top" align="left">11.37 &#x000B1; 2.17</td>
<td valign="top" align="left">56.20 &#x000B1; 5.38</td>
</tr> <tr>
<td valign="top" align="left">FCN</td>
<td valign="top" align="left">Late</td>
<td valign="top" align="left">0.80 &#x000B1; 0.02</td>
<td valign="top" align="left">21.31 &#x000B1; 1.85</td>
<td valign="top" align="left">52.29 &#x000B1; 5.33</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">GatedFusion</td>
<td valign="top" align="left">0.85 &#x000B1; 0.01</td>
<td valign="top" align="left">10.86 &#x000B1; 2.03</td>
<td valign="top" align="left"><bold>68.86</bold>&#x000B1;4.98</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Concat</td>
<td valign="top" align="left">0.84 &#x000B1; 0.02</td>
<td valign="top" align="left">11.05 &#x000B1; 2.01</td>
<td valign="top" align="left">63.60 &#x000B1; 7.96</td>
</tr> <tr>
<td valign="top" align="left">ResNet1D-GRU</td>
<td valign="top" align="left">Late</td>
<td valign="top" align="left">0.82 &#x000B1; 0.03</td>
<td valign="top" align="left">22.29 &#x000B1; 1.98</td>
<td valign="top" align="left">51.04 &#x000B1; 6.14</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">GatedFusion</td>
<td valign="top" align="left">0.85 &#x000B1; 0.02</td>
<td valign="top" align="left">11.00 &#x000B1; 1.59</td>
<td valign="top" align="left">60.88 &#x000B1; 2.59</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Concat</td>
<td valign="top" align="left">0.83 &#x000B1; 0.05</td>
<td valign="top" align="left">10.42 &#x000B1; 1.92</td>
<td valign="top" align="left">56.72 &#x000B1; 5.83</td>
</tr></tbody>
</table>
</table-wrap></sec>


<sec>
<title>4.2 <italic>k</italic>-Drive performance</title>
<p>In this section we present the results of training on the <italic>n</italic>-Back data and evaluation on the <italic>k</italic>-Drive dataset. <xref ref-type="table" rid="T5">Table 5</xref> shows the unimodal classification results of the <italic>k</italic>-Drive scenario. In <xref ref-type="fig" rid="F5">Figure 5</xref> we combine the results of <italic>n</italic>-Back and <italic>k</italic>-Drive to show how the performance of the unimodal models changes between the two scenarios. Notably, the eye tracker is no longer the most effective modality in this setting. Interestingly, the ECG and EMG modalities show improved performance compared to the <italic>n</italic>-Back scenario. However, there is no clear trend in the performance of the models. While classic machine learning models tend to outperform DL models in the <italic>n</italic>-Back, the results are now more ambiguous. For example, the ResNet1D-GRU model outperforms logistic regression.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>This table shows the F1-score for models trained on <italic>n</italic>-Back and evaluated on <italic>k</italic>-Drive.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>ECG</bold></th>
<th valign="top" align="left"><bold>EDA</bold></th>
<th valign="top" align="left"><bold>EMG</bold></th>
<th valign="top" align="left"><bold>EYE</bold></th>
<th valign="top" align="left"><bold>PPG</bold></th>
<th valign="top" align="left"><bold>RESP</bold></th>
<th valign="top" align="left"><bold>SKIN</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Logistic regression</td>
<td valign="top" align="left">0.81 &#x000B1; 0.02</td>
<td valign="top" align="left">0.69 &#x000B1; 0.03</td>
<td valign="top" align="left"><bold>0.86</bold>&#x000B1;0.00</td>
<td valign="top" align="left">0.71 &#x000B1; 0.04</td>
<td valign="top" align="left">0.70 &#x000B1; 0.03</td>
<td valign="top" align="left"><bold>0.85</bold>&#x000B1;0.05</td>
<td valign="top" align="left">0.65 &#x000B1; 0.07</td>
</tr> <tr>
<td valign="top" align="left">SVM</td>
<td valign="top" align="left">0.76 &#x000B1; 0.06</td>
<td valign="top" align="left">0.71 &#x000B1; 0.04</td>
<td valign="top" align="left">0.39 &#x000B1; 0.17</td>
<td valign="top" align="left">0.68 &#x000B1; 0.05</td>
<td valign="top" align="left">0.75 &#x000B1; 0.05</td>
<td valign="top" align="left">0.73 &#x000B1; 0.03</td>
<td valign="top" align="left"><bold>0.86</bold>&#x000B1;0.01</td>
</tr> <tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left"><bold>0.83</bold>&#x000B1;0.04</td>
<td valign="top" align="left">0.63 &#x000B1; 0.03</td>
<td valign="top" align="left">0.65 &#x000B1; 0.12</td>
<td valign="top" align="left">0.59 &#x000B1; 0.03</td>
<td valign="top" align="left">0.70 &#x000B1; 0.09</td>
<td valign="top" align="left">0.80 &#x000B1; 0.06</td>
<td valign="top" align="left">0.77 &#x000B1; 0.04</td>
</tr> <tr>
<td valign="top" align="left">FCN</td>
<td valign="top" align="left">0.78 &#x000B1; 0.07</td>
<td valign="top" align="left">0.67 &#x000B1; 0.09</td>
<td valign="top" align="left">0.83 &#x000B1; 0.04</td>
<td valign="top" align="left">0.73 &#x000B1; 0.03</td>
<td valign="top" align="left"><bold>0.78</bold>&#x000B1;0.06</td>
<td valign="top" align="left">0.76 &#x000B1; 0.05</td>
<td valign="top" align="left">0.73 &#x000B1; 0.19</td>
</tr> <tr>
<td valign="top" align="left">ResNet1D-GRU</td>
<td valign="top" align="left">0.81 &#x000B1; 0.02</td>
<td valign="top" align="left"><bold>0.72</bold>&#x000B1;0.05</td>
<td valign="top" align="left"><bold>0.86</bold>&#x000B1;0.03</td>
<td valign="top" align="left">0.77 &#x000B1; 0.02</td>
<td valign="top" align="left">0.71 &#x000B1; 0.04</td>
<td valign="top" align="left">0.69 &#x000B1; 0.02</td>
<td valign="top" align="left">0.83 &#x000B1; 0.06</td>
</tr> <tr>
<td valign="top" align="left">ResNet1D</td>
<td valign="top" align="left">0.79 &#x000B1; 0.05</td>
<td valign="top" align="left">0.67 &#x000B1; 0.09</td>
<td valign="top" align="left">0.83 &#x000B1; 0.07</td>
<td valign="top" align="left"><bold>0.79</bold>&#x000B1;0.03</td>
<td valign="top" align="left">0.76 &#x000B1; 0.01</td>
<td valign="top" align="left">0.71 &#x000B1; 0.06</td>
<td valign="top" align="left">0.62 &#x000B1; 0.22</td>
</tr></tbody>
</table>
</table-wrap>

<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>F1-score with standard deviation of unimodal models trained on the <italic>n</italic>-Back data and evaluated on <italic>n</italic>-Back (x-axis) and <italic>k</italic>-Drive (y-axis). Points above the bisector show better performance on the shifted dataset and performance below the bisector shows worse performance.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-06-1371181-g0005.tif"/>
</fig>



<p>Next, we examine the results of multimodal approaches in <xref ref-type="table" rid="T6">Table 6</xref>. <xref ref-type="fig" rid="F6">Figure 6</xref> shows the comparison of the in-distribution F1-Score with the performance on the <italic>k</italic>-Drive dataset. In contrast to the in-distribution results, where early fusion outperforms late fusion, now the early fusion classification is inferior. This could be due to the changed importance of the modalities. For example, the reduced ability of the eye tracker to discriminate between low and high task load in <italic>k</italic>-Drive, based on <italic>n</italic>-Back training, affects early fusion where eye tracker features are critical. In late fusion, each modality contributes equally to the final prediction, mitigating this problem. For DL methods, this phenomenon is less pronounced, with late fusion slightly underperforming compared to intermediate fusion. Consequently, in FCN models, intermediate fusion achieves the best overall classification performance. It does not deviate from the predictive performance in the <italic>n</italic>-Back test, which makes it particularly robust. The <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref> includes detailed tables on calibration errors and rejection ratios for the models and modalities. It is important to note that multimodal fusion improved the rejection ratio compared to the best unimodal models within a model category.</p>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p><italic>k</italic>-Drive fusion results.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>Fusion</bold></th>
<th valign="top" align="left"><bold>F1-Score (&#x02191;)</bold></th>
<th valign="top" align="left"><bold>Calibration error (&#x02193;)</bold></th>
<th valign="top" align="left"><bold>Rejection ratio (&#x02191;)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Logistic regression</td>
<td valign="top" align="left">Late</td>
<td valign="top" align="left">0.82 &#x000B1; 0.03</td>
<td valign="top" align="left">29.25 &#x000B1; 1.99</td>
<td valign="top" align="left">52.57 &#x000B1; 11.33</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Concat</td>
<td valign="top" align="left">0.76 &#x000B1; 0.08</td>
<td valign="top" align="left">24.94 &#x000B1; 6.23</td>
<td valign="top" align="left">30.88 &#x000B1; 15.07</td>
</tr> <tr>
<td valign="top" align="left">SVM</td>
<td valign="top" align="left">Late</td>
<td valign="top" align="left">0.82 &#x000B1; 0.01</td>
<td valign="top" align="left">24.22 &#x000B1; 1.50</td>
<td valign="top" align="left">52.89 &#x000B1; 4.83</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Concat</td>
<td valign="top" align="left">0.74 &#x000B1; 0.07</td>
<td valign="top" align="left">25.49 &#x000B1; 6.03</td>
<td valign="top" align="left">36.50 &#x000B1; 6.04</td>
</tr> <tr>
<td valign="top" align="left">XGBoost</td>
<td valign="top" align="left">Late</td>
<td valign="top" align="left">0.74 &#x000B1; 0.05</td>
<td valign="top" align="left">29.36 &#x000B1; 2.13</td>
<td valign="top" align="left">45.33 &#x000B1; 8.31</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Concat</td>
<td valign="top" align="left">0.65 &#x000B1; 0.09</td>
<td valign="top" align="left">28.90 &#x000B1; 4.72</td>
<td valign="top" align="left">41.33 &#x000B1; 7.81</td>
</tr> <tr>
<td valign="top" align="left">FCN</td>
<td valign="top" align="left">Late</td>
<td valign="top" align="left">0.80 &#x000B1; 0.07</td>
<td valign="top" align="left">26.45 &#x000B1; 2.79</td>
<td valign="top" align="left"><bold>60.21</bold>&#x000B1;10.06</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">GatedFusion</td>
<td valign="top" align="left">0.80 &#x000B1; 0.03</td>
<td valign="top" align="left">21.34 &#x000B1; 1.85</td>
<td valign="top" align="left">40.76 &#x000B1; 9.13</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Concat</td>
<td valign="top" align="left"><bold>0.84</bold>&#x000B1;0.05</td>
<td valign="top" align="left"><bold>20.83</bold>&#x000B1;3.24</td>
<td valign="top" align="left">37.70 &#x000B1; 16.76</td>
</tr> <tr>
<td valign="top" align="left">ResNet1D-GRU</td>
<td valign="top" align="left">Late</td>
<td valign="top" align="left">0.77 &#x000B1; 0.03</td>
<td valign="top" align="left">24.69 &#x000B1; 2.25</td>
<td valign="top" align="left">51.45 &#x000B1; 6.92</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">GatedFusion</td>
<td valign="top" align="left">0.78 &#x000B1; 0.04</td>
<td valign="top" align="left">22.45 &#x000B1; 3.76</td>
<td valign="top" align="left">36.68 &#x000B1; 10.93</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Concat</td>
<td valign="top" align="left">0.75 &#x000B1; 0.04</td>
<td valign="top" align="left">23.24 &#x000B1; 4.43</td>
<td valign="top" align="left">35.92 &#x000B1; 18.20</td>
</tr></tbody>
</table>
</table-wrap>

<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>F1-score with standard deviation of fusion models trained on the <italic>n</italic>-Back data and evaluated on <italic>n</italic>-Back (x-axis) and <italic>k</italic>-Drive (y-axis). Points above the bisector show better performance on the shifted dataset and performance below the bisector shows worse performance.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fcomp-06-1371181-g0006.tif"/>
</fig>



<p><xref ref-type="table" rid="T7">Table 7</xref> shows the results of an ablation study regarding the impact of different normalization approaches on the logistic regression fusion model with concatenated features vectors. In the <italic>n</italic>-Back performance analysis, we observe an incremental enhancement in logistic regression accuracy. The performance is least effective with no normalization, improves with global normalization, and reaches its best results using subject normalization. This clearly demonstrates the relative superiority of subject normalization over global normalization and no normalization in in-domain contexts. The scenario changes markedly in the <italic>k</italic>-Drive context. In this case, global normalization performed significantly worse, in contrast to the no normalization and subject normalization scenarios. Subject normalization continues to show superior performance. Interestingly, however, it was found that no normalization outperformed the global normalization. To investigate this further, we used the AUROC metric. This choice was made because AUROC indicates the separability of predictions, thus revealing whether adjusting the threshold might improve performance. In these evaluations global normalization outperforms no normalization.</p>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>Ablation study of normalization techniques on performance in different scenarios for early fusion logistic regression.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Data</bold></th>
<th valign="top" align="left"><bold>Normalization</bold></th>
<th valign="top" align="left"><bold>F1-Score (&#x02191;)</bold></th>
<th valign="top" align="left"><bold>AUROC (&#x02191;)</bold></th>
<th valign="top" align="left"><bold>Calibration error (&#x02193;)</bold></th>
<th valign="top" align="left"><bold>Rejection ratio (&#x02191;)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><italic>n</italic>-Back</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">0.58 &#x000B1; 0.05</td>
<td valign="top" align="left">0.63 &#x000B1; 0.03</td>
<td valign="top" align="left">0.09 &#x000B1; 0.02</td>
<td valign="top" align="left">0.07 &#x000B1; 0.13</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Global</td>
<td valign="top" align="left">0.75 &#x000B1; 0.03</td>
<td valign="top" align="left">0.83 &#x000B1; 0.03</td>
<td valign="top" align="left">0.15 &#x000B1; 0.05</td>
<td valign="top" align="left">0.38 &#x000B1; 0.08</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Subject</td>
<td valign="top" align="left"><bold>0.86</bold>&#x000B1;0.02</td>
<td valign="top" align="left"><bold>0.93</bold>&#x000B1;0.03</td>
<td valign="top" align="left"><bold>0.07</bold>&#x000B1;0.02</td>
<td valign="top" align="left"><bold>0.59</bold>&#x000B1;0.21</td>
</tr> <tr>
<td valign="top" align="left"><italic>k</italic>-Drive</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">0.65 &#x000B1; 0.08</td>
<td valign="top" align="left">0.63 &#x000B1; 0.04</td>
<td valign="top" align="left"><bold>0.25</bold>&#x000B1;0.05</td>
<td valign="top" align="left">0.15 &#x000B1; 0.07</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Global</td>
<td valign="top" align="left">0.08 &#x000B1; 0.12</td>
<td valign="top" align="left">0.67 &#x000B1; 0.06</td>
<td valign="top" align="left">0.68 &#x000B1; 0.06</td>
<td valign="top" align="left">0.25 &#x000B1; 0.16</td>
</tr>
 <tr>
<td/>
<td valign="top" align="left">Subject</td>
<td valign="top" align="left"><bold>0.76</bold>&#x000B1;0.08</td>
<td valign="top" align="left"><bold>0.80</bold>&#x000B1;0.06</td>
<td valign="top" align="left"><bold>0.25</bold>&#x000B1;0.06</td>
<td valign="top" align="left"><bold>0.31</bold>&#x000B1;0.15</td>
</tr></tbody>
</table>
</table-wrap>

</sec></sec>
<sec sec-type="discussion" id="s5">
<title>5 Discussion</title>
<p>Our investigation into various cognitive load estimation models both trained and evaluated on <italic>n</italic>-Back has yielded interesting insights. Notably, the eye tracker-based model exhibited the highest classification performance among all the models examined. This aligns with previous studies where eye tracking outperformed other modalities (Aygun et al., <xref ref-type="bibr" rid="B6">2022</xref>; Oppelt et al., <xref ref-type="bibr" rid="B37">2023</xref>). It is important to emphasize that the efficacy of eye tracking can be context-dependent. Factors like lighting conditions can significantly affect pupil size, a crucial indicator of cognitive overload.</p>
<p>When considering the models, it becomes evident that classic machine learning models tend to outperform deep learning models for in-distribution data, especially when applied to more complex signals like electrocardiography (ECG). A potential reason may be the limited dataset size, resulting in fewer generalizable features being learned. This is consistent with the results of Aygun et al. (<xref ref-type="bibr" rid="B6">2022</xref>), where DL models underperformed compared to classic ML models.</p>
<p>In addition to unimodal models, we also trained models employing various fusion techniques. However, none of these models showed an improvement compared to the top performing unimodal models. Other publications, however, demonstrate that multimodal combinations can indeed enhance classification performance. For instance, Oppelt et al. (<xref ref-type="bibr" rid="B37">2023</xref>) illustrated that integrating eye tracking data with biosignals can further enhance the performance of the eye tracker. The reasons for this disparate behavior can be diverse, e.g., differences in hyperparameter optimization. Nevertheless, within the realm of deep learning models, it has been repeatedly demonstrated that achieving superior performance with fusion models is non-trivial compared to the best unimodal performance (Wilson et al., <xref ref-type="bibr" rid="B48">2021</xref>). Employing explainability methods might have provided insights into how different features influence classification performance across the two datasets. Such analysis could reveal not just the consistency of feature behavior across varied scenarios, but also show the specific contributions of each modality or feature within the intermediate fusion models. By combining explainable AI with robustness analysis, as demonstrated in this study, future work can focus on finding multi-faceted explanations (Longo et al., <xref ref-type="bibr" rid="B33">2024</xref>) that enhance system trustworthiness. This approach provides insights into both feature interactions for classification and the reliability of those interactions under changing scenarios.</p>
<p>The second research question focuses on the influence of different modeling decisions on performance in a different scenario. Concerning unimodal performance, we observe that the eye tracker exhibits poorer performance across all models compared to the <italic>n</italic>-Back scenario. This scenario lacks consistent lighting conditions, which could be the reason for to diminished performance. Interestingly, all other modalities perform better than in the <italic>n</italic>-Back. One possible explanation is that this scenario induces higher levels of cognitive overload, making the data more separable based on physiological features. It is also possible that another affective state, such as enjoyment of driving, has been induced, which has similar physiological characteristics to cognitive overload. Another explanation could be that variations in physiological modalities arise not primarily from a mental state, but rather from slightly increased movement due to switching between a tablet and a steering wheel. Regarding fusion methods, we note that late fusion tends to deliver superior classification performance under a distribution shift. One potential explanation is that the feature fusion models might overly focus on a specific subset of features, particularly those derived from the eye tracker. While fusion does not outperform the best unimodal performance, late fusion in particular provides a valuable compromise between in-distribution performance and robustness, as it provides good results in both scenarios. Choosing only the best unimodal model, i.e., an eye tracker model, would have significantly degraded performance. Another important finding of our work is that the ECE decreases for all models between the <italic>n</italic>-Back to the <italic>k</italic>-Drive scenario, even if the classification performance remains stable. While the rejection ratio also exhibits a decrease across models, the decline is notably less severe in the late fusion approach. This further emphasizes late fusion as an interesting fusion method to create robust and reliable models.</p>
<p>Several limitations of our study should be acknowledged. Firstly, a <italic>post-hoc</italic> calibration step for the models might have impacted their performance. Incorporating such a step could potentially lead to improved calibration scores. Secondly, our study was based on a single dataset, which limits the generalizability of our findings. Finally, it is important to acknowledge the presence of possible confounding factors within the dataset that could have influenced our results.</p></sec>
<sec id="s6">
<title>6 Conclusion and future work</title>
<p>In this paper, we contribute to the understanding of the capabilities and limitations of modeling task load in real world scenarios, especially considering the common occurrence of data shifts. To this end, we first analyzed various machine learning models and fusion approaches with in-distribution data in order to subsequently observe the influence of the distribution shift on the performance. In our investigation, we came to the conclusion that, on the one hand, late fusion is a good compromise to provide good classification performance and uncertainty estimation for both in-distribution and out-of-distribution data.</p>
<p>Future research should aim to investigate and improve the robustness of multimodal cognitive load estimation by exploring diverse datasets and examining different types of shifts. The initial study was conducted using the same hardware across scenarios. Investigating the effects of more significant shifts, such as those introduced by varying wearable devices, could prove beneficial. These shifts affect not only the stimulus but also the quality of the signal, providing insights into developing applications that perform reliably under real-world conditions. Furthermore, expanding our research to include new modalities such as EEG and utilizing other machine learning approaches could provide valuable insights. Another important aspect for future work is to investigate how these models can be adapted to new scenarios in a better way. This could be achieved by domain adaptation techniques that address how models can be effectively adapted to new domains in a supervised or unsupervised manner. Future research could also expand our robustness experiments to incorporate more precise indicators of cognitive load as described in Longo et al. (<xref ref-type="bibr" rid="B34">2022</xref>). For example, this can be done integrating a diverse range of factors such as effort and motivation through self-assessments. By combining these with measures like the performance, researchers could evaluate the robustness using a more precise annotation for cognitive load.</p></sec>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://adabase-dataset.github.io/">https://adabase-dataset.github.io/</ext-link>.</p></sec>
<sec sec-type="ethics-statement" id="s8">
<title>Ethics statement</title>
<p>Ethical approval was not required for the study involving humans in accordance with the local legislation and institutional requirements. Written informed consent to participate in this study was not required from the participants or the participants&#x00027; legal guardians/next of kin in accordance with the national legislation and the institutional requirements.</p></sec>
<sec sec-type="author-contributions" id="s9">
<title>Author contributions</title>
<p>AF: Conceptualization, Formal analysis, Investigation, Methodology, Software, Supervision, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. JD: Conceptualization, Formal analysis, Investigation, Methodology, Software, Validation, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. NL-R: Funding acquisition, Resources, Writing &#x02013; review &#x00026; editing. NH: Project administration, Resources, Writing &#x02013; review &#x00026; editing. MO: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Validation, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing.</p></sec>
</body>
<back>
<sec sec-type="funding-information" id="s10">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. The authors acknowledge the partial funding by the EU TEF-Health project which is part of the Digital Europe Programme of the EU (DIGITAL-2022-CLOUD-AI-02-TEFHEALTH) under grant agreement no. 101100700 and the Berlin Senate and the partial support by the Bavarian Ministry of Economic Affairs, Regional Development and Energy through the Center for Analytics-Data-Applications (ADA-Center) within the framework of &#x0201C;BAYERN DIGITAL II&#x0201D; (20-3410-2-9-8).</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fcomp.2024.1371181/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fcomp.2024.1371181/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/></sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abrantes</surname> <given-names>A.</given-names></name> <name><surname>Comitz</surname> <given-names>E.</given-names></name> <name><surname>Mosaly</surname> <given-names>P.</given-names></name> <name><surname>Mazur</surname> <given-names>L.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Classification of eeg features for prediction of working memory load,&#x0201D;</article-title> in <source>Advances in The Human Side of Service Engineering</source>, eds. T. Z. Ahram, and W. Karwowski (Cham: Springer International Publishing), <fpage>115</fpage>&#x02013;<lpage>126</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-41947-3_12</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Akiba</surname> <given-names>T.</given-names></name> <name><surname>Sano</surname> <given-names>S.</given-names></name> <name><surname>Yanase</surname> <given-names>T.</given-names></name> <name><surname>Ohta</surname> <given-names>T.</given-names></name> <name><surname>Koyama</surname> <given-names>M.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Optuna: a next-generation hyperparameter optimization framework,&#x0201D;</article-title> in <source>Proceedings of the 25rd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</source>. <pub-id pub-id-type="doi">10.1145/3292500.3330701</pub-id></citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Albuquerque</surname> <given-names>I.</given-names></name> <name><surname>Tiwari</surname> <given-names>A.</given-names></name> <name><surname>Parent</surname> <given-names>M.</given-names></name> <name><surname>Cassani</surname> <given-names>R.</given-names></name> <name><surname>Gagnon</surname> <given-names>J.-F.</given-names></name> <name><surname>Lafond</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>WAUC: a multi-modal database for mental workload assessment under physical activity</article-title>. <source>Front. Neurosci</source>. <volume>14</volume>:<fpage>549524</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2020.549524</pub-id><pub-id pub-id-type="pmid">33335465</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Antonenko</surname> <given-names>P. D.</given-names></name> <name><surname>Paas</surname> <given-names>F.</given-names></name> <name><surname>Grabner</surname> <given-names>R. H.</given-names></name> <name><surname>van Gog</surname> <given-names>T.</given-names></name></person-group> (<year>2010</year>). <article-title>Using electroencephalography to measure cognitive load</article-title>. <source>Educ. Psychol. Rev</source>. <volume>22</volume>, <fpage>425</fpage>&#x02013;<lpage>438</lpage>. <pub-id pub-id-type="doi">10.1007/s10648-010-9130-y</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Arevalo</surname> <given-names>J.</given-names></name> <name><surname>Solorio</surname> <given-names>T.</given-names></name> <name><surname>Montes-y Gomez</surname> <given-names>M.</given-names></name> <name><surname>Gonz&#x000E1;lez</surname> <given-names>F. A.</given-names></name></person-group> (<year>2020</year>). <article-title>Gated multimodal networks</article-title>. <source>Neural Comput. Applic</source>. <volume>32</volume>, <fpage>10209</fpage>&#x02013;<lpage>10228</lpage>. <pub-id pub-id-type="doi">10.1007/s00521-019-04559-1</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aygun</surname> <given-names>A.</given-names></name> <name><surname>Nguyen</surname> <given-names>T.</given-names></name> <name><surname>Haga</surname> <given-names>Z.</given-names></name> <name><surname>Aeron</surname> <given-names>S.</given-names></name> <name><surname>Scheutz</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>Investigating methods for cognitive workload estimation for assistive robots</article-title>. <source>Sensors</source> <volume>22</volume>:<fpage>6834</fpage>. <pub-id pub-id-type="doi">10.3390/s22186834</pub-id><pub-id pub-id-type="pmid">36146189</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ayres</surname> <given-names>P.</given-names></name> <name><surname>Lee</surname> <given-names>J. Y.</given-names></name> <name><surname>Paas</surname> <given-names>F.</given-names></name> <name><surname>van Merri&#x000EB;nboer</surname> <given-names>J. J. G.</given-names></name></person-group> (<year>2021</year>). <article-title>The validity of physiological measures to identify differences in intrinsic cognitive load</article-title>. <source>Front. Psychol</source>. <volume>12</volume>:<fpage>702538</fpage>. <pub-id pub-id-type="doi">10.3389/fpsyg.2021.702538</pub-id><pub-id pub-id-type="pmid">34566780</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Beatty</surname> <given-names>J.</given-names></name></person-group> (<year>1982</year>). <article-title>Task-evoked pupillary responses, processing load, and the structure of processing resources</article-title>. <source>Psychol. Bull</source>. <volume>91</volume>:<fpage>276</fpage>. <pub-id pub-id-type="doi">10.1037//0033-2909.91.2.276</pub-id><pub-id pub-id-type="pmid">7071262</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Beatty</surname> <given-names>J.</given-names></name> <name><surname>Lucero-Wagoner</surname> <given-names>B.</given-names></name></person-group> (<year>2000</year>). <source>The Pupillary System</source>. <publisher-loc>Cambridge</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name>, <fpage>142</fpage>&#x02013;<lpage>162</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Beh</surname> <given-names>W.-K.</given-names></name> <name><surname>Wu</surname> <given-names>Y.-H.</given-names></name> <name><surname>Wu</surname> <given-names>A.-Y. A.</given-names></name></person-group> (<year>2021</year>). <article-title>Maus: a dataset for mental workload assessment on n-back task using wearable sensor</article-title>. <source>arXiv preprint arXiv:2111.02561</source>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bergstra</surname> <given-names>J.</given-names></name> <name><surname>Bardenet</surname> <given-names>R.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>K&#x000E9;gl</surname> <given-names>B.</given-names></name></person-group> (<year>2011</year>). <article-title>&#x0201C;Algorithms for hyper-parameter optimization,&#x0201D;</article-title> in <source>NIPS</source>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Booth</surname> <given-names>B. M.</given-names></name> <name><surname>Mundnich</surname> <given-names>K.</given-names></name> <name><surname>Narayanan</surname> <given-names>S. S.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;A novel method for human bias correction of continuous- time annotations,&#x0201D;</article-title> in <source>2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</source>, 3091&#x02013;3095. <pub-id pub-id-type="doi">10.1109/ICASSP.2018.8461645</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bradley</surname> <given-names>M. M.</given-names></name> <name><surname>Miccoli</surname> <given-names>L.</given-names></name> <name><surname>Escrig</surname> <given-names>M. A.</given-names></name> <name><surname>Lang</surname> <given-names>P. J.</given-names></name></person-group> (<year>2008</year>). <article-title>The pupil as a measure of emotional arousal and autonomic activation</article-title>. <source>Psychophysiology</source> <volume>45</volume>, <fpage>602</fpage>&#x02013;<lpage>607</lpage>. <pub-id pub-id-type="doi">10.1111/j.1469-8986.2008.00654.x</pub-id><pub-id pub-id-type="pmid">18282202</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cain</surname> <given-names>B.</given-names></name></person-group> (<year>2007</year>). <source>A review of the mental workload literature</source>. DTIC Document.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>T.</given-names></name> <name><surname>Guestrin</surname> <given-names>C.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Xgboost: a scalable tree boosting system,&#x0201D;</article-title> in <source>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</source>, 785&#x02013;794. <pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cho</surname> <given-names>K.</given-names></name> <name><surname>van Merri&#x000EB;nboer</surname> <given-names>B.</given-names></name> <name><surname>Gulcehre</surname> <given-names>C.</given-names></name> <name><surname>Bahdanau</surname> <given-names>D.</given-names></name> <name><surname>Bougares</surname> <given-names>F.</given-names></name> <name><surname>Schwenk</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>&#x0201C;Learning phrase representations using RNN encoder-decoder for statistical machine translation,&#x0201D;</article-title> in <source>Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP)</source>, eds. A. Moschitti, B. Pang, and W. Daelemans (Doha, Qatar: Association for Computational Linguistics), <fpage>1724</fpage>&#x02013;<lpage>1734</lpage>. <pub-id pub-id-type="doi">10.3115/v1/D14-1179</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cortes</surname> <given-names>C.</given-names></name> <name><surname>Vapnik</surname> <given-names>V.</given-names></name></person-group> (<year>1995</year>). <article-title>Support-vector networks</article-title>. <source>Mach. Lear</source>. <volume>20</volume>, <fpage>273</fpage>&#x02013;<lpage>297</lpage>. <pub-id pub-id-type="doi">10.1007/BF00994018</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dau</surname> <given-names>H. A.</given-names></name> <name><surname>Bagnall</surname> <given-names>A.</given-names></name> <name><surname>Kamgar</surname> <given-names>K.</given-names></name> <name><surname>Yeh</surname> <given-names>C.-C. M.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Gharghabi</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>The UCR time series archive</article-title>. <source>IEEE/CAA J. Autom. Sinica</source> <volume>6</volume>, <fpage>1293</fpage>&#x02013;<lpage>1305</lpage>. <pub-id pub-id-type="doi">10.1109/JAS.2019.1911747</pub-id></citation>
</ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dolmans</surname> <given-names>T. C.</given-names></name> <name><surname>Poel</surname> <given-names>M.</given-names></name> <name><surname>van &#x00027;t Klooster</surname> <given-names>J.-W. J. R.</given-names></name> <name><surname>Veldkamp</surname> <given-names>B. P.</given-names></name></person-group> (<year>2021</year>). <article-title>Perceived mental workload classification using intermediate fusion multimodal deep learning</article-title>. <source>Front. Hum. Neurosci</source>. <volume>14</volume>:<fpage>609096</fpage>. <pub-id pub-id-type="doi">10.3389/fnhum.2020.609096</pub-id><pub-id pub-id-type="pmid">33505259</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Freiesleben</surname> <given-names>T.</given-names></name> <name><surname>Grote</surname> <given-names>T.</given-names></name></person-group> (<year>2023</year>). <article-title>Beyond generalization: a theory of robustness in machine learning</article-title>. <source>Synthese</source> <volume>202</volume>:<fpage>109</fpage>. <pub-id pub-id-type="doi">10.1007/s11229-023-04334-9</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fridman</surname> <given-names>A.</given-names></name> <name><surname>Reimer</surname> <given-names>B.</given-names></name> <name><surname>Mehler</surname> <given-names>B.</given-names></name> <name><surname>Freeman</surname> <given-names>W. T.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Cognitive load estimation in the wild,&#x0201D;</article-title> in <source>Proceedings of the 2018 CHI Conference on Human Factors in Computing Systems</source>, 1&#x02013;9. <pub-id pub-id-type="doi">10.1145/3173574.3174226</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gjoreski</surname> <given-names>M.</given-names></name> <name><surname>Gams</surname> <given-names>M. Z.</given-names></name> <name><surname>Lustrek</surname> <given-names>M.</given-names></name> <name><surname>Genc</surname> <given-names>P.</given-names></name> <name><surname>Garbas</surname> <given-names>J.-U.</given-names></name> <name><surname>Hassan</surname> <given-names>T.</given-names></name></person-group> (<year>2020a</year>). <article-title>Machine learning and end-to-end deep learning for monitoring driver distractions from physiological and visual signals</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>70590</fpage>&#x02013;<lpage>70603</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2020.2986810</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gjoreski</surname> <given-names>M.</given-names></name> <name><surname>Kolenik</surname> <given-names>T.</given-names></name> <name><surname>Knez</surname> <given-names>T.</given-names></name> <name><surname>Lu&#x00161;trek</surname> <given-names>M.</given-names></name> <name><surname>Gams</surname> <given-names>M.</given-names></name> <name><surname>Gjoreski</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2020b</year>). <article-title>Datasets for cognitive load inference using wearable sensors and psychological traits</article-title>. <source>Appl. Sci</source>. <volume>10</volume>:<fpage>3843</fpage>. <pub-id pub-id-type="doi">10.3390/app10113843</pub-id></citation>
</ref>
<ref id="B24">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>C.</given-names></name> <name><surname>Pleiss</surname> <given-names>G.</given-names></name> <name><surname>Sun</surname> <given-names>Y.</given-names></name> <name><surname>Weinberger</surname> <given-names>K. Q.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;On calibration of modern neural networks,&#x0201D;</article-title> in <source>Proceedings of the 34th International Conference on Machine Learning, ICML&#x00027;17</source> (<publisher-loc>JMLR.org</publisher-loc>), <fpage>1321</fpage>&#x02013;<lpage>1330</lpage>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Haapalainen</surname> <given-names>E.</given-names></name> <name><surname>Kim</surname> <given-names>S.</given-names></name> <name><surname>Forlizzi</surname> <given-names>J.</given-names></name> <name><surname>Dey</surname> <given-names>A. K.</given-names></name></person-group> (<year>2010</year>). <article-title>&#x0201C;Psycho-physiological measures for assessing cognitive load,&#x0201D;</article-title> in <source>Proceedings of the 12th ACM International Conference on Ubiquitous Computing</source>. <pub-id pub-id-type="doi">10.1145/1864349.1864395</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hart</surname> <given-names>S. G.</given-names></name> <name><surname>Staveland</surname> <given-names>L. E.</given-names></name></person-group> (<year>1988</year>). <article-title>&#x0201C;Development of NASA-TLX (task load index): results of empirical and theoretical research,&#x0201D;</article-title> in <source>Human Mental Workload</source>, eds. P. A. Hancock, and N. Meshkati (North-Holland: Elsevier), <fpage>139</fpage>&#x02013;<lpage>183</lpage>. <pub-id pub-id-type="doi">10.1016/S0166-4115(08)62386-9</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jaeggi</surname> <given-names>S. M.</given-names></name> <name><surname>Seewer</surname> <given-names>R.</given-names></name> <name><surname>Nirkko</surname> <given-names>A. C.</given-names></name> <name><surname>Eckstein</surname> <given-names>D.</given-names></name> <name><surname>Schroth</surname> <given-names>G.</given-names></name> <name><surname>Groner</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2003</year>). <article-title>Does excessive memory load attenuate activation in the prefrontal cortex? Load-dependent processing in single and dual tasks: functional magnetic resonance imaging study</article-title>. <source>NeuroImage</source> <volume>19</volume>, <fpage>210</fpage>&#x02013;<lpage>225</lpage>. <pub-id pub-id-type="doi">10.1016/S1053-8119(03)00098-3</pub-id><pub-id pub-id-type="pmid">12814572</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kesed&#x0017E;i&#x00107;</surname> <given-names>I.</given-names></name> <name><surname>&#x00160;arlija</surname> <given-names>M.</given-names></name> <name><surname>Bo&#x0017E;ek</surname> <given-names>J.</given-names></name> <name><surname>Popovi&#x00107;</surname> <given-names>S.</given-names></name> <name><surname>&#x00106;osi&#x00107;</surname> <given-names>K.</given-names></name></person-group> (<year>2021</year>). <article-title>Classification of cognitive load based on neurophysiological features from functional near-infrared spectroscopy and electrocardiography signals on n-back task</article-title>. <source>IEEE Sensors J</source>. <volume>21</volume>, <fpage>14131</fpage>&#x02013;<lpage>14140</lpage>. <pub-id pub-id-type="doi">10.1109/JSEN.2020.3038032</pub-id></citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kingma</surname> <given-names>D. P.</given-names></name> <name><surname>Ba</surname> <given-names>J.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Adam: a method for stochastic optimization,&#x0201D;</article-title> in <source>3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, 2015, Conference Track Proceedings</source>.</citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kirchner</surname> <given-names>W. K.</given-names></name></person-group> (<year>1958</year>). <article-title>Age differences in short-term retention of rapidly changing information</article-title>. <source>J. Exper. Psychol</source>. 55, 352. <pub-id pub-id-type="doi">10.1037/h0043688</pub-id><pub-id pub-id-type="pmid">13539317</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kumar</surname> <given-names>S.</given-names></name> <name><surname>He</surname> <given-names>D.</given-names></name> <name><surname>Qiao</surname> <given-names>G.</given-names></name> <name><surname>Donmez</surname> <given-names>B.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Classification of driver cognitive load based on physiological data: exploring recurrent neural networks,&#x0201D;</article-title> in <source>2022 International Conference on Advanced Robotics and Mechatronics (ICARM)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>19</fpage>&#x02013;<lpage>24</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Longo</surname> <given-names>L.</given-names></name></person-group> (<year>2022</year>). <article-title>Modeling cognitive load as a self-supervised brain rate with electroencephalography and deep learning</article-title>. <source>Brain Sci</source>. <volume>12</volume>:<fpage>1416</fpage>. <pub-id pub-id-type="doi">10.3390/brainsci12101416</pub-id><pub-id pub-id-type="pmid">36291349</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Longo</surname> <given-names>L.</given-names></name> <name><surname>Brcic</surname> <given-names>M.</given-names></name> <name><surname>Cabitza</surname> <given-names>F.</given-names></name> <name><surname>Choi</surname> <given-names>J.</given-names></name> <name><surname>Confalonieri</surname> <given-names>R.</given-names></name> <name><surname>Ser</surname> <given-names>J. D.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Explainable artificial intelligence (xai) 2.0: a manifesto of open challenges and interdisciplinary research directions</article-title>. <source>Inf. Fusion</source> <volume>106</volume>:<fpage>102301</fpage>. <pub-id pub-id-type="doi">10.1016/j.inffus.2024.102301</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Longo</surname> <given-names>L.</given-names></name> <name><surname>Wickens</surname> <given-names>C. D.</given-names></name> <name><surname>Hancock</surname> <given-names>G.</given-names></name> <name><surname>Hancock</surname> <given-names>P. A.</given-names></name></person-group> (<year>2022</year>). <article-title>Human mental workload: a survey and a novel inclusive definition</article-title>. <source>Front. Psychol</source>. <volume>13</volume>:<fpage>883321</fpage>. <pub-id pub-id-type="doi">10.3389/fpsyg.2022.883321</pub-id><pub-id pub-id-type="pmid">35719509</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Malinin</surname> <given-names>A.</given-names></name></person-group> (<year>2019</year>). <source>Uncertainty estimation in deep learning with application to spoken language assessment</source>. Doctoral dissertation.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Meteier</surname> <given-names>Q.</given-names></name> <name><surname>Capallera</surname> <given-names>M.</given-names></name> <name><surname>Ruffieux</surname> <given-names>S.</given-names></name> <name><surname>Angelini</surname> <given-names>L.</given-names></name> <name><surname>Abou Khaled</surname> <given-names>O.</given-names></name> <name><surname>Mugellini</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Classification of drivers&#x00027; workload using physiological signals in conditional automation</article-title>. <source>Front. Psychol</source>. <volume>12</volume>:<fpage>596038</fpage>. <pub-id pub-id-type="doi">10.3389/fpsyg.2021.596038</pub-id><pub-id pub-id-type="pmid">33679516</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oppelt</surname> <given-names>M. P.</given-names></name> <name><surname>Foltyn</surname> <given-names>A.</given-names></name> <name><surname>Deuschel</surname> <given-names>J.</given-names></name> <name><surname>Lang</surname> <given-names>N. R.</given-names></name> <name><surname>Holzer</surname> <given-names>N.</given-names></name> <name><surname>Eskofier</surname> <given-names>B. M.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>ADABase: a multimodal dataset for cognitive load estimation</article-title>. <source>Sensors</source> <volume>23</volume>:<fpage>340</fpage>. <pub-id pub-id-type="doi">10.3390/s23010340</pub-id><pub-id pub-id-type="pmid">36616939</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Orru</surname> <given-names>G.</given-names></name> <name><surname>Longo</surname> <given-names>L.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;The evolution of cognitive load theory and the measurement of its intrinsic, extraneous and germane loads: a review,&#x0201D;</article-title> in <source>Human Mental Workload: Models and Applications</source>, eds. L. Longo, and M. C. Leva (Cham: Springer International Publishing), <fpage>23</fpage>&#x02013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-14273-5_3</pub-id></citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Paas</surname> <given-names>F.</given-names></name> <name><surname>Tuovinen</surname> <given-names>J. E.</given-names></name> <name><surname>Tabbers</surname> <given-names>H.</given-names></name> <name><surname>Van Gerven</surname> <given-names>P. W. M.</given-names></name></person-group> (<year>2003</year>). <article-title>Cognitive load measurement as a means to advance cognitive load theory</article-title>. <source>Educ. Psychol</source>. <volume>38</volume>, <fpage>63</fpage>&#x02013;<lpage>71</lpage>. <pub-id pub-id-type="doi">10.1207/S15326985EP3801_8</pub-id></citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Paas</surname> <given-names>F. G. W. C.</given-names></name> <name><surname>Van Merri&#x000EB;nboer</surname> <given-names>J. J. G.</given-names></name></person-group> (<year>1994</year>). <article-title>Instructional control of cognitive load in the training of complex cognitive tasks</article-title>. <source>Educ. Psychol. Rev</source>. <volume>6</volume>, <fpage>351</fpage>&#x02013;<lpage>371</lpage>. <pub-id pub-id-type="doi">10.1007/BF02213420</pub-id></citation>
</ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Palinko</surname> <given-names>O.</given-names></name> <name><surname>Kun</surname> <given-names>A. L.</given-names></name></person-group> (<year>2012</year>). <article-title>&#x0201C;Exploring the effects of visual cognitive load and illumination on pupil diameter in driving simulators,&#x0201D;</article-title> in <source>Proceedings of the Symposium on Eye Tracking Research and Applications</source>, 413&#x02013;416. <pub-id pub-id-type="doi">10.1145/2168556.2168650</pub-id></citation>
</ref>
<ref id="B42">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Paszke</surname> <given-names>A.</given-names></name> <name><surname>Gross</surname> <given-names>S.</given-names></name> <name><surname>Massa</surname> <given-names>F.</given-names></name> <name><surname>Lerer</surname> <given-names>A.</given-names></name> <name><surname>Bradbury</surname> <given-names>J.</given-names></name> <name><surname>Chanan</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;Pytorch: an imperative style, high-performance deep learning library,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems 32</source> (<publisher-loc>Curran Associates, Inc.</publisher-loc>), <fpage>8024</fpage>&#x02013;<lpage>8035</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Pedregosa</surname> <given-names>F.</given-names></name> <name><surname>Varoquaux</surname> <given-names>G.</given-names></name> <name><surname>Gramfort</surname> <given-names>A.</given-names></name> <name><surname>Michel</surname> <given-names>V.</given-names></name> <name><surname>Thirion</surname> <given-names>B.</given-names></name> <name><surname>Grisel</surname> <given-names>O.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>Scikit-learn: machine learning in python</article-title>. <source>J. Mach. Learn. Res</source>. <volume>12</volume>, <fpage>2825</fpage>&#x02013;<lpage>2830</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.jmlr.org/papers/v12/pedregosa11a.html">https://www.jmlr.org/papers/v12/pedregosa11a.html</ext-link></citation>
</ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rahman</surname> <given-names>H.</given-names></name> <name><surname>Ahmed</surname> <given-names>M. U.</given-names></name> <name><surname>Barua</surname> <given-names>S.</given-names></name> <name><surname>Funk</surname> <given-names>P.</given-names></name> <name><surname>Begum</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Vision-based driver&#x00027;s cognitive load classification considering eye movement using machine learning and deep learning</article-title>. <source>Sensors</source> <volume>21</volume>:<fpage>8019</fpage>. <pub-id pub-id-type="doi">10.3390/s21238019</pub-id><pub-id pub-id-type="pmid">34884021</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Seitz</surname> <given-names>J.</given-names></name> <name><surname>Maedche</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Biosignal-based recognition of cognitive load: A systematic review of public datasets and classifiers,&#x0201D;</article-title> in <source>Information Systems and Neuroscience: NeuroIS Retreat 2022</source>, 35&#x02013;52. <pub-id pub-id-type="doi">10.1007/978-3-031-13064-9_4</pub-id></citation>
</ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Varma</surname> <given-names>S.</given-names></name> <name><surname>Simon</surname> <given-names>R.</given-names></name></person-group> (<year>2006</year>). <article-title>Bias in error estimation when using cross-validation for model selection</article-title>. <source>BMC Bioinform</source>. <volume>7</volume>, <fpage>1</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1186/1471-2105-7-91</pub-id><pub-id pub-id-type="pmid">16504092</pub-id></citation></ref>
<ref id="B47">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Yan</surname> <given-names>W.</given-names></name> <name><surname>Oates</surname> <given-names>T.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Time series classification from scratch with deep neural networks: a strong baseline,&#x0201D;</article-title> in <source>2017 International Joint Conference on Neural Networks (IJCNN)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>1578</fpage>&#x02013;<lpage>1585</lpage>. <pub-id pub-id-type="doi">10.1109/IJCNN.2017.7966039</pub-id></citation>
</ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wilson</surname> <given-names>J. C.</given-names></name> <name><surname>Nair</surname> <given-names>S.</given-names></name> <name><surname>Scielzo</surname> <given-names>S.</given-names></name> <name><surname>Larson</surname> <given-names>E. C.</given-names></name></person-group> (<year>2021</year>). <article-title>Objective measures of cognitive load using deep multi-modal learning: a use-case in aviation</article-title>. <source>Proc. ACM Inter. Mobile, Wear. Ubiquit. Technol</source>. <volume>5</volume>, <fpage>1</fpage>&#x02013;<lpage>35</lpage>. <pub-id pub-id-type="doi">10.1145/3448111</pub-id></citation>
</ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Huang</surname> <given-names>S.</given-names></name> <name><surname>Xu</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>P.</given-names></name> <name><surname>Wu</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>D.</given-names></name></person-group> (<year>2022</year>). <article-title>Cognitive workload recognition using EEG signals and machine learning: a review</article-title>. <source>IEEE Trans. Cogn. Dev. Syst</source>. <volume>14</volume>, <fpage>799</fpage>&#x02013;<lpage>818</lpage>. <pub-id pub-id-type="doi">10.1109/TCDS.2021.3090217</pub-id></citation>
</ref>
</ref-list>
</back>
</article>