<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1369566</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2024.1369566</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Robotics and AI</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Webcam-based gaze estimation for computer screen interaction</article-title>
<alt-title alt-title-type="left-running-head">Falch and Lohan</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frobt.2024.1369566">10.3389/frobt.2024.1369566</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Falch</surname>
<given-names>Lucas</given-names>
</name>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2573906/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lohan</surname>
<given-names>Katrin Solveig</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/31146/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff>
<institution>Institute for the Development of Mechatronic Systems EMS, Eastern Switzerland University of Applied Sciences (OST)</institution>, <addr-line>Buchs</addr-line>, <country>Switzerland</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1480917/overview">Richard Jiang</ext-link>, Lancaster University, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2640858/overview">Qijie Zhao</ext-link>, Shanghai University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2641878/overview">Changyuan Wang</ext-link>, Xi&#x2019;an Technological University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Lucas Falch, <email>lucas.falch@ost.ch</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>02</day>
<month>04</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>11</volume>
<elocation-id>1369566</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>26</day>
<month>02</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Falch and Lohan.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Falch and Lohan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>This paper presents a novel webcam-based approach for gaze estimation on computer screens. Utilizing appearance based gaze estimation models, the system provides a method for mapping the gaze vector from the user&#x2019;s perspective onto the computer screen. Notably, it determines the user&#x2019;s 3D position in front of the screen, using only a 2D webcam without the need for additional markers or equipment. The study presents a comprehensive comparative analysis, assessing the performance of the proposed method against established eye tracking solutions. This includes a direct comparison with the purpose-built Tobii Eye Tracker 5, a high-end hardware solution, and the webcam-based GazeRecorder software. In experiments replicating head movements, especially those imitating yaw rotations, the study brings to light the inherent difficulties associated with tracking such motions using 2D webcams. This research introduces a solution by integrating Structure from Motion (SfM) into the Convolutional Neural Network (CNN) model. The study&#x2019;s accomplishments include showcasing the potential for accurate screen gaze tracking with a simple webcam, presenting a novel approach for physical distance computation, and proposing compensation for head movements, laying the groundwork for advancements in real-world gaze estimation scenarios.</p>
</abstract>
<kwd-group>
<kwd>eye tracking</kwd>
<kwd>gaze tracking</kwd>
<kwd>webcam based</kwd>
<kwd>gaze on screen</kwd>
<kwd>gaze estimation</kwd>
<kwd>structue from motion</kwd>
</kwd-group>
<contract-num rid="cn001">100.440 IP-ICT</contract-num>
<contract-sponsor id="cn001">Innosuisse&#x2014;Schweizerische Agentur f&#xfc;r Innovationsf&#xf6;rderung<named-content content-type="fundref-id">10.13039/501100013348</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Robot Vision and Artificial Perception</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Gaze tracking serves as a prevalent technique for comprehending human attention. Its utility extends to gauging users&#x2019; attitudes and attention, making it a valuable tool in fields like market research, adaptive information systems, human robot interaction and more recently, entertainment. Despite a reduction in costs for specialized eye tracking equipment, their ubiquity remains limited. Although the idea of using webcams for eye tracking is not new, existing solutions have yet to achieve the level of accuracy and sampling rates found in even basic commercial eye trackers like those from <xref ref-type="bibr" rid="B28">Tobii Group (2023)</xref>.</p>
<p>Webcam-based eye tracking solutions often provide a gaze vector from the user&#x2019;s viewpoint, indicating the direction of the person&#x2019;s gaze. However, these solutions often lack the ability to project this gaze vector into the environment due to the unknown distance between, for instance, a computer screen and the individual. In this study, we present a novel webcam-based gaze tracking approach designed for precise determination of a user&#x2019;s gaze on a computer screen. Our approach utilizes appearance-based gaze estimation models, like the OpenVino model (<xref ref-type="bibr" rid="B21">OpenVINO, 2023</xref>) or a model trained on the ETH-XGaze dataset (<xref ref-type="bibr" rid="B14">hysts on github, 2023</xref>), to determine a unit gaze vector. This vector facilitates the calculation of the distance between the user and the computer screen. It is important to note that our approach works with any model providing a gaze vector from the user&#x2019;s view point. The accuracy of our method is thus contingent on the precision of the model supplying the unit gaze vector. Our approach considers the user&#x2019;s spatial positioning in front of the screen, allowing us to compute the user&#x2019;s physical distance from the screen using only a 2D webcam, eliminating the need for additional markers or equipment.</p>
<p>As part of this study, we conduct a comprehensive comparative analysis, evaluating the performance of our proposed method against established eye-tracking solutions. This includes a head-to-head assessment with the purpose-built Tobii Eye Tracker 5, a cutting-edge hardware solution. We also examine the webcam-based GazeRecorder software and provide insights into our choice of this particular software, which will be explained further in the subsequent section.</p>
<p>The organization of the rest of the paper is as follows. In <xref ref-type="sec" rid="s2">Section 2</xref>, the related work is presented while indicating the shortcomings of the existing work. <xref ref-type="sec" rid="s3">Section 3</xref> outlines the associated methodologies and underlying principles used in the present work. In <xref ref-type="sec" rid="s4">Section 4</xref> the developed methods are applied and experiments are conducted. A conclusion and directions for possible future work are presented in <xref ref-type="sec" rid="s5">Section 5</xref>.</p>
</sec>
<sec id="s2">
<title>2 Related work</title>
<p>Eye gaze tracking has been a topic of significant research interest for decades, with various techniques and applications developed to capture and analyze the direction and movement of the eyes in order to infer the user&#x2019;s attention, intention, or interest. There are various techniques available to track human eye gaze. Broadly, two main approaches dominate the field of eye gaze tracking: model-based and appearance-based (<xref ref-type="bibr" rid="B8">Ferhat and Vilari&#xf1;o 2016</xref>; <xref ref-type="bibr" rid="B16">Kar and Corcoran 2017</xref>).</p>
<p>
<bold>Model-based</bold> methods rely on meticulously crafted geometric models of the eye. In corneal reflection techniques, external lighting, often near infra-red (NIR) LEDs, create corneal glints used to extract the eye region and estimate gaze through 2D regression, mapping the vector between pupil center and glint to the corresponding gaze coordinate on the screen (<xref ref-type="bibr" rid="B31">Yoo and Chung, 2005</xref>; <xref ref-type="bibr" rid="B36">Zhu and Ji, 2005</xref>; <xref ref-type="bibr" rid="B13">Hennessey et al., 2006</xref>). Conversely, shape-based methods derive gaze direction from observed eye shapes like pupil centers and iris edges (<xref ref-type="bibr" rid="B4">Chen and Ji, 2008</xref>; <xref ref-type="bibr" rid="B10">Hansen and Pece 2005</xref>). These models are employed in 3D-based approaches to estimate corneal center, optical and visual axes, and ultimately, gaze coordinates by determining intersections with the scene (<xref ref-type="bibr" rid="B24">Sogo, 2013</xref>). Additionally, cross-ratio based methods utilize four LEDs placed at screen corners, reflecting off the cornea to estimate gaze projection on the screen plane (<xref ref-type="bibr" rid="B1">Arar et al., 2016</xref>).</p>
<p>However, <bold>appearance-based</bold> methods take a different approach, directly utilizing eye images as input. These methods often employ feature extraction techniques, like face and eye detection, before employing advanced algorithms like CNNs to estimate the point of regard (PoR) (<xref ref-type="bibr" rid="B26">Tan et al., 2002</xref>; <xref ref-type="bibr" rid="B23">Sewell and Komogortsev, 2010</xref>; <xref ref-type="bibr" rid="B18">Muhammad Usman Ghani and Chaudhry, 2013</xref>). For example, Chi et al. <xref ref-type="bibr" rid="B5">Chi et al. (2009)</xref> leverage an active infrared light source and a single camera to calculate the gaze and utilize a neural network to compensate for head movements by tracking relative positions between pupil and corneal reflex center. Recent advancements underscore the dynamic nature of this field, with newer methods emerging that do not rely on external lighting and the advent of large-scale datasets has significantly advanced deep learning methodologies by providing millions of annotated samples for training CNNs to map facial attributes and eye images to gaze directions. <xref ref-type="bibr" rid="B34">Zhang et al. (2015)</xref> presented the MPIIGaze dataset, where they employed a 3D facial shape model to estimate the 3D poses of detected faces. Their CNN architecture learned the mapping from head poses and eye images to gaze directions in the camera coordinate system, considering the 3D rotation of the head from its coordinate system to the camera&#x2019;s. <xref ref-type="bibr" rid="B25">Sugano et al. (2014)</xref> used a similar approach, training a 3D gaze estimator and performing 3D reconstruction to generate dense training data for eye images. Ground truth 3D gaze directions were provided in the 3D world coordinate system. <xref ref-type="bibr" rid="B17">Krafka et al. (2016)</xref> introduced the GazeCapture dataset and trained a CNN for eye tracking. They achieving tracking errors of 1.71 <italic>cm</italic> and 2.53 <italic>cm</italic> on iPhone and tablet devices, respectively. Their end-to-end CNN model for gaze prediction was trained without relying on features such as head pose or eye center location. The dataset was meticulously collected using iPhones and iPads with known camera locations and screen sizes and it is unclear how it performs on different devices, screens or cameras. Another contribution by <xref ref-type="bibr" rid="B35">Zhang et al. (2017)</xref> proposed a method for learning a gaze estimator solely from full-face images in an end-to-end manner. They introduced a spatial weights CNN method that leveraged information from the entire face. Additionally, the ETH-XGaze dataset (<xref ref-type="bibr" rid="B32">Zhang et al., 2020</xref>), with one million labeled samples, has emerged as a valuable resource for gaze estimation research. Numerous studies have utilized this dataset, with <xref ref-type="bibr" rid="B3">Cai et al. (2021)</xref> achieving top-ranking performance on the ETH-XGaze competition leaderboard, attaining an average angular error of 3.11&#xb0;. Many of these appearance-based method only provide a gaze vector and compare their approach between datasets and gaze vector accuracy without even considering the actual point where a person is looking.</p>
<p>Therefore, there remains a lack of consensus in the literature regarding how to quantify the accuracy of gaze estimation approaches. While it would seem logical to denote accuracy as the error in gaze angle estimation, some studies instead measure it as the Euclidean distance between the estimated point of gaze and the true Point of Regard (PoR). Typically, researchers determine the true and estimated PoR by instructing test subjects to focus on specific points displayed on a screen. The challenge with using Euclidean distance lies in the absence of information regarding the distance between the screen and the user, making it challenging to compare different methods effectively.</p>
<p>In a recent study <xref ref-type="bibr" rid="B12">Heck et al. (2023)</xref> analyzed 16 publicly available webcam based gaze estimation software options, offering a comprehensive overview of the techniques employed. The evaluation was performed with a fixed head position and an average user-to-screen distance of 60 cm, as only a subset of the solutions allowed for head movements. In their study, the online eye tracker WebGazer (<xref ref-type="bibr" rid="B22">Papoutsaki et al., 2016</xref>) with a mentioned accuracy of 4.06 <italic>cm</italic> detects the pupil in a video frame and uses the location to linearly estimate a gaze coordinate on the screen. In addition the eye is treated as a multi-dimensional feature vector using clmtrackr (<xref ref-type="bibr" rid="B2">Audun, 2023</xref>) and histogram equalization similar to <xref ref-type="bibr" rid="B30">Xu et al. (2015)</xref>. To map the pupil location and eye features onto the screen they use continual self-calibration through user interactions, by assuming that gaze locations on the screen match the coordinates of that interaction. Features and user interactions such as clicks and cursor movements are the inputs for a regularized linear regression model to match eye features to gaze locations. The most accurate webcam-based eye tracking solution observed is GazeRecorder (<xref ref-type="bibr" rid="B9">GazeRecorder, 2023</xref>) with an accuracy of 1.75 <italic>cm</italic>, a commercial solution, however details about the gaze estimation method is unknown. Allowing head movements typically results in a decrease in accuracy, even observed in GazeRecorder GazeRecorder (accessed 2023), which experienced a 9% reduction.</p>
<p>2D gaze estimation typically involves formulating a regression problem, where the input image is mapped to a 2-dimensional on-screen gaze location <italic>p</italic> using a regression function <italic>f</italic>(<italic>I</italic>), where <italic>f</italic> represents the regression function and <italic>p</italic> is typically defined on the target screen (<xref ref-type="bibr" rid="B35">Zhang et al., 2017</xref>). However, the trained regression function may not be directly applicable to different cameras without addressing differences in projection models. A common assumption in many approaches is that the target screen plane remains fixed in the camera coordinate system, limiting the freedom of camera movement after training, which poses a practical constraint. Many existing methods report only the error in centimeters of the target point without mentioning the distance between the target plane and the user. They also often compare different methods based on specific datasets, overlooking real-world scenarios where users sit in front of a screen and focus on specific points of interest.</p>
<p>In this research project, we aim to overcome this limitation by proposing a method to determine the physical distance between the user and the screen, using only the unit gaze vector provided by many appearance-based methods. In these methods, CNNs are typically trained on labeled datasets, yielding a gaze vector without projecting it onto a surface. In our proposed approach, we will select two trained CNN gaze vector models and provide a methodology to project the gaze vector onto a screen.</p>
</sec>
<sec sec-type="methods" id="s3">
<title>3 Methodology</title>
<sec id="s3-1">
<title>3.1 Acquisition of gaze vector</title>
<p>In this work, we use OpenVINO (Open Visual Inference and Neural Network Optimization) an open-source toolkit developed by Intel for computer vision and deep learning applications. The toolkit includes a range of pre-trained models to simplify the development of computer vision and deep learning applications. The trained models exhibit an optimized format, enabling them to run at maximum efficiency on Intel hardware. Specifically, we use the pre-trained &#x201c;gaze-estimation-adas-0002&#x201d; model, which relies on &#x201c;face-detection-adas-0001,&#x201d; &#x201c;head-pose-estimation-adas-0001&#x201d; and &#x201c;facial-landmarks-35-adas-0002&#x201d; models OpenVINO (accessed 2023). First, the face is detected in the webcam video stream. With that, the eye-centers and eyes are detected within the face using &#x201c;landmarks-regression-retail-0009&#x201d; model. The detected face is also the input of the head pose estimation model. The gaze estimation model provides a gaze vector corresponding to the direction of a person&#x2019;s gaze in a Cartesian coordinate system in which the <italic>z</italic>-axis is directed from person&#x2019;s eyes (mid-point between left and right eyes&#x2019; centers) to the camera center, the <italic>y</italic>-axis is vertical, and the <italic>x</italic>-axis is orthogonal to <italic>z</italic> and <italic>y</italic>. This gaze vector is a unit vector and the length is unknown. The subsequent sections will clarify the definition of the coordinate system and visually present them in <xref ref-type="fig" rid="F1">Figures 1</xref>, <xref ref-type="fig" rid="F2">2</xref>. We further make use of the &#x201c;pl gaze estimation&#x201d; model hysts on github (accessed 2023), sourced from a GitHub repository housing trained models derived from various datasets, including MPIIGaze (<xref ref-type="bibr" rid="B34">Zhang et al., 2015)</xref>, MPIIFaceGaze (<xref ref-type="bibr" rid="B35">Zhang et al., 2017</xref>), and ETH-XGaze (<xref ref-type="bibr" rid="B32">Zhang et al., 2020</xref>). Specifically, we utilize the model trained on the ETH-XGaze dataset, as showcased in their demo. While numerous gaze vector models are accessible, including offerings from Nvidia (<xref ref-type="bibr" rid="B19">NVIDIA, 2023</xref>), we opt for the Intel model, aligning with our hardware infrastructure. Additionally, an online competition assessing the accuracy of gaze vectors for the ETH-XGaze dataset is available (<xref ref-type="bibr" rid="B6">CodaLab, 2023</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Definition of the gaze coordinate system, denoted as <italic>G</italic>, where the <italic>x</italic>-axis indicates the leftward direction, the <italic>y</italic>-axis points upwards, and the <italic>z</italic>-axis extends towards the screen.</p>
</caption>
<graphic xlink:href="frobt-11-1369566-g001.tif"/>
</fig>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Definition of coordinate systems. <italic>S</italic> is the screen coordinate system, <italic>W</italic> the world or camera coordinate system and <italic>G</italic> the gaze coordinate system additionally depicted in <xref ref-type="fig" rid="F1">Figure 1</xref>. The green vectors illustrate the process of projecting a unit gaze vector onto the screen, as described in detail in <xref ref-type="sec" rid="s3-3">Section 3.3</xref>.</p>
</caption>
<graphic xlink:href="frobt-11-1369566-g002.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 Definition of coordinate systems</title>
<p>The OpenVino gaze estimation model&#x2019;s input is an image of a person and it delivers a unit gaze vector indicating the direction of the person&#x2019;s gaze. This gaze vector is represented within the coordinate system denoted as <italic>G</italic>, as illustrated in <xref ref-type="fig" rid="F1">Figures 1</xref>, <xref ref-type="fig" rid="F2">2</xref>. Specifically, the OpenVino model provides this unit gaze vector within the <italic>G</italic> coordinate system, with its location depicted between the eyes in <xref ref-type="fig" rid="F1">Figure 1</xref>. While the exact placement of the gaze coordinate system <italic>G</italic> is not specified on the OpenVino website, its precise location is inconsequential for our method. We determine its position relative to the screen through our proposed approach, as detailed in subsequent sections.</p>
<p>Regarding the ETH-XGaze dataset (<xref ref-type="bibr" rid="B33">Zhang et al., 2018</xref>; <xref ref-type="bibr" rid="B32">Zhang et al., 2020</xref>), they outline an image normalization procedure where they designate the face center as the midpoint of the four eye corners and two nose corners. Our alignment process for gaze vectors produced by CNN models involves ensuring that the <italic>z</italic>-component of the vector is positive when oriented towards the screen. Additionally, the <italic>x</italic>-component is positive when looking to the left, and the y-component is positive when looking upward. However, this definition is not obligatory for the functionality of our method; only the adjustment of the rotation matrix (Eq. <xref ref-type="disp-formula" rid="e8">8</xref>) is necessary to align it with the screen coordinate system <italic>S</italic>, and rotation matrix with the world coordinate system <italic>W</italic>, representing the camera coordinate system.</p>
<p>Because the CNN models deliver only a unit vector into the direction where a person is looking, the location of the gaze on the screen is unknown. Therefore, a projection of the gaze vector onto the screen is necessary. In this paper, we provide a methodology to project a gaze vector with unit length onto a screen with unknown distance to the screen.</p>
</sec>
<sec id="s3-3">
<title>3.3 Projecting gaze unit vector onto computer screen</title>
<p>In the upcoming sections, we will present a novel approach for calculating a homogeneous transformation matrix to determine the position of a user looking at a screen through a webcam. We&#x2019;ll employ the following notation for the homogeneous transformation matrix (Eq. <xref ref-type="disp-formula" rid="e1">1</xref>).<disp-formula id="e1">
<mml:math id="m1">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msup>
<mml:mrow>
<mml:mn mathvariant="bold">0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(1)</label>
</disp-formula>
<inline-formula id="inf1">
<mml:math id="m2">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> is the transformation matrix that transforms a vector in coordinate system <italic>B</italic> to coordinate system <italic>A</italic>. <inline-formula id="inf2">
<mml:math id="m3">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> are the unit vectors of the coordinate system <italic>B</italic> presented in coordinate system <italic>A</italic> and <inline-formula id="inf3">
<mml:math id="m4">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> is the displacement from <italic>A</italic> to coordinate system <italic>B</italic> presented in coordinate system <italic>A</italic>.</p>
<sec id="s3-3-1">
<title>3.3.1 Transformation matrix screen to gaze</title>
<p>The approach to ascertain the user&#x2019;s distance from the screen and subsequently project the gaze vector onto the computer screen involves the computation of the matrix <sup>
<italic>S</italic>
</sup>
<italic>T</italic>
<sub>
<italic>G</italic>
</sub>. The transformation matrix <sup>
<italic>S</italic>
</sup>
<italic>T</italic>
<sub>
<italic>G</italic>
</sub> allows the transformation of the gaze vector <sup>
<italic>G</italic>
</sup>
<inline-formula id="inf4">
<mml:math id="m5">
<mml:mi>g</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> from the gaze coordinate system <italic>G</italic> to the screen coordinate system <italic>S</italic> (Eqs <xref ref-type="disp-formula" rid="e2">2</xref>, <xref ref-type="disp-formula" rid="e3">3</xref>).<disp-formula id="e2">
<mml:math id="m6">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:none/>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x3d;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:msup>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi>g</mml:mi>
</mml:math>
<label>(2)</label>
</disp-formula>
<disp-formula id="e3">
<mml:math id="m7">
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msup>
<mml:mrow>
<mml:mn mathvariant="bold">0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
<disp-formula id="e4">
<mml:math id="m8">
<mml:mo>&#x3d;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
<label>(4)</label>
</disp-formula>The gaze unit vector, denoted as <inline-formula id="inf5">
<mml:math id="m9">
<mml:mmultiscripts>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:none/>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>, and the scalar <inline-formula id="inf6">
<mml:math id="m10">
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:math>
</inline-formula>, project the gaze vector onto the screen. To find <italic>&#x3bb;</italic> such that the gaze vector <inline-formula id="inf7">
<mml:math id="m11">
<mml:mmultiscripts>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:none/>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> intersects with the screen, the scalar product of the screen&#x2019;s <italic>z</italic>-axis and the vector located on the screen&#x2019;s plane must be zero (as shown in Eq. <xref ref-type="disp-formula" rid="e5">5</xref>).<disp-formula id="e5">
<mml:math id="m12">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:math>
<label>(5)</label>
</disp-formula>
<disp-formula id="e6">
<mml:math id="m13">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:none/>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x3d;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x22c5;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mn>0,0,1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
<label>(6)</label>
</disp-formula>For a clearer understanding of this equation, please refer to <xref ref-type="fig" rid="F2">Figure 2</xref>, where the dashed green arrowheads represent the vectors involved. The point <italic>p</italic> corresponds to the point on the screen where the person is looking. The vector <inline-formula id="inf8">
<mml:math id="m14">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> originates from the transformation matrix <inline-formula id="inf9">
<mml:math id="m15">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula>, which we will determine in the regression problem discussed in <xref ref-type="sec" rid="s3-3-3">Section 3.3.3</xref>. The scalar product with the <italic>z</italic>-coordinate of the screen coordinate system becomes zero when the vector <inline-formula id="inf10">
<mml:math id="m16">
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> lies on the plane of the screen, allowing us to determine the scalar <italic>&#x3bb;</italic>. Eq. <xref ref-type="disp-formula" rid="e6">6</xref> rotates the <italic>z</italic>-axis ([0,0,1]) in screen coordinates to the gaze coordinate system, where <inline-formula id="inf11">
<mml:math id="m17">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> follows from the definition of <xref ref-type="fig" rid="F2">Figure 2</xref>. With that we can compute the scaling factor <italic>&#x3bb;</italic> from Eq. <xref ref-type="disp-formula" rid="e5">5</xref>.<disp-formula id="e7">
<mml:math id="m18">
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mspace width="0.28em"/>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mrow>
<mml:mrow>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mspace width="0.28em"/>
<mml:mmultiscripts>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:none/>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
<p>As mentioned earlier the &#x201c;gaze-estimation-adas-0002&#x201d; model takes as input the two eye images and the head rotation angle and delivers a gaze unit vector, whereby the coordinate system of the gaze vector is rotated so that its <italic>x</italic>/<italic>y</italic>-plane is parallel to the <italic>x</italic>/<italic>y</italic>-plane of the camera. If the webcam is mounted on top of the screen and the <italic>x</italic>/<italic>y</italic>-plane of the webcam coincides with the <italic>x</italic>/<italic>y</italic>-plane of the screen we can determine the rotation matrix <inline-formula id="inf12">
<mml:math id="m19">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula>.<disp-formula id="e8">
<mml:math id="m20">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(8)</label>
</disp-formula>Therefore, the only unknown is the translation vector <inline-formula id="inf13">
<mml:math id="m21">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> in Eq. <xref ref-type="disp-formula" rid="e4">4</xref>. This translation vector is part of the homogeneous transformation matrix, which is determined through a regression analysis (<xref ref-type="sec" rid="s3-3-3">Section 3.3.3</xref>).</p>
</sec>
<sec id="s3-3-2">
<title>3.3.2 Calibration process</title>
<p>To gather data for the regression problem, we initiated a calibration process in which a user is presented with points on the screen and instructed to direct their gaze towards these points, which is the standard method. Our approach simplifies the calibration process, involving the presentation of just four calibration points, denoted as <sup>
<italic>S</italic>
</sup>
<inline-formula id="inf14">
<mml:math id="m22">
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">C</mml:mi>
</mml:math>
</inline-formula>, on the screen. These calibration points, serve as the user&#x2019;s focal points. Each point is observed for approximately 2 s, yielding multiple data points for the regression problem from a single point of focus. Transitioning to a new point of focus may introduce a brief lag in gaze redirection, leading to instances where the user does not precisely follow the presented point on the screen. Hence, in the data stream, any erroneous data points, arising when the user gazes at positions different from the presented point, are eliminated prior to the application of the regression model. It is important to note that users were directed to maintain a steady head position during the calibration process, despite not employing a headrest.</p>
</sec>
<sec id="s3-3-3">
<title>3.3.3 Regression problem</title>
<p>With a set <inline-formula id="inf15">
<mml:math id="m23">
<mml:mi mathvariant="script">C</mml:mi>
</mml:math>
</inline-formula> of known calibration points <sup>
<italic>S</italic>
</sup>
<inline-formula id="inf16">
<mml:math id="m24">
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">C</mml:mi>
</mml:math>
</inline-formula> displayed on the screen and the gaze vector <inline-formula id="inf17">
<mml:math id="m25">
<mml:mmultiscripts>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:none/>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> provided by any gaze vector estimation model, we can determine the translation vector <inline-formula id="inf18">
<mml:math id="m26">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> by formulating a regression problem in the following way (Eqs <xref ref-type="disp-formula" rid="e9">9</xref>&#x2013;<xref ref-type="disp-formula" rid="e12">12</xref>).<disp-formula id="e9">
<mml:math id="m27">
<mml:mtable class="align" columnalign="left">
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:munder>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mspace width="1em"/>
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">C</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
<label>(9)</label>
</disp-formula>
<disp-formula id="e10">
<mml:math id="m28">
<mml:mtext>with&#x2009;</mml:mtext>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:msup>
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mspace width="0.17em"/>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
<label>(10)</label>
</disp-formula>
<disp-formula id="e11">
<mml:math id="m29">
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mspace width="0.28em"/>
<mml:mo>&#x22c5;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mspace width="0.28em"/>
<mml:mmultiscripts>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(11)</label>
</disp-formula>
<disp-formula id="e12">
<mml:math id="m30">
<mml:mi>x</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>,</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>,</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
<label>(12)</label>
</disp-formula>The coordinates <inline-formula id="inf19">
<mml:math id="m31">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> represent the <italic>x</italic>, <italic>y</italic>, and <italic>z</italic> coordinates of the vector <inline-formula id="inf20">
<mml:math id="m32">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula>. <inline-formula id="inf21">
<mml:math id="m33">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> is derived in Eq. <xref ref-type="disp-formula" rid="e4">4</xref>, while <italic>&#x3bb;</italic>
<sup>
<italic>i</italic>
</sup> is obtained from Eq. <xref ref-type="disp-formula" rid="e7">7</xref>. Given the knowledge of the calibration points&#x2019; poses and thus the distances between these points on the screen, we have both a physical reference and a direction provided by the set of gaze vectors <sup>
<italic>G</italic>
</sup>
<italic>g</italic>
<sup>
<italic>i</italic>
</sup>. Leveraging this information, we can determine the 3D location of the gaze coordinate system <italic>G</italic> through the proposed regression problem. Subsequently, we can ascertain the homogeneous transformation matrix <inline-formula id="inf22">
<mml:math id="m34">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> and project the unit gaze vector <inline-formula id="inf23">
<mml:math id="m35">
<mml:mmultiscripts>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:none/>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> onto the computer screen, as detailed in <xref ref-type="sec" rid="s3-3">Section 3.3</xref>.</p>
</sec>
<sec id="s3-3-4">
<title>3.3.4 Improve gaze estimation</title>
<p>The accuracy of the gaze vector model, in this case, the OpenVino gaze vector, is constrained by inherent limitations. Notably, the accuracy of the gaze vector provided by OpenVino exhibits a Mean Absolute Error (MAE) of 6.95&#xb0; OpenVINO (accessed 2023). This inaccuracy became apparent in the experiment, where four points forming a rectangular pattern were presented on the screen. Examination of the raw gaze data from the OpenVino model shows that the recorded coordinates do not align with the expected rectangular pattern (<xref ref-type="fig" rid="F3">Figure 3</xref>) that were presented as calibration points on the screen. This occurs because variations in lighting conditions, such as light sources positioned at the side, can result in reflections in the eye, potentially distorting the output.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>On the left are the already filtered <italic>x</italic> and <italic>y</italic> coordinates of the raw unit gaze vector. Different colors present different calibration points. The red cross presents the median of the noisy gaze vector values. On the right are the raw gaze vectors projected onto the computer screen. The yellow star shows the location, where the points were displayed in the calibration process.</p>
</caption>
<graphic xlink:href="frobt-11-1369566-g003.tif"/>
</fig>
<p>However, during the calibration process we are able to reduce this inaccuracy, which we will explain next. A homogeneous transformation matrix, incorporating a gaze vector as the translation vector from the calibration, yields an accurate projection for a specific calibration point, albeit limited to that point alone. Consequently, we derive additional transformation matrices from calibration points, each accurate only for its respective calibration point location. Instead the transformation matrix determined in the previous <xref ref-type="sec" rid="s3-3-3">Section 3.3.3</xref> provides a comprehensive estimation across all points. The subsequent equations provide these additional estimations (Eqs <xref ref-type="disp-formula" rid="e13">13</xref>, <xref ref-type="disp-formula" rid="e14">14</xref>).<disp-formula id="e13">
<mml:math id="m36">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mi>I</mml:mi>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(13)</label>
</disp-formula>
<disp-formula id="e14">
<mml:math id="m37">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mo>&#x2212;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(14)</label>
</disp-formula>
<disp-formula id="e15">
<mml:math id="m38">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x3d;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
<label>(15)</label>
</disp-formula>
<italic>p</italic> &#x2208; <italic>P</italic> is the set of calibration points <italic>p</italic> and <inline-formula id="inf24">
<mml:math id="m39">
<mml:mmultiscripts>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:none/>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> is the unit gaze vector from a person looking at this point during the calibration process. <inline-formula id="inf25">
<mml:math id="m40">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> are the calibration points displayed on the screen during the calibration process. The scaling factor <italic>&#x3bb;</italic> is computed over Eq. <xref ref-type="disp-formula" rid="e7">7</xref> by using <inline-formula id="inf26">
<mml:math id="m41">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> from the transformation matrix determined over the regression process. The matrix <inline-formula id="inf27">
<mml:math id="m42">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> computed in Eq. <xref ref-type="disp-formula" rid="e15">15</xref> transforms the unit gaze vector for this specific point to the screen with zero error.</p>
<p>In total, this procedure provides <italic>n</italic> &#x2b; 1 transformations <inline-formula id="inf28">
<mml:math id="m43">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula>, with <italic>n</italic> for the number of calibration points and an additional one determined through the regression problem, providing the distance between the user and the computer screen.</p>
<p>To determine the final gaze point on the screen, we first compute an initial gaze point on the screen through the transformation matrix determined through the regression analysis in <xref ref-type="sec" rid="s3-3-3">Section 3.3.3</xref>, which will provide an initial guess. We improve this guess by computing the euclidean to the known calibration points on the screen and choosing as additional transformation matrix, the one closest to this calibration point. As an approximation we define the median between these two gaze estimation points as the final point on the screen.</p>
<p>Additionally, we propose a framework to compensate for head movements, specifically where a user moves to the left or right in front of the screen. To consider these movements we make use of structure from motion.</p>
</sec>
<sec id="s3-3-5">
<title>3.3.5 Structure from motion for head movements</title>
<p>Structure from Motion (SfM) is a powerful technique in computer vision that aims to extract three-dimensional (3D) information from a collection of two-dimensional (2D) frames. One of the fundamental problems in SfM is to simultaneously estimate multiple crucial parameters from a set of point correspondences between two images (<xref ref-type="bibr" rid="B11">Hartley and Zisserman 2004</xref>). These parameters include the 3D coordinates of points in space (<sup>
<italic>W</italic>
</sup>
<italic>p</italic>), the relative motion of cameras (<sup>
<italic>W</italic>
</sup>
<italic>R</italic>, <sup>
<italic>W</italic>
</sup>
<italic>t</italic>), and the intrinsic properties of the cameras (<italic>K</italic>
<sub>1</sub>, <italic>K</italic>
<sub>2</sub>). The intrinsic properties of the camera are determined over the OpenCV camera calibration library (<xref ref-type="bibr" rid="B20">OpenCV, 2023</xref>). For feature matching between consecutive frames we use the OpenVino &#x201c;facial-landmarks-35-adas-0002&#x201d; library (<xref ref-type="bibr" rid="B15">Intel Corporation, 2023</xref>). With these features we determine the essential matrix <italic>E</italic> and recover the rotation matrix and translation vector between two frames. By triangulating the points we are able to get 3D points in the world/camera coordinate system. However, we do not know the scaling factor, because there is no reference distance in the video stream. Therefore, we formulate a complete regression problem to determine the homogeneous transformation matrix <sup>
<italic>W</italic>
</sup>
<italic>T</italic>
<sub>
<italic>G</italic>
</sub>(<italic>t</italic>). Moving the head to the left or to the right shifts the gaze coordinate system <italic>G</italic>. With SfM we are able to determine a unit translation vector of the transformation matrix <sup>
<italic>W</italic>
</sup>
<italic>T</italic>
<sub>
<italic>G</italic>
</sub>(<italic>t</italic>) from the world coordinate system <italic>W</italic> to the gaze coordinate system <italic>G</italic>. As mentioned earlier, given the unknown scaling factor, we will derive the gaze vector <inline-formula id="inf29">
<mml:math id="m44">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:none/>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> on the screen using the following equations (Eqs <xref ref-type="disp-formula" rid="e16">16</xref>&#x2013;<xref ref-type="disp-formula" rid="e18">18</xref>).<disp-formula id="e16">
<mml:math id="m45">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:none/>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi>g</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(16)</label>
</disp-formula>
<disp-formula id="e17">
<mml:math id="m46">
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(17)</label>
</disp-formula>
<disp-formula id="e18">
<mml:math id="m47">
<mml:mo>&#x3d;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
<label>(18)</label>
</disp-formula>With <inline-formula id="inf30">
<mml:math id="m48">
<mml:mi>&#x3bc;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:math>
</inline-formula> and <italic>&#x3bb;</italic> defined in Eq. <xref ref-type="disp-formula" rid="e7">7</xref>, we compute <inline-formula id="inf31">
<mml:math id="m49">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> over the following equation. To enhance readability, we will hereafter omit explicit time dependencies in the variables:<disp-formula id="e19">
<mml:math id="m50">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:msup>
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
<label>(19)</label>
</disp-formula>
<disp-formula id="e20">
<mml:math id="m51">
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(20)</label>
</disp-formula>
<disp-formula id="e21">
<mml:math id="m52">
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x2b;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(21)</label>
</disp-formula>
</p>
<p>With Eqs <xref ref-type="disp-formula" rid="e19">19</xref>&#x2013;<xref ref-type="disp-formula" rid="e21">21</xref> we can compute <inline-formula id="inf32">
<mml:math id="m53">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> in the following way (Eqs <xref ref-type="disp-formula" rid="e22">22</xref>, <xref ref-type="disp-formula" rid="e23">23</xref>).<disp-formula id="e22">
<mml:math id="m54">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x3d;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x2b;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
<label>(22)</label>
</disp-formula>
<disp-formula id="e23">
<mml:math id="m55">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(23)</label>
</disp-formula>
</p>
<p>
<sup>
<italic>W</italic>
</sup>
<inline-formula id="inf33">
<mml:math id="m56">
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> is determined over SfM (<xref ref-type="sec" rid="s3-3-5">Section 3.3.5</xref>) and with the following regression model (Eqs <xref ref-type="disp-formula" rid="e24">24</xref>&#x2013;<xref ref-type="disp-formula" rid="e27">27</xref>) we determine <italic>&#x3bc;</italic> and <inline-formula id="inf34">
<mml:math id="m57">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula>.<disp-formula id="e24">
<mml:math id="m58">
<mml:munder>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mspace width="1em"/>
<mml:munderover accentunder="false" accent="true">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="script">C</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:math>
<label>(24)</label>
</disp-formula>
<disp-formula id="e25">
<mml:math id="m59">
<mml:mtable class="align" columnalign="left">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:mtext>with</mml:mtext>
</mml:mtd>
<mml:mtd columnalign="left">
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>&#x3d;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
<label>(25)</label>
</disp-formula>
<disp-formula id="e26">
<mml:math id="m60">
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mspace width="0.28em"/>
<mml:mo>&#x22c5;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mspace width="0.28em"/>
<mml:mmultiscripts>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(26)</label>
</disp-formula>
<disp-formula id="e27">
<mml:math id="m61">
<mml:mi>x</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>,</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mo>,</mml:mo>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
<label>(27)</label>
</disp-formula>
<inline-formula id="inf35">
<mml:math id="m62">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> are the <italic>x</italic>, <italic>y</italic>, <italic>z</italic> coordinates of the vector <inline-formula id="inf36">
<mml:math id="m63">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula>. By solving this regression problem and knowing the parameters in <italic>x</italic> we can compute the transformation matrix <inline-formula id="inf37">
<mml:math id="m64">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> in Eq. <xref ref-type="disp-formula" rid="e19">19</xref>. To collect data, in order to, solve the regression problem we perform the calibration procedure described in <xref ref-type="sec" rid="s3-3-2">Section 3.3.2</xref>. The methods presented in this work have been implemented in Python and are publicly available on GitHub (<xref ref-type="bibr" rid="B7">Falch, 2023</xref>).</p>
</sec>
</sec>
</sec>
<sec id="s4">
<title>4 Experiments and results</title>
<sec id="s4-1">
<title>4.1 Evaluating the validity of the proposed method</title>
<p>In this study, our primary focus is not on evaluating the precision of the OpenVino CNN gaze estimation model or the model trained on the ETH-XGaze dataset. Instead, our goal is to showcase our proposed method, demonstrated on a single user. As our work does not involve the determination of a gaze vector, we utilize the pre-existing OpenVino gaze vector model. The reported accuracy of the OpenVino gaze vector, as per internal assessments, is 6.95&#xb0; OpenVINO (accessed 2023). In addition we showcase our method on the demo program for gaze estimation, specifically the ETH-XGaze model available on github hysts on github (accessed 2023). Further available models are based on MIIGaze (<xref ref-type="bibr" rid="B34">Zhang et al. 2015</xref>), MPIIFaceGaze (<xref ref-type="bibr" rid="B35">Zhang et al., 2017</xref>) or any other model available online.</p>
<p>For this demonstration, we engage a user in a simple calibration process. The user is asked to maintain a stable head position and focus on four specific calibration points displayed near the corners of the screen within a rectangle. However, due to the inaccuracy of OpenVino&#x2019;s CNN model, the raw gaze vector provided by the OpenVino model lacks precision, and the result is not a perfect rectangle (see <xref ref-type="fig" rid="F3">Figure 3</xref>).</p>
<p>To determine the distance between the user and the computer screen and to project a unit gaze vector onto the screen, we utilize the proposed regression model, which yields an estimated transformation matrix denoted as <inline-formula id="inf38">
<mml:math id="m65">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula>. With this transformation matrix we can now determine the 3D position of the user in front of the computer screen with only a 2D webcam, without additional hardware or markers.</p>
<p>To validate the user&#x2019;s position in front of the screen, represented by the translation vector of the homogeneous transformation matrix <inline-formula id="inf39">
<mml:math id="m66">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula>, we conduct physical measurements. This involves determining the distance between the user and the screen and measuring the location from the upper left corner of the screen to the gaze coordinate system, as defined in <xref ref-type="fig" rid="F2">Figure 2</xref>. This verifies our approach and is therefore the first method we know of, computing the actual physical distance, respectively location of a user in front of a computer screen with just four calibration points and without using stereo imaging. Notably, the accuracy of this 3D position is contingent on the precision of the model providing the gaze vector, in our case, the OpenVino model and the model trained on ETH-XGaze. To further validate our method, we calculate the exact vector that the OpenVino model should have delivered using the inverse of the transformation matrix <inline-formula id="inf40">
<mml:math id="m67">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:msup>
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>. Employing the transformation matrix <sup>
<italic>G</italic>
</sup>
<italic>T</italic>
<sub>
<italic>S</italic>
</sub>, we can then convert the displayed calibration points into the gaze coordinate system, normalizing them in the process. The calculated normalized gaze vectors obtained through this process are then again used as input for the same regression problem, resulting in the same transformation matrix as initially derived, which concludes our validation. By converting the calibration points into the gaze coordinate system, we can assess the accuracy of the applied gaze estimation model by computing the angle between two vectors (Eq. <xref ref-type="disp-formula" rid="e28">28</xref>).<disp-formula id="e28">
<mml:math id="m68">
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">deg</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>cos</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>g</mml:mi>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(28)</label>
</disp-formula>
<sup>
<italic>G</italic>
</sup>
<italic>g</italic> is the gaze vector provided by the gaze estimation model and <sup>
<italic>G</italic>
</sup>
<inline-formula id="inf41">
<mml:math id="m69">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>g</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> is the point displayed on the screen transformed into the gaze coordinate system. For this specific trial we get an average error of the four calibration points of 3.3&#xb0;, which is in this case lower than the reporter 6.95&#xb0;. However, it is worth noting that the error varies across the field of view, as illustrated in <xref ref-type="fig" rid="F3">Figure 3</xref>. The same holds for the ETH-XGaze dataset, as illustrated in Figure 8 in the paper of <xref ref-type="bibr" rid="B32">Zhang et al. (2020)</xref>.</p>
</sec>
<sec id="s4-2">
<title>4.2 Comparison of gaze tracking software</title>
<p>In this section, we conducted a comparative analysis of our method with two established gaze tracking solutions: the Tobii Eye Tracker 5 (<xref ref-type="bibr" rid="B27">Tobii, 2023</xref>) and GazeRecorder GazeRecorder (accessed 2023), which is free for non-commercial use, however the used methods and gaze tracking procedure is not available. The reported accuracy on their website is 1&#xb0;, while <xref ref-type="bibr" rid="B12">Heck et al. (2023)</xref> state an accuracy of 1.43&#xb0; in their review paper. GazeRecorder stands out as a webcam-based gaze tracking system for computer screens, as reported by <xref ref-type="bibr" rid="B12">Heck et al. (2023)</xref>. The system is the most accurate among webcam-based gaze tracking software. Although, it is important to note that GazeRecorder necessitates a comprehensive calibration process, involving around 30 calibration points.</p>
<p>The Tobii Eye Tracker 5 is purpose-built for tracking user gaze on a computer screen. This hardware is discreetly mounted beneath the screen and relies on a sophisticated combination of infrared cameras and stereo imaging to precisely monitor the user&#x2019;s gaze. The precise details of its proprietary technique are kept confidential as a closely held company secret.</p>
<p>In this evaluation, we pitted our method against the high-end Tobii hardware and GazeRecorder.</p>
<sec id="s4-2-1">
<title>4.2.1 Comparison with no head movements</title>
<p>For this comparison we use the same trial as described in the previous <xref ref-type="sec" rid="s4">Section 4</xref>. Given that both the GazeRecorder software and our method rely on the webcam, we had to conduct an additional trial for GazeRecorder, as the webcam cannot be simultaneously used by both applications. In the experiment, the user was positioned in front of a computer screen at a distance of 800 <italic>mm</italic>. The screen had a resolution of 2560 &#xd7; 1440 <italic>pixels</italic> and physical dimensions of 597 <italic>mm</italic> &#xd7; 336 <italic>mm</italic>.</p>
<p>To ensure accuracy, we synchronized the recording of both webcam-based gaze tracking solutions with data from the Tobii Eye Tracker 5. In <xref ref-type="fig" rid="F4">Figure 4</xref>, the trajectory of the presented target is depicted in blue, serving as the ground truth. The trajectory recorded by the Tobii Eye Tracker 5 is represented in orange, while our method in combination with the OpenVino model is shown in green and our method in combination with ETH-XGaze dataset is shown in black. Both the OpenVino model and the model trained on the ETH-XGaze dataset yield distinct outputs, both of which serve as inputs in our method for projecting the unit gaze vector onto the screen. This highlights the critical importance of an accurate unit gaze vector for our method. Despite their differences, the overall error along this exemplary trajectory between the two models is comparable (<xref ref-type="table" rid="T1">Table 1</xref>). It is worth noting that the number of samples collected by the Tobii Eye Tracker 5 surpasses that of the webcam based solutions.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>The proposed method is here compared with the Tobii Eye Tracker 5.</p>
</caption>
<graphic xlink:href="frobt-11-1369566-g004.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>The table presents the RMSE for the three gaze tracking solutions along a predefined trajectory without head movements.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left"/>
<th rowspan="2" align="left">Tobii eye tracker 5</th>
<th rowspan="2" align="left">GazeRecorder</th>
<th align="center">Proposed method</th>
<th align="center">Proposed method</th>
</tr>
<tr>
<th align="center">OpenVino</th>
<th align="center">ETX-Gaze</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">RMSE</td>
<td align="left">15 <italic>mm</italic>/0.9&#xb0;</td>
<td align="left">55 <italic>mm</italic>/2.65&#xb0;</td>
<td align="center">53 <italic>mm</italic>/3.3&#xb0;</td>
<td align="center">50 <italic>mm</italic>/3.2&#xb0;</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In <xref ref-type="fig" rid="F5">Figure 5</xref>, we present a comparison between GazeRecorder, Tobii Eye Tracker 5, and the ground truth. The color scheme in this figure corresponds to that in <xref ref-type="fig" rid="F4">Figure 4</xref> with the green trajectory for GazeRecorder. Notably, GazeRecorder exhibits a lower sampling rate compared to the Tobii Eye Tracker 5. This lower sampling rate can be attributed to the webcam&#x2019;s lower frame rate, which affects the rate at which data is captured.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>The GazeRecorder is here compared with the Tobii Eye Tracker 5.</p>
</caption>
<graphic xlink:href="frobt-11-1369566-g005.tif"/>
</fig>
<p>Furthermore, we calculated the Root Mean Squared Error (RMSE) among the three gaze tracking solutions for this trial. The RMSE was computed for the trajectories shown in both <xref ref-type="fig" rid="F4">Figures 4</xref>, <xref ref-type="fig" rid="F5">5</xref>. The computed RMSE values are presented in <xref ref-type="table" rid="T1">Table 1</xref>. The RMSE in millimeters represents the discrepancy between the ground truth on the screen and the gaze point provided by each gaze tracking system, at an 800 <italic>mm</italic> screen-user distance. Additionally, the RMSE in degrees measures the angular error between the ground truth gaze vector and the corresponding vector delivered by each of the three gaze tracking systems. It is evident that comparable results were achieved between GazeRecorder, which stands out as one of the most accurate webcam-based gaze tracking solutions, and our proposed method. However, it is essential to note that these solutions still cannot rival the precision of the highly sophisticated Tobii Eye Tracker 5 solution.</p>
<p>For this specific trial, we did not attain the accuracy reported by GazeRecorder and <xref ref-type="bibr" rid="B12">Heck et al. (2023)</xref>. This deviation could be attributed to the experiments not being conducted in a controlled laboratory environment, with not ideal lightning conditions, potential influences from the user&#x2019;s attention and gaze tracking behavior when following a trajectory. However, it is essential to highlight that, when computing the RMSE in degrees for this trial, our method yields similar results to GazeRecorder. This serves as an additional indication that our approach to compute a transformation matrix from the gaze coordinate system to the screen is effective, enabling the precise projection of a unit gaze vector onto the screen.</p>
</sec>
<sec id="s4-2-2">
<title>4.2.2 Head movements</title>
<p>In the context of head movements, we differentiate between two types: head movements where the user simply turns their head, corresponding to yaw and pitch movements and head movements where the user moves their head from left to right, or up and down.</p>
<p>For the next experiment, a point is presented at the center of the screen. The user is instructed to focus their gaze on this point while performing head movements: turning their head once to the left, once to the right, and once up and down. This essentially involves yaw and pitch movements. The movements were in the range &#xb1;30&#xb0; for yaw movements and &#xb1;25&#xb0; for pitch movements. The results are presented in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>The Root Mean Squared Error (RMSE) is calculated to compare the performance of the three gaze tracking solutions.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left">Tobii eye tracker 5</th>
<th align="left">GazeRecorder</th>
<th align="left">Proposed method</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">RMSE</td>
<td align="left">10 <italic>mm</italic>/0.5&#xb0;</td>
<td align="left">80 <italic>mm</italic>/4.1&#xb0;</td>
<td align="left">80 <italic>mm</italic>/5.1&#xb0;</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>In this particular experiment, a user maintains their gaze on a fixed point while turning their head.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>In the following experiment, the user is once more tasked fixing their gaze on a displayed point. However, this time, the user is instructed to refrain from moving their head, but rather, to shift their body to the left and then to the right, all while maintaining their focus on the point. The movement was in a range of &#xb1;100 <italic>mm</italic> form the initial position, which was in the middle of the screen at a distance of 800 <italic>mm</italic>. The outcome of this experiment is detailed in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>The Root Mean Squared Error (RMSE) is calculated to compare the performance of the three gaze tracking solutions.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left">Tobii eye tracker 5</th>
<th align="left">GazeRecorder</th>
<th align="left">Proposed method</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">RMSE</td>
<td align="left">10 <italic>mm</italic>/0.6&#xb0;</td>
<td align="left">30 <italic>mm</italic>/1.9&#xb0;</td>
<td align="left">60 <italic>mm</italic>/4&#xb0;</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>In this particular experiment, a user moves to the left and then to the right in front of the screen.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The accuracy and precision of our projection depends solely on the gaze vector estimation model used. For instance, the Mean Absolute Error (MAE) of OpenVino is 6.95&#xb0;, with a standard deviation of 3.58&#xb0;. Accuracy and precision metrics for models trained on the ETH-XGaze dataset are available on the ETH-XGaze competition website CodaLab (accessed 2023). These initial experiments, though conducted on only one person, already indicate that 2D webcam-based solutions, including GazeRecorder, fail to effectively accommodate precise head movements. These experiments merely serve to highlight the persistent limitations associated with head movements when using 2D webcam-based gaze tracking. This limitation, also recognized by <xref ref-type="bibr" rid="B12">Heck et al. (2023)</xref>, becomes evident after just one trial. Specifically, lateral movements, such as those to the left or right, are not adequately addressed in any of the trained models. For example, GazeRecorder restricts horizontal head movements by positioning the user centrally and issuing alerts for excessive shifts. In the subsequent section, we delve into the challenges associated with lateral movements, illustrate the issue using the OpenVino model, and propose a solution to address these challenges using SfM.</p>
</sec>
<sec id="s4-2-3">
<title>4.2.3 Applying structure from motion</title>
<p>The proposed method compensates for horizontal head movements in which a user shifts left or right by utilizing SfM to track the user&#x2019;s movement. To achieve this, we apply the model introduced in <xref ref-type="sec" rid="s3-3-5">Section 3.3.5</xref>. In our first experiment using this model, we replicate the conditions detailed in <xref ref-type="sec" rid="s4-2-1">Section 4.2.1</xref>, where the user is instructed to follow a target trajectory while keeping their head stationary. The results of this experiment align with those presented in <xref ref-type="table" rid="T1">Table 1</xref>, showcasing similar outcomes and reinforcing the validity of the approach.</p>
<p>In the subsequent experiment, participants are instructed turning their head to the left and right (yaw-movement). Head movements can introduce subtle changes in the translation vector of <inline-formula id="inf42">
<mml:math id="m70">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> since the eyes are not precisely aligned with the head&#x2019;s axis of rotation. However, it is essential to note that the OpenVino CNN model already compensates for these head movements. Therefore, both yaw and pitch rotations need to be accounted for in the translation vector of <inline-formula id="inf43">
<mml:math id="m71">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula>.</p>
<p>Furthermore, when a user moves to the left or right without actually performing a yaw head movement, the 2D camera perspective makes it appear as if a yaw movement occurred. This is because the user is filmed from the side, causing the linear movement to mimic a yaw movement from the CNN model&#x2019;s perspective. These particular movements are not considered in the CNN model&#x2019;s calculations.</p>
<p>With the proposed method, which determines the actual location of a person in front of the computer screen, we are able to illustrate the mentioned two scenarios in <xref ref-type="fig" rid="F6">Figure 6</xref>. One movement features a pure yaw movement and another involves lateral head movements to the left and right without head rotation. The visual representation demonstrates that a pure yaw movement leads to a translation in the <italic>x</italic>-direction of the <inline-formula id="inf44">
<mml:math id="m72">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula> translation vector, as indicated in blue. However, when a user moves laterally, a significant change occurs in the <italic>x</italic>-direction of the translation vector, along with a shift in the yaw angle. This change in the yaw angle happens even when the user does not physically turn their head, a phenomenon attributed to the side-view camera perspective, as previously explained.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>The initial segment represents a yaw movement, while the subsequent segment illustrates a pure lateral shift. The left axis corresponds to the translation vector&#x2019;s <italic>x</italic>-direction in <inline-formula id="inf45">
<mml:math id="m73">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula>, while the right axis denotes the yaw angle in degrees.</p>
</caption>
<graphic xlink:href="frobt-11-1369566-g006.tif"/>
</fig>
<p>To accurately distinguish between these movements, a CNN model should undergo specialized training. This training process should include an additional input: the user&#x2019;s position in front of the screen, determined through SfM techniques. With existing eye gaze datasets like (<xref ref-type="bibr" rid="B32">Zhang et al., 2020</xref> or <xref ref-type="bibr" rid="B29">Tonsen et al., 2016</xref>) we propose to account for lateral head movements and improve webcam based gaze tracking.</p>
</sec>
</sec>
</sec>
<sec id="s5">
<title>5 Conclusion and future work</title>
<p>In this study, we presented a novel method for webcam based gaze estimation on a computer screen requiring only four calibration points. Our focus was not on evaluating the precision of any appearance based CNN gaze estimation model but on demonstrating our proposed method on a single user. We presented a methodology to project a gaze vector onto a screen with an unknown distance. The proposed regression model determined a transformation matrix, <inline-formula id="inf46">
<mml:math id="m74">
<mml:mmultiscripts>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:none/>
<mml:mprescripts/>
<mml:none/>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:mmultiscripts>
</mml:math>
</inline-formula>, allowing the conversion of the gaze vector from the gaze coordinate system to the screen coordinate system.</p>
<p>Physical validation of the user&#x2019;s position in front of the screen confirmed the soundness of our approach, representing a novel method for determining the physical distance and location of a user without stereo imaging. The method&#x2019;s accuracy is contingent on the precision of the gaze vector model, here the OpenVINO model and a model trained on the ETH-XGaze dataset.</p>
<p>Comparisons with established gaze-tracking solutions, Tobii Eye Tracker 5 and GazeRecorder, showcased comparable results, indicating the potential efficacy of our method. Further experiments addressed head movements, highlighting the limitations of 2D webcam-based solutions and proposing compensation using SfM techniques.</p>
<p>Future work should focus on refining the proposed method to enhance accuracy, especially in uncontrolled environments. Specialized training of CNN models, incorporating user position via SfM, should be explored using data sets like (<xref ref-type="bibr" rid="B32">Zhang et al., 2020</xref> or <xref ref-type="bibr" rid="B29">Tonsen et al., 2016</xref>). The CNN trained with these adjustments should be able to effectively differentiate between lateral movements to the left or right and specific head movements, such as yaw and pitch rotations.</p>
<p>Moreover, there is potential for enhancing sensitivity to cope with varying lighting conditions. The introduction of supplementary filtering mechanisms may aid in bolstering accuracy, particularly in challenging lighting scenarios.</p>
<p>Generally, webcam-based gaze tracking solutions, exhibit a notable sensitivity to varying lighting conditions. It is essential to underscore that the overall accuracy of our proposed method is intrinsically tied to the precision of the raw gaze vector generated by the gaze estimation model. In this study we have conclusively demonstrated that webcam-based solutions, while promising, still cannot attain the level of accuracy achieved by sophisticated gaze tracking hardware, which frequently leverages advanced technologies such as stereo vision and infrared cameras. Consequently, the pursuit of further research endeavors is imperative in order to bridge the existing gap and elevate webcam-based solutions to a level of accuracy comparable to that of purpose-built eye tracking hardware.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>Publicly available gaze estimation models were analyzed in this study. The implementation of the concept presented in this paper can be found here: <ext-link ext-link-type="uri" xlink:href="https://github.com/FalchLucas/WebCamGazeEstimation">https://github.com/FalchLucas/WebCamGazeEstimation</ext-link>.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>LF: Conceptualization, Formal Analysis, Investigation, Methodology, Software, Validation, Writing&#x2013;original draft, Writing&#x2013;review and editing. KL: Funding acquisition, Project administration, Resources, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The authors declare that financial support was received for the research, authorship, and/or publication of this article. This project received funding from Innosuisse&#x2014;the Swiss Innovation Agency, under the grant number 100.440 IP-ICT, for the development of Innovative Visualization Tools for Big Battery Data.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arar</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Thiran</surname>
<given-names>J.-P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>A regression-based user calibration framework for real-time gaze estimation</article-title>. <source>IEEE Trans. Circuits Syst. Video Technol.</source> <volume>26</volume>, <fpage>2069</fpage>&#x2013;<lpage>2082</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Audun</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>clmtrackr: javascript library for precise tracking of facial features via constrained local models</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://github.com/auduno/clmtrackr">https://github.com/auduno/clmtrackr</ext-link>.</comment>
</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <source>Gaze estimation with an ensemble of four architectures</source>. <comment>
<italic>arXiv preprint arXiv:2107</italic>.<italic>01980</italic>
</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.2107.01980</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2008</year>). &#x201c;<article-title>3d gaze estimation with a single camera without ir illumination</article-title>,&#x201d; in <conf-name>2008 19th International Conference on Pattern Recognition</conf-name>, <fpage>1</fpage>&#x2013;<lpage>4</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chi</surname>
<given-names>J. N.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>Y. T.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>Eye gaze calculation based on nonlinear polynomial and generalized regression neural network</article-title>,&#x201d; in <conf-name>5th International Conference on Natural Computation, ICNC 2009</conf-name>, <fpage>617</fpage>&#x2013;<lpage>623</lpage>. <pub-id pub-id-type="doi">10.1109/ICNC.2009.599</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<collab>CodaLab</collab> (<year>2023</year>). <article-title>CodaLab - ETH-XGaze competition</article-title>
</citation>
</ref>
<ref id="B7">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Falch</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Gaze estimation</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://github.com/FalchLucas/GazeEstimation">https://github.com/FalchLucas/GazeEstimation</ext-link>.</comment>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ferhat</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Vilari&#xf1;o</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Low cost eye tracking: the current panorama</article-title>. <source>Comput. Intell. Neurosci.</source> <volume>2016</volume>, <fpage>1</fpage>, <lpage>14</lpage>. <pub-id pub-id-type="doi">10.1155/2016/8680541</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="web">
<collab>GazeRecorder</collab> (<year>2023</year>). <article-title>Gazepointer - real-time gaze tracker for hands-free mouse cursor control</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://gazerecorder.com/gazepointer/">https://gazerecorder.com/gazepointer/</ext-link>.</comment>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hansen</surname>
<given-names>D. W.</given-names>
</name>
<name>
<surname>Pece</surname>
<given-names>A. E. C.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Eye tracking in the wild</article-title>. <source>Comput. Vis. Image Underst.</source> <volume>98</volume>, <fpage>155</fpage>&#x2013;<lpage>181</lpage>. <pub-id pub-id-type="doi">10.1016/j.cviu.2004.07.013</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hartley</surname>
<given-names>R. I.</given-names>
</name>
<name>
<surname>Zisserman</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2004</year>). <source>Multiple view geometry in computer vision</source>. <edition>second edn</edition>. <publisher-name>Cambridge University Press</publisher-name>. <comment>0521540518</comment>.</citation>
</ref>
<ref id="B12">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Heck</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Deutscher</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Becker</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Webcam eye tracking for desktop and mobile devices: a systematicreview</article-title>,&#x201d; in <conf-name>Proceedings of the 56th Hawaii International Conference on System Science</conf-name>.</citation>
</ref>
<ref id="B13">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Hennessey</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Noureddin</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Lawrence</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2006</year>). &#x201c;<article-title>A single camera eye-gaze tracking system with free head motion</article-title>,&#x201d; in <conf-name>Proceedings of the 2006 Symposium on Eye Tracking Research and Applications</conf-name>, <conf-loc>San Diego, CA</conf-loc> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>87</fpage>&#x2013;<lpage>94</lpage>. <pub-id pub-id-type="doi">10.1145/1117309.1117349</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="web">
<collab>hysts on github</collab> (<year>2023</year>). <article-title>A demo program of gaze estimation models (mpiigaze, mpiifacegaze, eth-xgaze)</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://github.com/hysts/pytorch_mpiigaze_demo?tab=readme-ov-file">https://github.com/hysts/pytorch_mpiigaze_demo?tab&#x3d;readme-ov-file</ext-link>.</comment>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<collab>Intel Corporation</collab> (<year>2023</year>). <article-title>Openvino model: facial landmarks 35 adas 0002</article-title>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Corcoran</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A review and analysis of eye-gaze estimation systems, algorithms and performance evaluation methods in consumer platforms</article-title>. <source>IEEE Access</source> <volume>5</volume>, <fpage>16495</fpage>&#x2013;<lpage>16519</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2017.2735633</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Krafka</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Khosla</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kellnhofer</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kannan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Bhandarkar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Matusik</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). &#x201c;<article-title>Eye tracking for everyone</article-title>,&#x201d; in <conf-name>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <conf-loc>Los Alamitos, CA, USA</conf-loc> (<publisher-name>IEEE Computer Society</publisher-name>), <fpage>2176</fpage>&#x2013;<lpage>2184</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2016.239</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Muhammad Usman Ghani</surname>
<given-names>M. M. N. G.</given-names>
</name>
<name>
<surname>Chaudhry</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sohail</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Geelani</surname>
<given-names>M. N.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Gazepointer: a real time mouse pointer control implementation based on eye gaze tracking</article-title>. <source>INMIC</source>, <fpage>154</fpage>&#x2013;<lpage>159</lpage>. <pub-id pub-id-type="doi">10.1109/inmic.2013.6731342</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<collab>NVIDIA</collab> (<year>2023</year>). <article-title>NVIDIA TAO toolkit - GazeNet model</article-title>
</citation>
</ref>
<ref id="B20">
<citation citation-type="web">
<collab>OpenCV</collab> (<year>2023</year>). <article-title>Camera calibration with opencv</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://docs.opencv.org/4.x/dc/dbb/tutorial_py_calibration.html">https://docs.opencv.org/4.x/dc/dbb/tutorial_py_calibration.html</ext-link>.</comment>
</citation>
</ref>
<ref id="B21">
<citation citation-type="web">
<collab>OpenVINO</collab> (<year>2023</year>). <article-title>Gaze estimation adas 0002 model</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://docs.openvino.ai/latest/omz_models_model_gaze_estimation_adas_0002.html">https://docs.openvino.ai/latest/omz_models_model_gaze_estimation_adas_0002.html</ext-link>.</comment>
</citation>
</ref>
<ref id="B22">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Papoutsaki</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sangkloy</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Laskey</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Daskalova</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hays</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Webgazer: scalable webcam eye tracking using user interactions</article-title>,&#x201d; in <conf-name>Proceedings of the Twenty-Fifth International Joint Conference on Artificial Intelligence</conf-name> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>AAAI Press</publisher-name>), <fpage>3839</fpage>&#x2013;<lpage>3845</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Sewell</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Komogortsev</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>Real-time eye gaze tracking with an unmodified commodity webcam employing a neural network</article-title>,&#x201d; in <conf-name>CHI EA &#x2019;10 CHI &#x2019;10 Extended Abstracts on Human Factors in Computing Systems</conf-name>, <conf-loc>Atlanta, GA</conf-loc> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>3739</fpage>&#x2013;<lpage>3744</lpage>. <pub-id pub-id-type="doi">10.1145/1753846.1754048</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sogo</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Gazeparser: an open-source and multiplatform library for low-cost eye tracking and analysis</article-title>. <source>Behav. Res. Methods</source> <volume>45</volume>, <fpage>684</fpage>&#x2013;<lpage>695</lpage>. <pub-id pub-id-type="doi">10.3758/s13428-012-0286-x</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Sugano</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Matsushita</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sato</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Learning-by-synthesis for appearance-based 3d gaze estimation</article-title>,&#x201d; in <conf-name>2014 IEEE Conference on Computer Vision and Pattern Recognition</conf-name>, <fpage>1821</fpage>&#x2013;<lpage>1828</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2014.235</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Tan</surname>
<given-names>K.-H.</given-names>
</name>
<name>
<surname>Kriegman</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ahuja</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2002</year>). &#x201c;<article-title>Appearance-based eye gaze estimation</article-title>,&#x201d; in <conf-name>Proceedings Sixth IEEE Workshop on Applications of Computer Vision, 2002</conf-name> (<publisher-name>WACV 2002</publisher-name>), <fpage>191</fpage>&#x2013;<lpage>195</lpage>. <pub-id pub-id-type="doi">10.1109/ACV.2002.1182180</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<collab>Tobii</collab> (<year>2023</year>). <article-title>Tobii eye tracker 5</article-title>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<collab>Tobii group</collab>. (<year>2023</year>). <article-title>Tobii group</article-title>
</citation>
</ref>
<ref id="B29">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Tonsen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Sugano</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bulling</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Labelled pupils in the wild: a dataset for studying pupil detection in unconstrained environments</article-title>,&#x201d; in <conf-name>ETRA &#x2019;16 Proceedings of the Ninth Biennial ACM Symposium on Eye Tracking Research and Applications</conf-name>, <conf-loc>Charleston, SC</conf-loc> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <fpage>139</fpage>&#x2013;<lpage>142</lpage>. <pub-id pub-id-type="doi">10.1145/2857491.2857520</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ehinger</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Finkelstein</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kulkarni</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Turkergaze: crowdsourcing saliency with webcam based eye tracking</article-title>. <comment>
<italic>ArXiv</italic> abs/1504.06755</comment>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yoo</surname>
<given-names>D. H.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>M. J.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>A no<italic>vel non</italic>-intrusive eye gaze estimation using cross-ratio under large head motion</article-title>. <source>Comput. Vis. Image Underst.</source> <volume>98</volume>, <fpage>25</fpage>&#x2013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1016/j.cviu.2004.07.011</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Beeler</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Bradley</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hilliges</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Eth-xgaze: a large scale dataset for gaze estimation under extreme head pose and gaze variation</article-title>,&#x201d; in <conf-name>European Conference on Computer Vision (ECCV)</conf-name>, <conf-loc>Cham</conf-loc>. Editors <person-group person-group-type="editor">
<name>
<surname>Vedaldi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bischof</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Brox</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Frahm</surname>
<given-names>J.-M.</given-names>
</name>
</person-group> (<publisher-name>Springer</publisher-name>). <comment>of Lecture Notes in Computer Science</comment>, <fpage>12350</fpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-58558-7_22</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Sugano</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bulling</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Revisiting data normalization for appearance-based gaze estimation</article-title>,&#x201d; in <conf-name>ETRA &#x2019;18 Proceedings of the 2018 ACM Symposium on Eye Tracking Research and Applications</conf-name>, <conf-loc>Warsaw, Poland</conf-loc> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>). <pub-id pub-id-type="doi">10.1145/3204493.3204548</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Sugano</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fritz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bulling</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Appearance-based gaze estimation in the wild</article-title>,&#x201d; in <conf-name>2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <fpage>4511</fpage>&#x2013;<lpage>4520</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2015.7299081</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Sugano</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fritz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bulling</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>It&#x2019;s written all over your face: full-face appearance-based gaze estimation</article-title>,&#x201d; in <conf-name>2017 IEEE Conference on Computer Vision and Pattern Recognition Workshops</conf-name> (<publisher-name>CVPRW</publisher-name>), <fpage>2299</fpage>&#x2013;<lpage>2308</lpage>. <pub-id pub-id-type="doi">10.1109/CVPRW.2017.284</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2005</year>). &#x201c;<article-title>Eye gaze tracking under natural head movements</article-title>,&#x201d; in <conf-name>2005 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR&#x2019;05) 1</conf-name>, <fpage>918</fpage>&#x2013;<lpage>923</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>