﻿<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.0 20120330//EN" "http://jats.nlm.nih.gov/publishing/1.0/JATS-journalpublishing1.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-id journal-id-type="nlm-ta">Intell. Robot.</journal-id>
      <journal-id journal-id-type="publisher-id">IR</journal-id>
      <journal-title-group>
        <journal-title>Intelligence &amp; Robotics</journal-title>
      </journal-title-group>
      <issn pub-type="epub">2770-3541</issn>
      <publisher>
        <publisher-name>OAE Publishing Inc.</publisher-name>
      </publisher>
    </journal-meta>
    <article-meta>
	<article-id>IR-2026-041701</article-id>
      <article-id pub-id-type="doi">10.20517/ir.2026.29</article-id>
      <article-categories>
        <subj-group>
          <subject>Research Article</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Unlocking the head: unleashing deep learning and depth camera for free head movement gaze estimation</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <name>
            <surname>Wang</surname>
            <given-names>Zihao</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
          <xref ref-type="aff" rid="I2">
            <sup>2</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Wang</surname>
            <given-names>Jiangtao</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
          <xref ref-type="aff" rid="I2">
            <sup>2</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Ng</surname>
            <given-names>Anna Ching Mei</given-names>
          </name>
          <xref ref-type="aff" rid="I3">
            <sup>3</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Ng</surname>
            <given-names>Tit</given-names>
          </name>
          <xref ref-type="aff" rid="I3">
            <sup>3</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Qian</surname>
            <given-names>Wei</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
          <xref ref-type="aff" rid="I2">
            <sup>2</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Elirehema</surname>
            <given-names>Mbazingwa</given-names>
          </name>
          <xref ref-type="aff" rid="I4">
            <sup>4</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" corresp="yes">
          <name>
            <surname>Monkam</surname>
            <given-names>Patrice</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
          <xref ref-type="aff" rid="I2">
            <sup>2</sup>
          </xref>
          <xref ref-type="corresp" rid="cor1" />
          <contrib-id contrib-id-type="orcid">https://orcid.org/0000-0003-3805-9457</contrib-id>
        </contrib>
        <contrib contrib-type="author" corresp="yes">
          <name>
            <surname>Qi</surname>
            <given-names>Shouliang</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
          <xref ref-type="aff" rid="I2">
            <sup>2</sup>
          </xref>
          <xref ref-type="corresp" rid="cor1" />
          <contrib-id contrib-id-type="orcid">https://orcid.org/0000-0003-0977-1939</contrib-id>
        </contrib>
      </contrib-group>
      <aff id="I1">
        <sup>1</sup>College of Medicine and Biological Information Engineering, Northeastern University, Shenyang 110819, Liaoning, China.</aff>
      <aff id="I2">
        <sup>2</sup>Key Laboratory of Intelligent Computing in Medical Image, Ministry of Education, Northeastern University, Shenyang 110169, Liaoning, China.</aff>
      <aff id="I3">
        <sup>3</sup>Shenzhen Jingmei Health Technology Co., Ltd., Shenzhen 518000, Guangdong, China.</aff>
      <aff id="I4">
        <sup>4</sup>Department of Electronics and Telecommunication Engineering, Dar es Salaam Institute of Technology, Dar es Salaam 2958, Tanzania.</aff>
      <author-notes>
        <corresp id="cor1">Correspondence to: Dr. Patrice Monkam, Prof. Shouliang Qi, College of Medicine and Biological Information Engineering, Northeastern University, Shenyang 110819, Liaoning, China. E-mail: <email>patrice123china1@gmail.com</email>; <email>qisl@bmie.neu.edu.cn</email></corresp>
        <fn fn-type="other">
          <p>
            <bold>Received:</bold> 17 Apr 2026 |  <bold>First Decision:</bold> 8 Jul 2026 | <bold>Revised:</bold> 28 Jul 2026 | <bold>Accepted:</bold> 1 Sep 2026 | <bold>Published:</bold> 23 Sep 2026</p>
        </fn>
        <fn fn-type="other">
          <p>
            <bold>Academic Editor:</bold> Jinhai Li | <bold>Copy Editor:</bold> Pei-Yun Wang | <bold>Production Editor:</bold> Pei-Yun Wang</p>
        </fn>
      </author-notes>
      <pub-date pub-type="ppub">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>23</day>
        <month>9</month>
        <year>2026</year>
      </pub-date>
      <volume>6</volume>
	  <issue>3</issue>
      <fpage>621</fpage>
	  <lpage>42</lpage>
      <permissions>
        <copyright-statement>© The Author(s) 2026.</copyright-statement>
        <license xlink:href="https://creativecommons.org/licenses/by/4.0/">
          <license-p>© The Author(s) 2026. <bold>Open Access</bold> This article is licensed under a Creative Commons Attribution 4.0 International License (<uri xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</uri>), which permits unrestricted use, sharing, adaptation, distribution and reproduction in any medium or format, for any purpose, even commercially, as long as you give appropriate credit to the original author(s) and the source, provide a link to the Creative Commons license, and indicate if changes were made.</license-p>
        </license>
      </permissions>
      <abstract>
        <p>Gaze estimation has applications such as visual attention analysis and human-computer interaction. However, performance under free head movement conditions can be further enhanced. This study aims to improve the accuracy and robustness of appearance-based gaze estimation without head fixation. We develop a gaze-tracking system using a consumer depth camera and a deep-learning-based gaze estimation model [Gaze-Point-Net (GPN)]. A customized YOLOv5-based detector is developed to simultaneously localize the eyes and mouth, providing reliable facial landmarks for gaze estimation. The detected regions and depth image were used to calculate a vector of location and posture. In GPN, a double-channel convolutional neural network with a Squeeze-and-Excitation module extracts features from binocular images, and these features are concatenated with the location-and-posture vector to complete the multimodal gaze-position prediction task. Our GPN presents competitive performance in four experiments: (1) The error distribution in the sequential point test ranges from 4 to 10 degrees; (2) In the random point test, the calibrated and filtered GPN achieved a pixel error of 185.19 ± 116.57 pixels and an angular error of 4.52° ± 2.87°, demonstrating superior performance over counterpart methods; (3) In the trajectory tracking test, approximately 81.91% of gaze points were located within a 200-pixel tolerance radius of the target trajectory; (4) In the browsing test, the generated gaze heat and trajectory maps showed satisfactory results. The developed eye-tracking system, integrating a depth camera and deep learning models, demonstrates competitive performance and strong potential for several eye-tracking applications.</p>
      </abstract>
      <kwd-group>
        <kwd>Gaze estimation</kwd>
        <kwd>eye movement</kwd>
        <kwd>deep learning</kwd>
        <kwd>depth camera</kwd>
        <kwd>free head movement</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec id="sec1">
      <title>1. INTRODUCTION</title>
      <p>Gaze estimation is an important research direction in computer vision. It can utilize devices such as color cameras and infrared equipment to capture features such as faces and eyes, and then calculate the position that the user is looking at on a screen or in a natural environment. This facilitates interaction between the user and the device or can provide data support for operations or tests by collecting physiological information from the user. Gaze estimation technology has a wide range of applications<sup>[<xref ref-type="bibr" rid="B1">1</xref>,<xref ref-type="bibr" rid="B2">2</xref>]</sup>, and can be used in education<sup>[<xref ref-type="bibr" rid="B3">3</xref>,<xref ref-type="bibr" rid="B4">4</xref>]</sup>, scientific research<sup>[<xref ref-type="bibr" rid="B5">5</xref>,<xref ref-type="bibr" rid="B6">6</xref>]</sup>, medicine<sup>[<xref ref-type="bibr" rid="B7">7</xref>,<xref ref-type="bibr" rid="B8">8</xref>]</sup>, entertainment<sup>[<xref ref-type="bibr" rid="B9">9</xref>,<xref ref-type="bibr" rid="B10">10</xref>]</sup>, automatic driving<sup>[<xref ref-type="bibr" rid="B11">11</xref>,<xref ref-type="bibr" rid="B12">12</xref>]</sup>, business<sup>[<xref ref-type="bibr" rid="B13">13</xref>]</sup>, sport<sup>[<xref ref-type="bibr" rid="B14">14</xref>]</sup>, and so on. As shown in <xref ref-type="fig" rid="fig1">Figure 1</xref>, gaze estimation results can be represented by a gaze-trajectory plot and a gaze heat map.</p>
      <fig id="fig1" position="float" width="460">
        <label>Figure 1</label>
        <caption>
          <p>Illustration of gaze estimation results, including the predicted gaze trajectory (left) and gaze distribution heat map (right). These images are used solely for visualization and illustrative purposes. They were obtained from the Kodak Lossless True Color Image Suite, originally released by Eastman Kodak Company (1999); the publicly accessible mirror at <uri xlink:href="https://r0k.us/graphics/kodak/">https://r0k.us/graphics/kodak/</uri> (accessed: 22.2.2026) was used for retrieval.</p>
        </caption>
        <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="ir6029.fig.1.jpg" />
      </fig>
      <p>Gaze estimation systems primarily operate in the following three modes. (1) Head-mounted eye tracking. In this approach, gaze estimation is performed while wearing devices such as glasses or headsets. This method achieves high-precision gaze estimation, especially when equipped with gaze estimation hardware. As documented in<sup>[<xref ref-type="bibr" rid="B15">15</xref>,<xref ref-type="bibr" rid="B16">16</xref>]</sup>, this method has been refined. The advantage of this approach is that the device’s relative position to the head remains relatively stable, eliminating the need to account for variations in head movement. It allows for a focus on eye image information alone, resulting in higher accuracy within appearance-based methods; (2) Desktop-based gaze estimation. This method employs sampling devices, such as cameras, to collect information and utilizes model-based or appearance-based techniques for gaze estimation. It is primarily used to predict a user’s point of focus while using a computer or looking at a monitor<sup>[<xref ref-type="bibr" rid="B17">17</xref>,<xref ref-type="bibr" rid="B18">18</xref>]</sup>. Notably, the camera’s position is fixed relative to the monitor. It caters to scenarios where the user’s head position remains relatively stable or changes slightly. Given the possibility of head movement and rotations, the method must account for these variables during gaze estimation; (3) Pose-based gaze estimation. This approach is often applied in long-range prediction scenarios, where the camera is positioned at a significant distance from the subject. It relies on the subject’s overall body posture and head orientation to determine gaze direction and identify the object of interest. As shown in<sup>[<xref ref-type="bibr" rid="B19">19</xref>]</sup>, substantial progress has been made with this approach. This method involves numerous variables, and the task does not require extremely high precision. We used the second method, which builds a desktop line-of-sight tracking system.</p>
      <p>There are currently three common methods for gaze estimation. (1) 2D mapping method. These approaches involve personalized calibration and utilize a mapping model to estimate the focal point corresponding to new eye features<sup>[<xref ref-type="bibr" rid="B20">20</xref>,<xref ref-type="bibr" rid="B21">21</xref>]</sup>; (2) 3D model method. These methods employ eye structure and geometric imaging models to estimate the three-dimensional line of sight. It has higher requirements for cameras and lighting but offers greater flexibility in handling individual differences and head movements<sup>[<xref ref-type="bibr" rid="B22">22</xref>,<xref ref-type="bibr" rid="B23">23</xref>]</sup>; (3) Appearance-based method. The appearance-based method uses eye or facial images as input and determines gaze position through the training of a mapping model<sup>[<xref ref-type="bibr" rid="B24">24</xref>,<xref ref-type="bibr" rid="B25">25</xref>]</sup>. It is robust because it uses a large number of statistical samples, does not require specific eye features to be extracted, and requires less equipment. However, adapting it to individual activities remains challenging. It offers simplicity, efficiency, high accuracy, a small number of model parameters, and low computational requirements. Because of its minimal hardware requirements, it is suitable for deployment in scenarios involving lightweight imaging devices. Therefore, appearance-based methods have become a mainstream approach for gaze estimation.</p>
      <p>With advances in deep learning for computer vision, neural networks have shown state-of-the-art performance in gaze estimation. Appearance-based gaze estimation methods use mathematical models and take inputs such as binocular images to infer gaze direction. Consequently, neural network models in deep learning are widely applied in appearance-based gaze estimation methods. LeNet and VGG models are widely used in gaze estimation tasks<sup>[<xref ref-type="bibr" rid="B2">2</xref>]</sup>.</p>
      <p>Appearance-based gaze estimation methods have been widely used on home computers<sup>[<xref ref-type="bibr" rid="B2">2</xref>]</sup>. When the subject’s head is significantly tilted, or farther away, the system may have difficulty accurately detecting eye position and posture, resulting in inaccurate or even failed gaze estimation. Appearance-based gaze estimation methods use traditional cameras for sampling, which is largely influenced by environmental lighting conditions. This article aims to address these issues to some extent.</p>
      <p>In this study, we develop a desktop-based gaze estimation system capable of accurate gaze estimation under free head movement and varying lighting conditions. The main contributions are summarized as follows: (1) A desktop-based gaze estimation system is developed to support a wide range of gaze-related research through flexible experimental configurations while providing reliable gaze estimation; (2) A curated gaze estimation dataset is collected and compiled to support system development and comprehensive evaluation, with plans to make the dataset available following completion of the necessary data-sharing procedures and receipt of the required permissions; (3) Unlike conventional approaches that first estimate the gaze direction in the camera coordinate system and subsequently transform it into the world coordinate system to determine the screen gaze point<sup>[<xref ref-type="bibr" rid="B26">26</xref>]</sup>, the proposed system directly regresses the gaze position on the screen, simplifying the estimation pipeline; (4) Depth information is incorporated to estimate the user’s spatial position and head pose, enabling robust gaze estimation under unrestricted head movement.</p>
    </sec>
    <sec id="sec2">
      <title>2. MATERIALS AND METHODS</title>
      <sec id="sec2-1">
        <title>2.1. Overview of the unlocking-the-head gaze estimation system</title>
        <p>The procedure of our gaze estimation system is as follows [<xref ref-type="fig" rid="fig2">Figure 2A</xref>]: Firstly, we employ an improved You Only Look Once (YOLO) model<sup>[<xref ref-type="bibr" rid="B27">27</xref>]</sup> to locate and crop eye images. We utilize an Intel RealSense depth camera<sup>[<xref ref-type="bibr" rid="B28">28</xref>]</sup> to calculate the three-dimensional spatial positions of the eyes and mouth. Next, we utilize a multimodal gaze point network to predict screen gaze points. To accomplish this, we need to create and collect two datasets: one for training the YOLO object detection model and another for training the Gaze-Point-Net (GPN) gaze regression model. In our implementation, the resolution of the feature maps is increased in the detection head to improve the recognition of small-sized objects (eyes and mouth regions in the facial images). Moreover, the left-eye and right-eye classes are merged into a single “eye” class, since the two eyes exhibit highly similar appearance characteristics. The eyes and mouth are detected simultaneously during facial landmark detection. While the eye images are used in the subsequent gaze estimation process, the detected mouth position provides an additional facial landmark for characterizing the facial geometry. Combined with the binocular eye positions captured by the Intel RealSense depth camera, the three landmarks (left eye, right eye, and mouth) define a facial plane whose normal vector provides two angular degrees of freedom corresponding to head pitch and yaw in the camera coordinate system. The 3D spatial coordinates of the left and right eyes, together with these two angular components, constitute the 8-dimensional position/posture vector used in this study. Roll is not explicitly represented as an independent angular variable but is implicitly encoded in the relative 3D spatial configuration of the two eyes.</p>
        <fig id="fig2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Overview of the unlocking-the-head gaze estimation system. (A) The deployment of the system; (B) The training of the universal GPN model; (C) The calibration of the GPN model. GPN: Gaze-Point-Net; RGB: Red, Green, Blue; YOLO: You Only Look Once.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="ir6029.fig.2.jpg" />
        </fig>
        <p>
          <xref ref-type="fig" rid="fig2">Figure 2B</xref> illustrates the training process of GPN. We collected training data from volunteers, consisting of standardized binocular images and corresponding three-dimensional information computed from image depth. These datasets were assembled and utilized to train a universal model.</p>
        <p>We trained our model on a large dataset, and to further improve its fit to each user, we designed a calibration function [<xref ref-type="fig" rid="fig2">Figure 2C</xref>]. With this function, we can enhance the accuracy of the model’s predictions and deploy the same model on screens of different sizes to maintain consistent task accuracy when the camera position may not be stable.</p>
      </sec>
      <sec id="sec2-2">
        <title>2.2. Construction and training of the GPN model</title>
        <p>Traditional gaze estimation tasks are usually performed under the condition of a fixed head, so traditional methods do not need to consider the head’s spatial position and pose information. Since the six degrees of freedom of the head are fixed, neural network models in traditional methods accept only binocular images or use a monocular model and compute binocular images through a mirror-flipping method.</p>
        <p>To address the asymmetry between pupil direction and eyelid features in binocular images, we designed GPN based on LeNet, similar to existing studies<sup>[<xref ref-type="bibr" rid="B2">2</xref>]</sup>. Since binocular images are different, the feature extractors cannot use the same parameters for feature extraction. Therefore, we adopted a double-channel convolutional neural network feature extractor with non-shared parameters to extract features from binocular images and used multiple fully connected layers to compress and extract features. Additionally, we connected (concatenated) head spatial position and pose vectors in the hidden layer to construct a multimodal neural network for completing the multimodal gaze position prediction task. Below is the network structure of GPN [<xref ref-type="fig" rid="fig3">Figure 3</xref>].</p>
        <fig id="fig3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>GPN structure. The 8-dimensional position/posture representation comprises the 3D spatial coordinates of the left and right eyes (6 dimensions) and two angular components of the facial-plane normal representing head pitch and yaw (2 dimensions). GPN: Gaze-Point-Net; FC: fully connected.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="ir6029.fig.3.jpg" />
        </fig>
        <p>Our neural network architecture mainly consists of two parts: the SELayer<sup>[<xref ref-type="bibr" rid="B29">29</xref>]</sup> and GAZE attention. The SELayer is a module used to enhance the neural network’s feature expression ability. It takes the input feature map and performs adaptive average pooling to obtain a global feature vector. After processing through two fully connected layers and an activation function, the SELayer generates a weight vector used to scale the input feature map. The SELayer includes an adaptive average pooling layer, two fully connected layers, a rectified linear unit (ReLU) activation function to construct nonlinear combinations, and a Sigmoid activation function to limit the parameter range.</p>
        <p>The GAZE attention is a neural network consisting of convolutional and fully connected layers. Its input includes two monochrome eye images (36 × 60 pixels each) and head position and pose information. The output is a 2D tensor representing the gaze point coordinates.</p>
        <p>The convolutional layers of this neural network adopt a non-shared-parameter dual-channel convolutional neural network structure. Each channel contains two convolutional modules. The first convolutional module consists of 20 convolutional kernels with a size of 5 × 5, a stride of 1, and a padding of 2 to keep the feature size the same. This is followed by a batch normalization layer, a ReLU activation function, and a 2 × 2 max-pooling layer. After the first pooling operation, the attention mechanism introduced earlier enhances the network’s focus on the input data. This mechanism adaptively weights each channel, allowing the neural network to focus more on extracting important features while ignoring irrelevant ones. The second convolutional module consists of 50 convolutional kernels with a size of 5 × 5, a stride of 1, and a padding of 2 × 2 to ensure that the feature size remains the same. This is followed by a batch normalization layer, a ReLU activation function, and a maximum pooling layer with a size of 2.</p>
      </sec>
      <sec id="sec2-3">
        <title>2.3. Training of the augmented GPN</title>
        <p>During training, we found that the model underestimated the y-coordinate of the predicted gaze point. Therefore, we redesigned the loss function. Our data and labels are both represented as two-dimensional coordinates, denoted by <italic>c</italic>1 = (<italic>x</italic>1, <italic>y</italic>1) and <italic>c</italic>2 = (<italic>x</italic>2, <italic>y</italic>2), respectively, where <italic>c</italic>1 and <italic>c</italic>2 represent the predicted and ground-truth gaze points. Here, <italic>x</italic>1 and <italic>y</italic>1 denote the predicted horizontal and vertical coordinates, while <italic>x</italic>2 and <italic>y</italic>2 denote the corresponding ground-truth coordinates. The proposed loss function consists of the absolute errors of the x- and y-coordinates, the squared error of the y-coordinate, and the Euclidean distance between the predicted and ground-truth gaze points, weighted by 2:2:1:4, respectively [Equation (1)]. These weighting coefficients were selected empirically through preliminary experiments to balance the contributions of the individual loss terms and promote stable model convergence. The squared y-coordinate error term further emphasizes deviations in the vertical coordinate, particularly at larger y-coordinate values.</p>
		<p><disp-formula> <label>(1)</label> <tex-math id="E1"> $$  L(c1,c2)=2|x1-x2|+2|y1-y2|+|y1-y2|^2+4||(x1,y1)-(x2,y2)||2 $$ </tex-math></disp-formula></p>
        <p>A total of 20 participants were randomly divided at the participant level into training, validation, and test sets at an 8:1:1 ratio, comprising 16, 2, and 2 participants, respectively. All samples from each participant were assigned exclusively to the corresponding subset, ensuring that no participant appeared in more than one subset.</p>
        <p>During training, we set the number of epochs to 50, the batch size to 16, and used the redesigned loss function as the objective function. To optimize the objective function, we used the stochastic gradient descent (SGD) optimizer with an initial learning rate of 0.0001 and a dynamic learning rate strategy. Specifically, we gradually reduced the learning rate by a factor of 0.1 at epochs 20 and 30. After each training batch, we used the validation set to calculate the mean absolute error (MAE) and determine whether to update the best model.</p>
      </sec>
      <sec id="sec2-4">
        <title>2.4. Calibration and filtering process</title>
        <p>The GPN model was trained using only the data from participants in the training subset, while data from the validation and test subsets were reserved for their respective evaluation purposes. Therefore, we can further improve the model’s accuracy and performance under specific users’ usage conditions through calibration. To this end, we designed the following calibration steps, following a similar approach to that described in<sup>[<xref ref-type="bibr" rid="B30">30</xref>]</sup>.</p>
        <p>First, we collected calibration data using a method similar to that used in collecting training data. The calibration data were collected by presenting nine gray boxes arranged in two rectangles on the display screen in sequence. The collected data include standardized binocular images, head position and posture vectors, and corresponding label data. Next, the collected data were assembled into a batch of calibration datasets, and the model was fine-tuned using the same loss function and optimizer as in training. Importantly, model fine-tuning at this stage used exclusively the calibration data from volunteers included in the training set.</p>
        <p>During model calibration, to prevent overfitting at the nine calibration points, we employed an approach similar to transfer learning by adjusting only the parameters of the last fully connected layer during fine-tuning. This method enables us to further improve the model’s accuracy and performance under specific user conditions.</p>
        <p>Moreover, a mean filter is applied to the calibrated gaze-point predictions. Specifically, the current gaze location is estimated as the mean of the five most recent predicted gaze locations. The filtered and unfiltered predictions are then compared to evaluate the effect of filtering on gaze estimation performance.</p>
      </sec>
      <sec id="sec2-5">
        <title>2.5. Experimental methods and measurements</title>
        <sec id="sec2-5-1">
          <title>2.5.1. Experiments</title>
          <p>After the completion of the entire system’s construction, we designed the following four experiments with reference to the method in<sup>[<xref ref-type="bibr" rid="B31">31</xref>]</sup>: sequential and random point testing, trajectory tracking testing, and browsing mode analysis. These experiments comprehensively analyze the accuracy, robustness, and real-time performance of the system.</p>
          <p>In the sequential point location test, we utilized our system design to track a gray box moving in a zigzag pattern from the upper left corner of the screen. Real-time gaze prediction was achieved through the use of a gaze estimation system. With participants’ consent, we recorded binocular images, facial images, head position, head pose information, and normalized screen position information of the gray box for data analysis.</p>
          <p>Similarly, in the random point location test, we used a gray block as a fixation target that jumped randomly between specific position and remained fixed for three seconds at each positions. To eliminate the influence of saccadic eye movements, we recorded data after the block had been stationary for one second.</p>
          <p>The fixation task with a stationary target in these experiments closely resembled the primary task of current eye-tracking devices, allowing us to assess the eye-tracking system’s performance accurately.</p>
          <p>To evaluate the system’s ability to track moving targets, we designed a circular trajectory tracking test. A circular target trajectory with a radius of 500 pixels was drawn at the center of the screen, and a moving box served as the gaze target. Gaze points were predicted and collected throughout the test. To quantify tracking accuracy, confidence radii of 300 and 200 pixels relative to the target trajectory were used as tolerance thresholds for determining whether the predicted gaze points were valid tracks.</p>
          <p>Finally, we demonstrated the gaze estimation system using a browsing mode analysis task. We selected five images containing highly distinctive objects, each displayed in full screen for five seconds. Simultaneously, we captured the sequence of gaze positions of the viewers and visualized them as gaze point trajectories and heat maps to analyze and present the performance of our model. By combining the captured gaze points with the positions of the target objects in the images, we analyzed the reliability, real-time capability, and accuracy of our model.</p>
        </sec>
        <sec id="sec2-5-2">
          <title>2.5.2. Measurement</title>
          <p>After computing the normalized gaze coordinates, they are mapped back to the 1,920 × 1,080 screen coordinate system to obtain the final projected gaze location, referred to as the fixation point (yellow arrow in <xref ref-type="fig" rid="fig4">Figure 4</xref>). Simultaneously, the corresponding ground-truth location is retained as the centroid of the target square displayed during the experiment (orange arrow in <xref ref-type="fig" rid="fig4">Figure 4</xref>). The predicted and ground-truth screen coordinates are then used to calculate the Euclidean distance between the two points on the screen.</p>
          <fig id="fig4" position="float" width="500">
            <label>Figure 4</label>
            <caption>
              <p>Camera coordinate system and gaze estimation. The image displayed on the screen was obtained from a publicly accessible online source and is included for illustrative purposes only. It is not part of the experimental dataset or quantitative evaluation.</p>
            </caption>
            <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="ir6029.fig.4.jpg" />
          </fig>
          <p>To further evaluate gaze estimation accuracy in three-dimensional space, the spatial coordinates of the eyes in the camera coordinate system, together with the known geometric relationship between the camera and the display, are used to determine the distances from the midpoint of the two eye rays to both the predicted and ground-truth fixation points. Based on these distances, the cosine theorem is applied to compute the angular error between the predicted and ground-truth gaze directions (red angle in <xref ref-type="fig" rid="fig4">Figure 4</xref>). The resulting angular error is subsequently analyzed under different experimental conditions.</p>
          <p>It is important to note that this study was approved by the Ethics Committee of Northeastern University (approval no. NEU-EC-2024B036S). All procedures were conducted in accordance with institutional guidelines and the Declaration of Helsinki (2024). Written informed consent was obtained from all participants prior to data collection. The experimental setup consisted of a 1,920 × 1,080 display and the Intel RealSense depth camera mounted directly below the screen. Data were collected from a single participant per session at a viewing distance ranging from 0.5 to 0.85 m.</p>
        </sec>
      </sec>
    </sec>
    <sec id="sec3">
      <title>3. RESULTS</title>
      <sec id="sec3-1">
        <title>3.1. Performance at sequential point test</title>
        <p>We conducted sequential point testing and collected 6,572 sets of experimental data. The results of this testing, along with error analysis at multiple scales, are presented in <xref ref-type="fig" rid="fig5">Figures 5</xref> and <xref ref-type="fig" rid="fig6">6</xref>. Moreover, the performance of the eyes-mouth detection model is evaluated.</p>
        <fig id="fig5" position="float">
          <label>Figure 5</label>
          <caption>
            <p>Performance analysis of GPN with different variables. (A) error analysis of the horizontal coordinates of the target points; (B) Error analysis of the vertical coordinates of the target points; (C) Error analysis of the horizontal and vertical coordinates of the target points; (D) Error analysis of the spatial horizontal location of volunteers’ heads; (E) Error analysis of the spatial vertical location of volunteers’ heads; (F) Error analysis of the spatial axis location of volunteers’ heads; (G) Error analysis of the pitch angle of volunteers’ heads; (H) Error analysis of the paw angle of volunteers’ heads. The central points indicate the mean values, and the blue shaded regions represent the corresponding variance. GPN: Gaze-Point-Net.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="ir6029.fig.5.jpg" />
        </fig>
        <fig id="fig6" position="float" width="550">
          <label>Figure 6</label>
          <caption>
            <p>Performance analysis of GPN with different light conditions. The central points indicate the mean values, and the blue shaded regions represent the corresponding variance. GPN: Gaze-Point-Net; RGB: Red, Green, Blue.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="ir6029.fig.6.jpg" />
        </fig>
        <p>During the error analysis, we considered various variables such as screen gaze position, head spatial position, head spatial pose, and lighting conditions. This comprehensive analysis allowed us to identify the factors that contribute to errors in our experimental data. By combining sequential point testing with thorough error analysis, we obtained a comprehensive understanding of our system’s performance and limitations under different conditions.</p>
        <p>
          <xref ref-type="fig" rid="fig5">Figure 5</xref> presents the error analysis for various variables. The horizontal axis represents the variables, while the vertical axis represents the average angular error within the corresponding interval. <xref ref-type="fig" rid="fig5">Figure 5A</xref> and <xref ref-type="fig" rid="fig5">B</xref> display the relationship between the horizontal and vertical positions of the target point on the screen and the angular error. <xref ref-type="fig" rid="fig5">Figure 5C</xref> presents a three-dimensional plot showing the relationship between the target point’s position on the screen and the angular error. The x and y axes represent the horizontal and vertical positions of the target point on the screen, respectively, while the z-axis represents the magnitude of the angular error. The error distribution ranges from 6 to 10 degrees.</p>
        <p>
          <xref ref-type="fig" rid="fig5">Figure 5D</xref> illustrates the relationship between the angular error and the horizontal distance between the observer’s gaze and the camera. The independent variable is the horizontal position of the midpoint between the eyes in the camera coordinate system, which represents the observer’s horizontal displacement. The horizontal distance is distributed from -0.1 to 0.15 meters, and the error distribution ranges from 4 to 10 degrees. <xref ref-type="fig" rid="fig5">Figure 5E</xref> demonstrates the relationship between the angular error and the vertical distance. The independent variable is the vertical position of the midpoint between the eyes in the camera coordinate system, representing the observer’s vertical displacement. The vertical distance values range from -0.15 to 0 meters, while the angular error distribution ranges from 5 to 10 degrees. <xref ref-type="fig" rid="fig5">Figure 5F</xref> presents the relationship between the angular error and the axial distance between the observer’s gaze and the camera. The independent variable is the axial position of the midpoint between the eyes in the camera coordinate system, which represents the observer’s forward and backward displacement and the distance between the observer and the display. The axial distance ranges from 0.5 to 0.85 meters, and the angular error distribution ranges from 4 to 8 degrees.</p>
        <p>Finally, <xref ref-type="fig" rid="fig5">Figure 5G</xref> displays the relationship between the angular error and the pitch angle of the observer’s head, representing the vertical swing amplitude. The pitch angle ranges from -70 to 20 degrees, and the angular error distribution ranges from 7 to 9 degrees. <xref ref-type="fig" rid="fig5">Figure 5H</xref> shows the relationship between the angular error and the yaw angle of the observer’s head, representing the horizontal swing amplitude. The yaw angle ranges from -10 to 38 degrees, and the angular error distribution ranges from 5 to 9 degrees.</p>
        <p>
          <xref ref-type="fig" rid="fig6">Figure 6</xref> illustrates the relationship between the mean and range of Red, Green, Blue (RGB) values and the angular error. The mean RGB values represent the intensity of the experimental ambient light, which is distributed in the range of 40-120. The range of RGB values represents the angle of the laboratory light source, where a larger light angle corresponds to a larger RGB range, distributed in the range of 80-200. In both cases, the errors are distributed within 5-10 degrees.</p>
      </sec>
      <sec id="sec3-2">
        <title>3.2. Performance at random point test</title>
        <p>We collected the results of random point tests for analysis. First, we plotted the distribution of points and the predicted standard deviation using the actual values and the predicted values. These are shown in <xref ref-type="fig" rid="fig7">Figure 7</xref>. In the left panel, the light-colored large circle represents the fixed target point in the test, with its center at the center of the target block and its diameter equal to the target block’s diagonal length. The dark-colored small circles represent the scattered positions of the predicted fixation points by the model. In the right part, the dark small circles represent the actual fixed target point locations, while the light large circles surrounding them indicate the range of predicted points corresponding to the labels. This figure shows the distribution of predicted points around the ground-truth values and the predictive standard deviation.</p>
        <fig id="fig7" position="float">
          <label>Figure 7</label>
          <caption>
            <p>Random point test results.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="ir6029.fig.7.jpg" />
        </fig>
        <p>During the experimental process, we retained the predicted data from the uncalibrated model and the unfiltered results, resulting in four data groups. These data are illustrated in <xref ref-type="fig" rid="fig8">Figure 8</xref>. The left graph depicts the functional relationship between angular error and the true values on the X-axis, where the scatter points represent the distribution of the data. The scatter points are plotted at a ratio of 1:100. The four curves represent the quadratic polynomial fit for each dataset, illustrating how the average angular error varies with the x-coordinate. The right graph presents a similar error analysis, with the Y-axis representing the true values. As shown in the graph, filtering the raw data has minimal effect on reducing the average error, whereas filtering after calibration substantially reduces it.</p>
        <fig id="fig8" position="float">
          <label>Figure 8</label>
          <caption>
            <p>Data filtering and model calibration analysis.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="ir6029.fig.8.jpg" />
        </fig>
        <p>To analyze the performance of the model, we conducted a series of control experiments and tested the following mathematical models: (1) K-nearest neighbors (KNN). In machine learning, KNN is commonly used for pattern recognition tasks. We expanded the image into a one-dimensional vector and concatenated it with the head position and attitude information, then used the KNN method with a neighborhood size of 3 for testing<sup>[<xref ref-type="bibr" rid="B30">30</xref>]</sup>; (2) Random forest (RF)<sup>[<xref ref-type="bibr" rid="B32">32</xref>]</sup>. RF is an effective regression method in machine learning. We tested the model using the same data-processing method, with a decision tree size of 300, a maximum depth of 20, and a maximum of 65 features at each node; (3) Linear regression (LR)<sup>[<xref ref-type="bibr" rid="B33">33</xref>]</sup>. LR is a simple regression method in machine learning and has been applied effectively to medical image analysis tasks<sup>[<xref ref-type="bibr" rid="B33">33</xref>]</sup>. We used the same data processing method and tested the model using LR.</p>
        <p>Finally, utilizing various models including KNN, RF, LR, GPN, and calibrated GPN, we obtained the prediction results for random-point testing. The results from different models were compared, and the error analysis results are presented in <xref ref-type="fig" rid="fig9">Figure 9</xref>. <xref ref-type="fig" rid="fig9">Figure 9A</xref> shows the error analysis graph with the true values on the x-axis, which has the same pattern as <xref ref-type="fig" rid="fig8">Figure 8</xref>, consisting of an error scatter plot and a fitting curve. In contrast, <xref ref-type="fig" rid="fig9">Figure 9B</xref> shows the error analysis graph with the true values on the y-axis.</p>
        <fig id="fig9" position="float">
          <label>Figure 9</label>
          <caption>
            <p>Performance comparison analysis of GPN and other models. (A) scatterplot and fitted curve analysis of the horizontal coordinates of the target points between different models; (B) Scatterplot and fitted curve analysis of the vertical coordinates of the target points between different models; (C) Analysis of mean angular errors for different models; error bars represent the standard deviations of angular errors across the test samples (<italic>n</italic> = 2); (D) Analysis of mean pixel errors for different models; error bars represent the standard deviations of pixel errors across the test samples (<italic>n</italic> = 2). GPN: Gaze-Point-Net; KNN: K-nearest neighbors; LR: linear regression; RF: random forest.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="ir6029.fig.9.jpg" />
        </fig>
        <p>Scattered points represent the error distribution of data sampled at a certain proportion, and a quadratic function is used to fit the scattered points into a curve. The blue, cyan, and yellow colors represent the results of KNN, LR, and RF, respectively. Pink represents the uncalibrated model results, while red represents the calibrated model results with added data-filtering operations. Our model shows a significantly smaller error distribution and lower standard deviation. Additionally, applying the filtering operations after calibration substantially improves the model’s performance.</p>
        <p>
          <xref ref-type="fig" rid="fig9">Figure 9C</xref> and <xref ref-type="fig" rid="fig9">D</xref> present numerical comparisons of the average angular error and average pixel error for the aforementioned models, respectively. The two figures depict the KNN, RF, LR, general GPN, and calibrated GPN models from left to right, representing progressively lighter shades. The results indicate that our model’s performance is comparable to RF before calibration but improves significantly after calibration.</p>
        <p>
          <xref ref-type="table" rid="t1">Table 1</xref> presents the mean and standard deviation of pixel error and angular error for different models. The screen has a pixel range of 2202, and its visible viewing angle ranges from 60 to 70 degrees; the calibrated GPN also includes the subsequent data-filtering operation. Our calibrated GPN achieves a pixel error of 185.19 ± 116.57 and a angular error of 4.52 ± 2.87, which is much smaller than that of KNN, RF, LR, and GPN.</p>
        <table-wrap id="t1">
          <label>Table 1</label>
          <caption>
            <p>Pixel and angular errors of the calibrated GPN and its counterparts</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;">
                  <bold>Model</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Pixel error</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Angular error</bold>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>KNN<sup>[<xref ref-type="bibr" rid="B30">30</xref>]</sup></td>
                <td>458.36 ± 263.17</td>
                <td>11.16 ± 6.19</td>
              </tr>
              <tr>
                <td>RF<sup>[<xref ref-type="bibr" rid="B32">32</xref>]</sup></td>
                <td>292.75 ± 152.27</td>
                <td>7.15 ± 3.72</td>
              </tr>
              <tr>
                <td>LR<sup>[<xref ref-type="bibr" rid="B33">33</xref>]</sup></td>
                <td>491.26 ± 267.23</td>
                <td>12.12 ± 6.75</td>
              </tr>
              <tr>
                <td>GPN</td>
                <td>357.03 ± 223.80</td>
                <td>8.85 ± 5.81</td>
              </tr>
              <tr>
                <td>
                  <bold>Calibrated GPN</bold>
                </td>
                <td>
                  <bold>185.19 ± 116.57</bold>
                </td>
                <td>
                  <bold>4.52 ± 2.87</bold>
                </td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn>
              <p>Bold content indicates the proposed approach, which achieved the best performance. GPN: Gaze-Point-Net; KNN: K-nearest neighbors; RF: random forest; LR: linear regression.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>We employed multiple regression task evaluation metrics to analyze the performance of the models used in our study, as shown in <xref ref-type="table" rid="t2">Table 2</xref>. As in <xref ref-type="table" rid="t1">Table 1</xref>, the calibrated GPN includes the subsequent data-filtering operation. The metrics include the root mean squared error (RMSE), MAE, R-squared (R<sup>2</sup>), mean absolute percentage error (MAPE), and mean squared percentage error (MSPE).</p>
        <table-wrap id="t2">
          <label>Table 2</label>
          <caption>
            <p>Multiple regression task evaluation metrics for the GPN and its counterparts</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;">
                  <bold>Model</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>RMSE</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>MAE</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>R<sup>2</sup></bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>MAPE</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>MSPE</bold>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>KNN<sup>[<xref ref-type="bibr" rid="B30">30</xref>]</sup></td>
                <td>373.7343</td>
                <td>291.3501</td>
                <td>0.2357</td>
                <td>130.6664</td>
                <td>1,025.7574</td>
              </tr>
              <tr>
                <td>RF<sup>[<xref ref-type="bibr" rid="B32">32</xref>]</sup></td>
                <td>233.3364</td>
                <td>187.8462</td>
                <td>0.6088</td>
                <td>90.7369</td>
                <td>437.8502</td>
              </tr>
              <tr>
                <td>LR<sup>[<xref ref-type="bibr" rid="B33">33</xref>]</sup></td>
                <td>392.678</td>
                <td>311.6328</td>
                <td>0.0536</td>
                <td>112.1944</td>
                <td>579.4017</td>
              </tr>
              <tr>
                <td>GPN</td>
                <td>293.8332</td>
                <td>232.2993</td>
                <td>0.3423</td>
                <td>62.4100</td>
                <td>185.9423</td>
              </tr>
              <tr>
                <td>
                  <bold>Calibrated GPN</bold>
                </td>
                <td>
                  <bold>156.5305</bold>
                </td>
                <td>
                  <bold>121.8607</bold>
                </td>
                <td>
                  <bold>0.8224</bold>
                </td>
                <td>
                  <bold>46.6341</bold>
                </td>
                <td>
                  <bold>124.8420</bold>
                </td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn>
              <p>Bold content indicates the proposed approach, which achieved the best performance. GPN: Gaze-Point-Net; RMSE: root mean squared error; MAE: mean absolute error; R<sup>2</sup>: R-squared; MAPE: mean absolute percentage error; MSPE: mean squared percentage error; KNN: K-nearest neighbors; RF: random forest; LR: linear regression.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>Based on the analysis of RMSE and MAE, our model exhibits significantly smaller errors compared to other models. Furthermore, for MAPE and MSPE, our model shows smaller relative errors and fewer instances of large errors. Additionally, the R<sup>2</sup> coefficient indicates a stronger correlation between our model’s regression results and the reference standards.</p>
        <p>The relationship between the horizontal and vertical coordinates of the ground truth values on the screen and the angular error is illustrated in <xref ref-type="fig" rid="fig10">Figure 10</xref> using a three-dimensional plot. From the final coordinate error analysis graph, the model’s prediction error was within 5° in most locations. After calibration and filtering, the average angular prediction error is substantially reduced to 4.52°, demonstrating the effectiveness of the proposed method in achieving accurate gaze estimation across different target locations.</p>
        <fig id="fig10" position="float" width="350">
          <label>Figure 10</label>
          <caption>
            <p>Three-dimensional error analysis.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="ir6029.fig.10.jpg" />
        </fig>
      </sec>
      <sec id="sec3-3">
        <title>3.3. Performance at trajectory tracking test</title>
        <p>The results of the trajectory tracking test are shown in <xref ref-type="fig" rid="fig11">Figure 11</xref>. The left part displays the distribution heatmap of prediction errors. The heatmap depicts the trajectory of the moving target, with brighter colors indicating larger errors and darker colors indicating smaller errors. The right part shows the distribution of scattered points within a certain confidence region. The light-colored circular rings in the background represent tolerance boundaries around the target trajectory, with the radius indicating the allowable distance from the trajectory. Scatter plots depict the distribution of predicted gaze points. A confidence radius is defined to determine whether predicted gaze points fall within a specified distance from the target trajectory and are therefore considered valid tracks. With a tolerance radius of 300 pixels, the tracking accuracy reaches 93.75%. When the tolerance radius is reduced to 200 pixels, the accuracy is 81.91%.</p>
        <fig id="fig11" position="float">
          <label>Figure 11</label>
          <caption>
            <p>Circular trajectory tracking test results.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="ir6029.fig.11.jpg" />
        </fig>
      </sec>
      <sec id="sec3-4">
        <title>3.4. Analysis of browsing patterns</title>
        <p>
          <xref ref-type="fig" rid="fig12">Figure 12</xref> presents the browsing-mode analysis, in which the gaze positions of a participant were simultaneously recorded using a professional eye tracker (Tobii Pro) and the proposed eye-tracking system while freely viewing each image<sup>[<xref ref-type="bibr" rid="B34">34</xref>]</sup>. The first column [<xref ref-type="fig" rid="fig12">Figure 12A</xref>] shows four representative images viewed by the participant. The second and fourth columns [<xref ref-type="fig" rid="fig12">Figure 12B</xref> and <xref ref-type="fig" rid="fig12">D</xref>] present the heat maps generated by the professional eye tracker and the proposed system, respectively, while the third and fifth columns [<xref ref-type="fig" rid="fig12">Figure 12C</xref> and <xref ref-type="fig" rid="fig12">E</xref>] show the corresponding gaze trajectory maps. The heat maps illustrate the distribution and density of gaze points, with deeper red indicating higher gaze density and thus the observer’s focal areas of attention. The trajectory maps plot gaze points in chronological order, illustrating the temporal progression of visual attention across the image. Overall, the heat maps and trajectory maps generated by the proposed system closely resemble those obtained from the professional eye tracker, with only minor differences in gaze-point locations. These results demonstrate that the proposed system can provide gaze-tracking patterns comparable to those of the professional eye tracker during free-viewing tasks.</p>
        <fig id="fig12" position="float">
          <label>Figure 12</label>
          <caption>
            <p>Representative examples of browsing-pattern analysis. (A) Original images viewed by the participant; (B) and (D) heat maps generated by the Tobii Pro eye tracker and the proposed system, respectively; (C) and (E) corresponding gaze trajectory maps. They were obtained from the Kodak Lossless True Color Image Suite, originally released by Eastman Kodak Company (1999); the publicly accessible mirror at <uri xlink:href="https://r0k.us/graphics/kodak/">https://r0k.us/graphics/kodak/</uri> (accessed: 22.2.2026) was used for retrieval, and the Berkeley Segmentation Dataset and Benchmark (BSDS300) (<uri xlink:href="https://www2.eecs.berkeley.edu/Research/Projects/CS/vision/bsds/">https://www2.eecs.berkeley.edu/Research/Projects/CS/vision/bsds/</uri>)<sup>[<xref ref-type="bibr" rid="B34">34</xref>]</sup>.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="ir6029.fig.12.jpg" />
        </fig>
      </sec>
    </sec>
    <sec id="sec4">
      <title>4. DISCUSSION</title>
      <sec id="sec4-1">
        <title>4.1. Principal findings</title>
        <p>In this study, we designed a comprehensive system for gaze estimation and evaluated its performance through various experiments, drawing on the concept of explicit gaze control in neural talking-head synthesis<sup>[<xref ref-type="bibr" rid="B35">35</xref>]</sup>. We demonstrated the feasibility of gaze estimation without infrared cameras and showed that a multimodal approach incorporating head spatial position and pose information can enhance the accuracy and generalization performance of gaze estimation. Furthermore, while training the object detection model, we found that increasing the resolution of the feature maps in YOLOv5 can improve the model’s ability to recognize small-sized objects. Additionally, combining similar but different object categories (the left and right eyes) into the same class can significantly improve detection performance for that class.</p>
        <p>In the computational workflow, we discovered that in the process of using neural networks for calculations, separating and then fusing features - namely, first computing the head’s position and pose information while using perspective transformation to remove pose information from eye images, and then utilizing neural networks for feature fusion - improved the accuracy to some extent.</p>
        <p>We directly regress the target fixation point, which reduces computational complexity and makes calibration more efficient. Calibrating the gaze estimation model before each prediction task greatly improves its accuracy.</p>
        <p>Finally, our model performs much better in fixed-point gaze tasks (Random point test) compared to scanning tasks (Sequential point test) involving moving points. This is because our random-point testing procedure averages angular errors and standard deviations after the random point remains fixed for a certain period, yielding slightly lower errors than sequential-point testing involving continuous movement of the target point.</p>
      </sec>
      <sec id="sec4-2">
        <title>4.2. Comparison with previous studies</title>
        <p>Since existing gaze estimation studies often adopt different datasets, input modalities, and evaluation protocols, direct performance comparisons across datasets may not be fully reliable. Therefore, <xref ref-type="table" rid="t3">Table 3</xref> provides a reference comparison with representative methods reported in the literature, rather than a strict quantitative benchmark. The results highlight the effectiveness of our integrated gaze estimation system under the experimental setting adopted in this study.</p>
        <table-wrap id="t3">
          <label>Table 3</label>
          <caption>
            <p>Comparison with representative gaze estimation methods reported in the literature</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;">
                  <bold>Study</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Key aspects</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Performance</bold>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>Our method</td>
                <td>- Gaze estimation<break />- Eye movement<break />- Depth camera<break />- Free head movement<break />- Calibration</td>
                <td>Angular error = 4.52° ± 2.87°<break />Pixel error: = 185.19 ± 116.57</td>
              </tr>
              <tr>
                <td>Arar <italic>et al.</italic>, 2017<sup>[<xref ref-type="bibr" rid="B36">36</xref>]</sup></td>
                <td>- Eye movements<break />- HCI<break />- Gaze estimation<break />- Video-based remote eye trackers<break />- User calibration<break />- Estimation bias<break />- Explicit and implicit user calibration methods<break />- Regression-based user calibration techniques<break />- Weighted least squares regression<break />- Real-time cross-ratio based gaze estimation<break />- State-of-the-art user calibration methods</td>
                <td>Angular error = 1.01° ± 1°</td>
              </tr>
              <tr>
                <td>Bao <italic>et al.</italic>, 2022<sup>[<xref ref-type="bibr" rid="B37">37</xref>]</sup></td>
                <td>- Deep learning-based approaches<break />- Appearance-based gaze estimation<break />- Unseen environments<break />- Generalizing gaze estimation algorithm<break />- Rotation-consistency property<break />- Unsupervised domain adaptation<break />- Distribution loss<break />- Cross-domain gaze estimation tasks<break />- Experimental results<break />- Baselines<break />- Computer vision tasks<break />- Physical constraints</td>
                <td>Angular error = 5.70°</td>
              </tr>
              <tr>
                <td>Zhang <italic>et al.</italic>, 2020<sup>[<xref ref-type="bibr" rid="B38">38</xref>]</sup></td>
                <td>- ETH-XGaze<break />- HCI<break />- Custom datasets<break />- Comparison across methods<break />- ETH-XGaze dataset<break />- High-resolution images<break />- Extreme head poses<break />- Ground truth gaze targets<break />- Robustness of Gaze estimation methods<break />- Standardized experimental protocol<break />- Benchmark website<break />- Unified gaze estimation research</td>
                <td>Angular error = 8° at extreme angle</td>
              </tr>
              <tr>
                <td>Kim <italic>et al.</italic>, 2020<sup>[<xref ref-type="bibr" rid="B39">39</xref>]</sup></td>
                <td>- Smart interactive environments<break />- User intent prediction<break />- Gaze estimation technology<break />- Interaction techniques<break />- Deep learning-based approach<break />- Low-light conditions<break />- Eye image enhancement<break />- MPIIGaze dataset</td>
                <td>Angular error = 9.52° when the light is dim</td>
              </tr>
              <tr>
                <td>Ren <italic>et al.</italic>, 2023<sup>[<xref ref-type="bibr" rid="B40">40</xref>]</sup></td>
                <td>- Low-light environment<break />- Gaze estimation<break />- Feature fusion<break />- Multi-level information elements<break />- Gaze conduction principle<break />- Multi-level information element fusion model<break />- Optimized input modes and network structures<break />- GazeCapture dataset<break />- Average error</td>
                <td>Angular error = 4.38°</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn>
              <p>HCI: Human-computer interaction.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>In existing appearance-based gaze estimation methods, it has been demonstrated that combining head spatial position and three-dimensional posture methods is highly effective for gaze estimation under free-viewing conditions<sup>[<xref ref-type="bibr" rid="B36">36</xref>]</sup>. Our method includes a process for correcting the head roll angle. When the head undergoes angular changes along the axial rotation, we use a matrix transformation to correct the inverse rotation. This eliminates roll-angle features in the image. We incorporate this corrected feature as an input to the model alongside other modalities, using the multimodal approach we propose.</p>
        <p>The gaze estimation method for head rotation in Ref.<sup>[<xref ref-type="bibr" rid="B37">37</xref>]</sup> directly computes on images without reverse-rotation correction. In comparison, their method relies on separate feature extraction, whereas our approach more thoroughly extracts rotation features. Our method first disentangles the features before fusing them, resulting in improved performance. Ultimately, our method achieves an error 1.18° lower than that of their approach.</p>
        <p>Similarly, our approach incorporates perspective transformation for separating the head’s yaw and pitch angles. As a result, our method demonstrates a certain degree of adaptability to both horizontal and vertical head rotations. The method described in<sup>[<xref ref-type="bibr" rid="B38">38</xref>]</sup> focuses on gaze estimation for baseline angles. They trained their model on datasets with significant angle variations and achieved an error of approximately 8 degrees. <xref ref-type="fig" rid="fig5">Figure 5G</xref> and <xref ref-type="fig" rid="fig5">H</xref> illustrate the error analysis results of our method in these two dimensions. Compared with this specialized approach tailored to the specific conditions, our method achieves a performance close to its capabilities.</p>
        <p>We acquired data under various lighting conditions by adjusting the illumination. Furthermore, during the image preprocessing phase, techniques such as histogram equalization and data standardization were applied to mitigate the impact of variations in image RGB values. In contrast to previous work<sup>[<xref ref-type="bibr" rid="B39">39</xref>]</sup>, which focused on dim-lighting scenarios (resulting in an error of 9.52 degrees), our approach demonstrates an average error reduction of 5.00 degrees. As illustrated in <xref ref-type="fig" rid="fig6">Figure 6</xref>, our method appears to be less affected by variations in light intensity and direction, thus exhibiting strong performance across different lighting conditions.</p>
        <p>The method mentioned in<sup>[<xref ref-type="bibr" rid="B40">40</xref>]</sup> is a gaze tracking approach for mobile devices. It utilizes the front camera of a smartphone to capture images, locates 68 facial landmarks using a model, and performs estimation. Our method, by contrast, is deployed on desktop platforms and benefits from higher computational power. We employ the YOLO model for target detection; in comparison, our method exhibits higher fault tolerance, faster computation speed, and lower resource consumption. On desktop platforms, we achieve performance similar to theirs with greater efficiency and resource savings.</p>
        <p>Our approach shares some similarities with the method proposed in<sup>[<xref ref-type="bibr" rid="B41">41</xref>]</sup>, as both utilize facial-related information for gaze estimation. However, Ref.<sup>[<xref ref-type="bibr" rid="B41">41</xref>]</sup> employs an long short-term memory (LSTM)-based framework to model temporal dependencies from continuous facial sequences, whereas our method focuses on individual-frame gaze estimation by integrating head pose information, binocular features, and calibration strategies. This design reduces the dependency on temporal information and enables a more lightweight implementation for scenarios where continuous data acquisition is unavailable or unnecessary. Although Ref.<sup>[<xref ref-type="bibr" rid="B41">41</xref>]</sup> demonstrates the effectiveness of temporal modeling for gaze estimation, a direct quantitative comparison between the two methods is difficult due to differences in datasets, input settings, and experimental protocols. Nevertheless, the comparison highlights different design considerations: temporal modeling can improve performance when sequential data are available, while our framework provides a practical alternative for frame-based gaze estimation with reduced dependence on continuous recordings. Future work will further investigate this issue using unified experimental protocols and larger datasets.</p>
        <p>Compared to existing methods, one advantage is our calibration approach. We fine-tune neural networks for calibration, enabling robust calibration under various conditions such as changes in characters, lighting, camera parameters, camera field of view, camera position, screen size, and more. This calibration significantly enhances accuracy. During system deployment, our approach eliminates the need for fixed camera placement, head fixation, and external conditions like controlled lighting. This allows tasks to be accomplished without setting up a series of external conditions, providing a more flexible and efficient solution.</p>
      </sec>
      <sec id="sec4-3">
        <title>4.3. Limitations and future work</title>
        <p>Despite the promising performance of the proposed gaze estimation system, several limitations remain.</p>
        <p>First, the current system relies solely on images captured under visible light conditions, without incorporating infrared sensors. Consequently, its performance is sensitive to illumination variations. Although data augmentation and training strategies were employed to improve robustness, illumination effects cannot be fully eliminated. This limitation could be addressed by introducing advanced data enhancement techniques, such as diffusion-based models, to improve image quality under challenging lighting conditions and enhance robustness.</p>
        <p>Second, the target detection module operates on a per-frame basis without enforcing temporal consistency. This leads to fluctuations in detection results across consecutive frames, causing inconsistencies in the extracted eye regions and resulting in random directional deviations in gaze predictions. To overcome this limitation, temporal consistency constraints could be incorporated, and gaze estimation models designed for continuous video streams could be developed to leverage inter-frame information and reduce prediction drift.</p>
        <p>Third, the current pipeline lacks fine-grained preprocessing of eye images, such as explicit segmentation of the iris and eyelids, which limits the effectiveness and accuracy of feature extraction. In addition, the absence of temporal modeling prevents mitigation of randomness inherent in single-frame predictions. This issue could be alleviated by integrating eye-region segmentation methods to isolate key anatomical structures and adopting sequence-based modeling approaches to better exploit temporal dynamics.</p>
        <p>Finally, although the current study adopts a participant-level data partitioning strategy to ensure subject-disjoint training, validation, and testing, further evaluation using more rigorous subject-independent protocols, such as leave-one-subject-out cross-validation, would provide additional evidence of the proposed method’s robustness and generalizability. Moreover, a systematic ablation analysis of individual components within the proposed framework would further clarify the contribution of each module and provide deeper insights into the model design. Future studies will investigate larger and more diverse cohorts, incorporate more comprehensive cross-participant validation strategies, and conduct extensive ablation studies to further assess the model’s applicability and reliability in real-world scenarios.</p>
      </sec>
    </sec>
    <sec id="sec5">
      <title>5. CONCLUSION</title>
      <p>In this study, we developed a gaze estimation system that integrates direct screen position regression, binocular gaze estimation, multimodal feature fusion, and model calibration into a unified framework. By directly regressing screen positions, the system eliminates the coordinate transformation process, thereby simplifying the estimation pipeline, improving computational efficiency, reducing error accumulation, and increasing fault tolerance. A binocular model with unshared parameters is adopted to minimize the influence of binocular disparity encountered in monocular models. In addition, a multimodal convolutional neural network (CNN) model incorporating head spatial position and pose information enhances the system’s adaptability to natural head movements and rotations. Perspective transformation is further applied to separate binocular image pose information from multimodal feature fusion, thereby improving the overall estimation performance. Finally, combined with the model calibration strategy, the proposed system supports gaze estimation in a wide range of daily-life applications. The proposed system also exhibits strong calibration capability, enabling effective adaptation to variations across different operating conditions. Moreover, the model remains relatively compact, and when combined with the high computational efficiency of the YOLO-based detector, the overall inference speed does not limit the camera’s sampling frequency. Consequently, the integrated system achieves both high efficiency and accuracy during inference, enabling reliable fixed-point gaze estimation in practical applications.</p>
    </sec>
  </body>
  <back>
    <sec>
      <title>DECLARATIONS</title>
      <sec>
        <title>Authors’ contributions</title>
        <p>Conceptualization, methodology, software, investigation, data curation, writing - original draft: Wang, Z.</p>
        <p>Software, validation, formal analysis, visualization: Wang, J.</p>
        <p>Resources, investigation, validation: Ng, A. C. M.</p>
        <p>Resources, project administration, validation: Ng, T.</p>
        <p>Validation, formal analysis, writing - review and editing: Qian, W.</p>
        <p>Writing - review and editing, validation: Elirehema, M.</p>
        <p>Supervision, writing - review and editing, funding acquisition: Monkam, P.</p>
        <p>Supervision, conceptualization, writing - review and editing, funding acquisition: Qi, S.</p>
      </sec>
      <sec>
        <title>Availability of data and materials</title>
        <p>The dataset generated and analyzed in this study is not currently publicly available due to pending data-sharing permissions. It may be obtained from the corresponding author upon reasonable request, subject to the required ethical and data-sharing approvals.</p>
      </sec>
      <sec>
        <title>AI and AI-assisted tools statement</title>
        <p>During the preparation of this manuscript, the authors used ChatGPT (version 5.0, released 2025-08-07) solely for language polishing and grammar checking. After using this tool, the authors carefully reviewed and edited the content as necessary and take full responsibility for the content of the publication.</p>
      </sec>
      <sec>
        <title>Financial support and sponsorship</title>
        <p>This study was supported by the National Natural Science Foundation of China (82472076), the Fundamental Research Funds for the Central Universities (N26LPY002, N25BJD013), and Open Funding from Shenzhen Jingmei Health Technology Company Ltd.</p>
      </sec>
      <sec>
        <title>Conflicts of interest</title>
        <p>Ng, A. C. M. and Ng, T. are affiliated with Shenzhen Jingmei Health Technology Co., Ltd., while the other authors have declared no conflicts of interest.</p>
      </sec>
      <sec>
        <title>Ethical approval and consent to participate</title>
        <p>This study was approved by the Ethics Committee of Northeastern University (approval no. NEU-EC-2024B036S) and was conducted in accordance with institutional and ethical guidelines and the Declaration of Helsinki (2024). Written informed consent was obtained from all experimental participants.</p>
      </sec>
      <sec>
        <title>Consent for publication</title>
        <p>Written informed consent was obtained from all experimental participants for the publication of this study, including the use of their data and any facial images contained within the figures.</p>
      </sec>
      <sec>
        <title>Copyright</title>
        <p>© The Author(s) 2026.</p>
      </sec>
    </sec>
    <ref-list>
      <ref id="B1">
        <label>1</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Casado-Aranda</surname>
              <given-names>L</given-names>
            </name>
            <name>
              <surname>Sánchez-Fernández</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Ibáñez-Zapata</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Evaluating communication effectiveness through eye tracking: benefits, state of the art, and unresolved questions</article-title>
          <source>Int J Bus Commun</source>
          <year>2023</year>
          <volume>60</volume>
          <fpage>24</fpage>
          <lpage>61</lpage>
          <pub-id pub-id-type="doi">10.1177/2329488419893746</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B2">
        <label>2</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Liu</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Chi</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Yang</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Yin</surname>
              <given-names>X</given-names>
            </name>
          </person-group>
          <article-title>In the eye of the beholder: a survey of gaze tracking techniques</article-title>
          <source>Pattern Recognit</source>
          <year>2022</year>
          <volume>132</volume>
          <fpage>108944</fpage>
          <pub-id pub-id-type="doi">10.1016/j.patcog.2022.108944</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B3">
        <label>3</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Fan</surname>
              <given-names>K</given-names>
            </name>
            <name>
              <surname>Cao</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Meng</surname>
              <given-names>Z</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Predicting the reader’s English level from reading fixation patterns using the siamese convolutional neural network</article-title>
          <source>IEEE Trans Neural Syst Rehabil Eng</source>
          <year>2022</year>
          <volume>30</volume>
          <fpage>1071</fpage>
          <lpage>80</lpage>
          <pub-id pub-id-type="doi">10.1109/tnsre.2022.3157768</pub-id>
          <pub-id pub-id-type="pmid">35259110</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B4">
        <label>4</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Wang</surname>
              <given-names>FS</given-names>
            </name>
            <name>
              <surname>Gianduzzo</surname>
              <given-names>C</given-names>
            </name>
            <name>
              <surname>Meboldt</surname>
              <given-names>M</given-names>
            </name>
            <name>
              <surname>Lohmeyer</surname>
              <given-names>Q</given-names>
            </name>
          </person-group>
          <article-title>An algorithmic approach to determine expertise development using object-related gaze pattern sequences</article-title>
          <source>Behav Res Methods</source>
          <year>2022</year>
          <volume>54</volume>
          <fpage>493</fpage>
          <lpage>507</lpage>
          <pub-id pub-id-type="doi">10.3758/s13428-021-01652-z</pub-id>
          <pub-id pub-id-type="pmid">34258709</pub-id>
          <pub-id pub-id-type="pmcid">PMC8863757</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B5">
        <label>5</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Fan</surname>
              <given-names>X</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>F</given-names>
            </name>
            <name>
              <surname>Song</surname>
              <given-names>D</given-names>
            </name>
            <name>
              <surname>Lu</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Liu</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>GazMon: eye gazing enabled driving behavior monitoring and prediction</article-title>
          <source>IEEE Trans Mob Comput</source>
          <year>2021</year>
          <volume>20</volume>
          <fpage>1420</fpage>
          <lpage>33</lpage>
          <pub-id pub-id-type="doi">10.1109/tmc.2019.2962764</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B6">
        <label>6</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Čegovnik</surname>
              <given-names>T</given-names>
            </name>
            <name>
              <surname>Stojmenova</surname>
              <given-names>K</given-names>
            </name>
            <name>
              <surname>Jakus</surname>
              <given-names>G</given-names>
            </name>
            <name>
              <surname>Sodnik</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>An analysis of the suitability of a low-cost eye tracker for assessing the cognitive load of drivers</article-title>
          <source>Appl Ergon</source>
          <year>2018</year>
          <volume>68</volume>
          <fpage>1</fpage>
          <lpage>11</lpage>
          <pub-id pub-id-type="doi">10.1016/j.apergo.2017.10.011</pub-id>
          <pub-id pub-id-type="pmid">29409621</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B7">
        <label>7</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Li</surname>
              <given-names>P</given-names>
            </name>
            <name>
              <surname>Hou</surname>
              <given-names>X</given-names>
            </name>
            <name>
              <surname>Duan</surname>
              <given-names>X</given-names>
            </name>
            <name>
              <surname>Yip</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Song</surname>
              <given-names>G</given-names>
            </name>
            <name>
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <article-title>Appearance-based gaze estimator for natural interaction control of surgical robots</article-title>
          <source>IEEE Access</source>
          <year>2019</year>
          <volume>7</volume>
          <fpage>25095</fpage>
          <lpage>110</lpage>
          <pub-id pub-id-type="doi">10.1109/access.2019.2900424</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B8">
        <label>8</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Chhimpa</surname>
              <given-names>GR</given-names>
            </name>
            <name>
              <surname>Kumar</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Garhwal</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Dhiraj</surname>
              <given-names />
            </name>
          </person-group>
          <article-title>Development of a real-time eye movement-based computer interface for communication with improved accuracy for disabled people under natural head movements</article-title>
          <source>J Real Time Image Proc</source>
          <year>2023</year>
          <volume>20</volume>
          <fpage>81</fpage>
          <pub-id pub-id-type="doi">10.1007/s11554-023-01336-1</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B9">
        <label>9</label>
        <nlm-citation publication-type="journal">
          <article-title>Ramirez Gomez, A.; Lankes, M. Eyesthetics: making sense of the aesthetics of playing with gaze</article-title>
          <source>Proc ACM Hum Comput Interact</source>
          <year>2021</year>
          <volume>5</volume>
          <fpage>1</fpage>
          <lpage>24</lpage>
          <pub-id pub-id-type="doi">10.1145/3474686</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B10">
        <label>10</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Papavlasopoulou</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Sharma</surname>
              <given-names>K</given-names>
            </name>
            <name>
              <surname>Melhart</surname>
              <given-names>D</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Investigating gaze interaction to support children’s gameplay</article-title>
          <source>Int J Child Comput Interact</source>
          <year>2021</year>
          <volume>30</volume>
          <fpage>100349</fpage>
          <pub-id pub-id-type="doi">10.1016/j.ijcci.2021.100349</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B11">
        <label>11</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Deane</surname>
              <given-names>O</given-names>
            </name>
            <name>
              <surname>Toth</surname>
              <given-names>E</given-names>
            </name>
            <name>
              <surname>Yeo</surname>
              <given-names>SH</given-names>
            </name>
          </person-group>
          <article-title>Deep-SAGA: a deep-learning-based system for automatic gaze annotation from eye-tracking data</article-title>
          <source>Behav Res Methods</source>
          <year>2023</year>
          <volume>55</volume>
          <fpage>1372</fpage>
          <lpage>91</lpage>
          <pub-id pub-id-type="doi">10.3758/s13428-022-01833-4</pub-id>
          <pub-id pub-id-type="pmid">35650384</pub-id>
          <pub-id pub-id-type="pmcid">PMC10126076</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B12">
        <label>12</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Yuan</surname>
              <given-names>G</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Yan</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Fu</surname>
              <given-names>X</given-names>
            </name>
          </person-group>
          <article-title>Self-calibrated driver gaze estimation via gaze pattern learning</article-title>
          <source>Knowl Based Syst</source>
          <year>2022</year>
          <volume>235</volume>
          <fpage>107630</fpage>
          <pub-id pub-id-type="doi">10.1016/j.knosys.2021.107630</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B13">
        <label>13</label>
        <nlm-citation publication-type="book">
          <person-group person-group-type="author">
            <name>
              <surname>Namnakani</surname>
              <given-names>O</given-names>
            </name>
            <name>
              <surname>Abdrabou</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Grizou</surname>
              <given-names>J</given-names>
            </name>
            <etal />
          </person-group>
          <comment>Comparing dwell time, pursuits and gaze gestures for gaze interaction on handheld mobile devices. In <italic>Proceedings of the 2023 CHI Conference on Human Factors in Computing Systems</italic>. Association for Computing Machinery; 2023. pp. 1-17.</comment>
          <pub-id pub-id-type="doi">10.1145/3544548.3580871</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B14">
        <label>14</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Hacques</surname>
              <given-names>G</given-names>
            </name>
            <name>
              <surname>Dicks</surname>
              <given-names>M</given-names>
            </name>
            <name>
              <surname>Komar</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Seifert</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>Visual control during climbing: variability in practice fosters a proactive gaze pattern</article-title>
          <source>PLoS One</source>
          <year>2022</year>
          <volume>17</volume>
          <fpage>e0269794</fpage>
          <pub-id pub-id-type="doi">10.1371/journal.pone.0269794</pub-id>
          <pub-id pub-id-type="pmid">35687600</pub-id>
          <pub-id pub-id-type="pmcid">PMC9187105</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B15">
        <label>15</label>
        <nlm-citation publication-type="book">
          <person-group person-group-type="author">
            <name>
              <surname>Fischer</surname>
              <given-names>T</given-names>
            </name>
            <name>
              <surname>Chang</surname>
              <given-names>HJ</given-names>
            </name>
            <name>
              <surname>Demiris</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <comment>RT-GENE: real-time eye gaze estimation in natural environments. In <italic>Computer Vision - ECCV 2018</italic>. Cham: Springer International Publishing; 2018. pp. 339-57.</comment>
          <pub-id pub-id-type="doi">10.1007/978-3-030-01249-6_21</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B16">
        <label>16</label>
        <nlm-citation publication-type="book">
          <person-group person-group-type="author">
            <name>
              <surname>Wu</surname>
              <given-names>Z</given-names>
            </name>
            <name>
              <surname>Rajendran</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Van As</surname>
              <given-names>T</given-names>
            </name>
            <name>
              <surname>Badrinarayanan</surname>
              <given-names>V</given-names>
            </name>
            <name>
              <surname>Rabinovich</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <comment>EyeNet: a multi-task deep network for off-axis eye gaze estimation. In <italic>2019 IEEE/CVF International Conference on Computer Vision Workshop (ICCVW)</italic>, Seoul, Korea. Ocr 27-28, 2019. IEEE; 2019. pp. 3683-7.</comment>
          <pub-id pub-id-type="doi">10.1109/ICCVW.2019.00455</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B17">
        <label>17</label>
        <nlm-citation publication-type="book">
          <person-group person-group-type="author">
            <name>
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Jiang</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Li</surname>
              <given-names>J</given-names>
            </name>
            <etal />
          </person-group>
          <comment>Contrastive regression for domain adaptation on gaze estimation. In <italic>2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</italic>, New Orleans, USA. Jun 18-24, 2022. IEEE; 2022. pp. 19354-63.</comment>
          <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.01877</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B18">
        <label>18</label>
        <nlm-citation publication-type="book">
          <person-group person-group-type="author">
            <name>
              <surname>Nonaka</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Nobuhara</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Nishino</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <comment>Dynamic 3D gaze from afar: deep gaze estimation from temporal eye-head-body coordination. In <italic>2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</italic>, New Orleans, USA. Jun 18-24, 2022. IEEE; 2022. pp. 2182-91.</comment>
          <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.00223</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B19">
        <label>19</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Fatahipour</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Mosavi</surname>
              <given-names>MR</given-names>
            </name>
            <name>
              <surname>Fariborz</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Uncalibrated eye gaze estimation using SE-ResNext with unconstrained head movement and ambient light change</article-title>
          <source>Research Square</source>
          <year>2023</year>
          <pub-id pub-id-type="doi">10.21203/rs.3.rs-2666872/v1</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B20">
        <label>20</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Ma</surname>
              <given-names>C</given-names>
            </name>
            <name>
              <surname>Baek</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Choi</surname>
              <given-names>K</given-names>
            </name>
            <name>
              <surname>Ko</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Improved remote gaze estimation using corneal reflection-adaptive geometric transforms</article-title>
          <source>Opt Eng</source>
          <year>2014</year>
          <volume>53</volume>
          <fpage>053112</fpage>
          <pub-id pub-id-type="doi">10.1117/1.oe.53.5.053112</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B21">
        <label>21</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Shin</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Choi</surname>
              <given-names>K</given-names>
            </name>
            <name>
              <surname>Kim</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Ko</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>A novel single IR light based gaze estimation method using virtual glints</article-title>
          <source>IEEE Trans Consum Electron</source>
          <year>2015</year>
          <volume>61</volume>
          <fpage>254</fpage>
          <lpage>60</lpage>
          <pub-id pub-id-type="doi">10.1109/tce.2015.7150601</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B22">
        <label>22</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Chi</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Liu</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>F</given-names>
            </name>
            <name>
              <surname>Chi</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Hou</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>3-D gaze-estimation method using a multi-camera-multi-light-source system</article-title>
          <source>IEEE Trans Instrum Meas</source>
          <year>2020</year>
          <volume>69</volume>
          <fpage>9695</fpage>
          <lpage>708</lpage>
          <pub-id pub-id-type="doi">10.1109/tim.2020.3006681</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B23">
        <label>23</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Liu</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Chi</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Hu</surname>
              <given-names>W</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>3D model-based gaze tracking via iris features with a single camera and a single light source</article-title>
          <source>IEEE Trans Human Mach Syst</source>
          <year>2021</year>
          <volume>51</volume>
          <fpage>75</fpage>
          <lpage>86</lpage>
          <pub-id pub-id-type="doi">10.1109/thms.2020.3035176</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B24">
        <label>24</label>
        <nlm-citation publication-type="book">
          <person-group person-group-type="author">
            <name>
              <surname>Qin</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Shimoyama</surname>
              <given-names>T</given-names>
            </name>
            <name>
              <surname>Sugano</surname>
              <given-names>Y</given-names>
            </name>
          </person-group>
          <comment>Learning-by-novel-view-synthesis for full-face appearance-based 3D gaze estimation. In <italic>2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)</italic>, New Orleans, USA. Jun 19-20, 2022. IEEE; 2022. pp. 4977-87.</comment>
          <pub-id pub-id-type="doi">10.1109/CVPRW56347.2022.00546</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B25">
        <label>25</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Li</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Chen</surname>
              <given-names>Z</given-names>
            </name>
            <name>
              <surname>Zhong</surname>
              <given-names>Y</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Appearance-based gaze estimation for ASD diagnosis</article-title>
          <source>IEEE Trans Cybern</source>
          <year>2022</year>
          <volume>52</volume>
          <fpage>6504</fpage>
          <lpage>17</lpage>
          <pub-id pub-id-type="doi">10.1109/tcyb.2022.3165063</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B26">
        <label>26</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Wang</surname>
              <given-names>Q</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Dang</surname>
              <given-names>R</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Style transformed synthetic images for real world gaze estimation by using residual neural network with embedded personal identities</article-title>
          <source>Appl Intell</source>
          <year>2023</year>
          <volume>53</volume>
          <fpage>2026</fpage>
          <lpage>41</lpage>
          <pub-id pub-id-type="doi">10.1007/s10489-022-03481-9</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B27">
        <label>27</label>
        <nlm-citation publication-type="book">
          <person-group person-group-type="author">
            <name>
              <surname>Khanam</surname>
              <given-names>R</given-names>
            </name>
            <name>
              <surname>Hussain</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <comment>What is YOLOv5: a deep look into the internal features of the popular object detector. <italic>arXiv</italic> <bold>2024</bold>, arXiv:2407.20892. Available online: <uri xlink:href="https://doi.org/10.48550/arXiv.2407.20892">https://doi.org/10.48550/arXiv.2407.20892</uri>. (accessed 2026-09-11)</comment>
        </nlm-citation>
      </ref>
      <ref id="B28">
        <label>28</label>
        <nlm-citation publication-type="book">
          <person-group person-group-type="author">
            <name>
              <surname>Keselman</surname>
              <given-names>L</given-names>
            </name>
            <name>
              <surname>Iselin Woodfill</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Grunnet-Jepsen</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Bhowmik</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <comment>Intel(R) realSense(TM) stereoscopic depth cameras. In <italic>2017 IEEE Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)</italic>, Honolulu, USA. Jul 21-26, 2017. IEEE; 2017. pp. 1267-76.</comment>
          <pub-id pub-id-type="doi">10.1109/CVPRW.2017.167</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B29">
        <label>29</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Hu</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Shen</surname>
              <given-names>L</given-names>
            </name>
            <name>
              <surname>Albanie</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Sun</surname>
              <given-names>G</given-names>
            </name>
            <name>
              <surname>Wu</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Squeeze-and-excitation networks</article-title>
          <source>IEEE Trans Pattern Anal Mach Intell</source>
          <year>2020</year>
          <volume>42</volume>
          <fpage>2011</fpage>
          <lpage>23</lpage>
          <pub-id pub-id-type="doi">10.1109/tpami.2019.2913372</pub-id>
          <pub-id pub-id-type="pmid">31034408</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B30">
        <label>30</label>
        <nlm-citation publication-type="book">
          <person-group person-group-type="author">
            <name>
              <surname>Park</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>De Mello</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Molchanov</surname>
              <given-names>P</given-names>
            </name>
            <name>
              <surname>Iqbal</surname>
              <given-names>U</given-names>
            </name>
            <name>
              <surname>Hilliges</surname>
              <given-names>O</given-names>
            </name>
            <name>
              <surname>Kautz</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <comment>Few-shot adaptive gaze estimation. In <italic>2019 IEEE/CVF International Conference on Computer Vision (ICCV)</italic>, Seoul, Korea. Oct 27 - Nov 02, 2019. IEEE; 2019. pp. 9367-76.</comment>
          <pub-id pub-id-type="doi">10.1109/ICCV.2019.00946</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B31">
        <label>31</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Valliappan</surname>
              <given-names>N</given-names>
            </name>
            <name>
              <surname>Dai</surname>
              <given-names>N</given-names>
            </name>
            <name>
              <surname>Steinberg</surname>
              <given-names>E</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Accelerating eye movement research via accurate and affordable smartphone eye tracking</article-title>
          <source>Nat Commun</source>
          <year>2020</year>
          <volume>11</volume>
          <fpage>4553</fpage>
          <pub-id pub-id-type="doi">10.1038/s41467-020-18360-5</pub-id>
          <pub-id pub-id-type="pmid">32917902</pub-id>
          <pub-id pub-id-type="pmcid">PMC7486382</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B32">
        <label>32</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Criminisi</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Shotton</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Konukoglu</surname>
              <given-names>E</given-names>
            </name>
          </person-group>
          <article-title>Decision forests: a unified framework for classification, regression, density estimation, manifold learning and semi-supervised learning</article-title>
          <source>Found Trends Comput Graph Vis</source>
          <year>2012</year>
          <volume>7</volume>
          <fpage>81</fpage>
          <lpage>227</lpage>
          <pub-id pub-id-type="doi">10.1561/0600000035</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B33">
        <label>33</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Zhang</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Cai</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Cui</surname>
              <given-names>C</given-names>
            </name>
            <name>
              <surname>Qi</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Zhao</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Predicting breast cancer response to neoadjuvant therapy by integrating radiomic and deep-learning features from early-and-peak phases of DCE-MRI</article-title>
          <source>BMC Cancer</source>
          <year>2025</year>
          <volume>25</volume>
          <fpage>1747</fpage>
          <pub-id pub-id-type="doi">10.1186/s12885-025-15095-8</pub-id>
          <pub-id pub-id-type="pmid">41219899</pub-id>
          <pub-id pub-id-type="pmcid">PMC12604257</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B34">
        <label>34</label>
        <nlm-citation publication-type="book">
          <person-group person-group-type="author">
            <name>
              <surname>Martin</surname>
              <given-names>D</given-names>
            </name>
            <name>
              <surname>Fowlkes</surname>
              <given-names>C</given-names>
            </name>
            <name>
              <surname>Tal</surname>
              <given-names>D</given-names>
            </name>
            <name>
              <surname>Malik</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <comment>A database of human segmented natural images and its application to evaluating segmentation algorithms and measuring ecological statistics. In <italic>Proceedings of 8th IEEE International Conference on Computer Vision. ICCV 2001</italic>, Vancouver, Canada. Jul 07-14, 2001. IEEE; 2001. pp. 416-423.</comment>
          <pub-id pub-id-type="doi">10.1109/ICCV.2001.937655</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B35">
        <label>35</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Doukas</surname>
              <given-names>MC</given-names>
            </name>
            <name>
              <surname>Ververas</surname>
              <given-names>E</given-names>
            </name>
            <name>
              <surname>Sharmanska</surname>
              <given-names>V</given-names>
            </name>
            <name>
              <surname>Zafeiriou</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Free-HeadGAN: neural talking head synthesis with explicit gaze control</article-title>
          <source>IEEE Trans Pattern Anal Mach Intell</source>
          <year>2023</year>
          <volume>45</volume>
          <fpage>9743</fpage>
          <lpage>56</lpage>
          <pub-id pub-id-type="doi">10.1109/tpami.2023.3253243</pub-id>
          <pub-id pub-id-type="pmid">37028333</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B36">
        <label>36</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Arar</surname>
              <given-names>NM</given-names>
            </name>
            <name>
              <surname>Gao</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Thiran</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>A regression-based user calibration framework for real-time gaze estimation</article-title>
          <source>IEEE Trans Circuits Syst Video Technol</source>
          <year>2017</year>
          <volume>27</volume>
          <fpage>2623</fpage>
          <lpage>38</lpage>
          <pub-id pub-id-type="doi">10.1109/tcsvt.2016.2595322</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B37">
        <label>37</label>
        <nlm-citation publication-type="book">
          <person-group person-group-type="author">
            <name>
              <surname>Bao</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Liu</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Lu</surname>
              <given-names>F</given-names>
            </name>
          </person-group>
          <comment>Generalizing gaze estimation with rotation consistency. In <italic>2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</italic>, New Orleeans, USA. Jun 18-24, 2022. IEEE; 2022. pp. 4197-206.</comment>
          <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.00417</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B38">
        <label>38</label>
        <nlm-citation publication-type="book">
          <person-group person-group-type="author">
            <name>
              <surname>Zhang</surname>
              <given-names>X</given-names>
            </name>
            <name>
              <surname>Park</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Beeler</surname>
              <given-names>T</given-names>
            </name>
            <name>
              <surname>Bradley</surname>
              <given-names>D</given-names>
            </name>
            <name>
              <surname>Tang</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Hilliges</surname>
              <given-names>O</given-names>
            </name>
          </person-group>
          <comment>ETH-XGaze: a large scale dataset for gaze estimation under extreme head pose and gaze variation. In <italic>Computer Vision - ECCV 2020</italic>. Cham: Springer International Publishing; 2020. pp. 365-81.</comment>
          <pub-id pub-id-type="doi">10.1007/978-3-030-58558-7_22</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B39">
        <label>39</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Kim</surname>
              <given-names>JH</given-names>
            </name>
            <name>
              <surname>Jeong</surname>
              <given-names>JW</given-names>
            </name>
          </person-group>
          <article-title>Gaze in the dark: gaze estimation in a low-light environment with generative adversarial networks</article-title>
          <source>Sensors</source>
          <year>2020</year>
          <volume>20</volume>
          <fpage>4935</fpage>
          <pub-id pub-id-type="doi">10.3390/s20174935</pub-id>
          <pub-id pub-id-type="pmid">32878209</pub-id>
          <pub-id pub-id-type="pmcid">PMC7506593</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B40">
        <label>40</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Ren</surname>
              <given-names>Z</given-names>
            </name>
            <name>
              <surname>Fang</surname>
              <given-names>F</given-names>
            </name>
            <name>
              <surname>Hou</surname>
              <given-names>G</given-names>
            </name>
            <name>
              <surname>Li</surname>
              <given-names>Z</given-names>
            </name>
            <name>
              <surname>Niu</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>Appearance-based gaze estimation with feature fusion of multi-level information elements</article-title>
          <source>J Comput Des Eng</source>
          <year>2023</year>
          <volume>10</volume>
          <fpage>1080</fpage>
          <lpage>109</lpage>
          <pub-id pub-id-type="doi">10.1093/jcde/qwad038</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B41">
        <label>41</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Li</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Huang</surname>
              <given-names>L</given-names>
            </name>
            <name>
              <surname>Chen</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Wang</surname>
              <given-names>X</given-names>
            </name>
            <name>
              <surname>Tan</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Appearance-based gaze estimation method using static transformer temporal differential network</article-title>
          <source>Mathematics</source>
          <year>2023</year>
          <volume>11</volume>
          <fpage>686</fpage>
          <pub-id pub-id-type="doi">10.3390/math11030686</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>