Publications
2024

SalChartQA: Question-driven Saliency on Information Visualisations
Yao Wang, Weitian Wang, Abdullah Abdelhafez, Mayar Elfares, Zhiming Hu, Mihai Bâce, Andreas Bulling
Proc. ACM SIGCHI Conference on Human Factors in Computing Systems (CHI), pp. 1--14, 2024.
AbstractLinksBibTeXProject
Understanding the link between visual attention and user's needs when visually exploring information visualisations is under-explored due to a lack of large and diverse datasets to facilitate these analyses. To fill this gap, we introduce SalChartQA – a novel crowd-sourced dataset that uses the BubbleView interface as a proxy for human gaze and a question-answering (QA) paradigm to induce different information needs in users. SalChartQA contains 74,340 answers to 6,000 questions on 3,000 visualisations. Informed by our analyses demonstrating the tight correlation between the question and visual saliency, we propose the first computational method to predict question-driven saliency on information visualisations. Our method outperforms state-of-the-art saliency models, improving several metrics, such as the correlation coefficient and the Kullback-Leibler divergence. These results show the importance of information needs for shaping attention behaviour and paving the way for new applications, such as task-driven optimisation of visualisations or explainable AI in chart question-answering.
@inproceedings{wang24_chi,
title = {SalChartQA: Question-driven Saliency on Information Visualisations},
author = {Yao Wang and Weitian Wang and Abdullah Abdelhafez and Mayar Elfares and Zhiming Hu and Mihai B{\^a}ce and Andreas Bulling},
year = {2024},
booktitle = {Proc. ACM SIGCHI Conference on Human Factors in Computing Systems (CHI)},
pages = {1--14},
doi = {10.1145/3613904.3642942},
}

Mouse2Vec: Learning Reusable Semantic Representations of Mouse Behaviour
Guanhua Zhang, Zhiming Hu, Mihai Bâce, Andreas Bulling
Proc. ACM SIGCHI Conference on Human Factors in Computing Systems (CHI), pp. 1--17, 2024.
AbstractLinksBibTeXProject
The mouse is a pervasive input device used for a wide range of interactive applications. However, computational modelling of mouse behaviour typically requires time-consuming design and extraction of handcrafted features, or approaches that are application-specific. We instead propose Mouse2Vec – a novel self-supervised method designed to learn semantic representations of mouse behaviour that are reusable across users and applications. Mouse2Vec uses a Transformer-based encoder-decoder architecture, which is specifically geared for mouse data: During pretraining, the encoder learns an embedding of input mouse trajectories while the decoder reconstructs the input and simultaneously detects mouse click events. We show that the representations learned by our method can identify interpretable mouse behaviour clusters and retrieve similar mouse trajectories. We also demonstrate on three sample downstream tasks that the representations can be practically used to augment mouse data for training supervised methods and serve as an effective feature extractor.
@inproceedings{zhang24_chi,
title = {Mouse2Vec: Learning Reusable Semantic Representations of Mouse Behaviour},
author = {Guanhua Zhang and Zhiming Hu and Mihai B{\^a}ce and Andreas Bulling},
year = {2024},
booktitle = {Proc. ACM SIGCHI Conference on Human Factors in Computing Systems (CHI)},
pages = {1--17},
doi = {10.1145/3613904.3642141},
}

Saliency3D: a 3D Saliency Dataset Collected on Screen
Yao Wang, Qi Dai, Mihai Bâce, Karsten Klein, Andreas Bulling
Proc. ACM International Symposium on Eye Tracking Research and Applications (ETRA), pp. 1--9, 2024.
AbstractLinksBibTeXProject
While visual saliency has recently been studied in 3D, the experimental setup for collecting 3D saliency data can be expensive and cumbersome. To address this challenge, we propose a novel experimental design that utilizes an eye tracker on a screen to collect 3D saliency data. Our experimental design reduces the cost and complexity of 3D saliency dataset collection. We first collect gaze data on a screen, then we map them to 3D saliency data through perspective transformation. Using this method, we collect a 3D saliency dataset (49,276 fixations) comprising 10 participants looking at sixteen objects. Moreover, we examine the viewing preferences for objects and discuss our findings in this study. Our results indicate potential preferred viewing directions and a correlation between salient features and the variation in viewing directions.
@inproceedings{wang24_etras,
title = {Saliency3D: a 3D Saliency Dataset Collected on Screen},
author = {Yao Wang and Qi Dai and Mihai B{\^a}ce and Karsten Klein and Andreas Bulling},
year = {2024},
booktitle = {Proc. ACM International Symposium on Eye Tracking Research and Applications (ETRA)},
pages = {1--9},
doi = {10.1145/3649902.3653350},
}

Learning User Embeddings from Human Gaze for Personalised Saliency Prediction
Florian Strohm, Mihai Bâce, Andreas Bulling
Proc. ACM on Human-Computer Interaction (PACM HCI), 8 (ETRA), pp. 1--18, 2024.
AbstractLinksBibTeXProject
Reusable embeddings of user behaviour have shown significant performance improvements for the personalised saliency prediction task. However, prior works require explicit user characteristics and preferences as input, which are often difficult to obtain. We present a novel method to extract user embeddings from pairs of natural images and corresponding saliency maps generated from a small amount of user-specific eye tracking data. At the core of our method is a Siamese convolutional neural encoder that learns the user embeddings by contrasting the image and personal saliency map pairs of different users. Evaluations on two saliency datasets show that the generated embeddings have high discriminative power, are effective at refining universal saliency maps to the individual users, and generalise well across users and images. Finally, based on our model's ability to encode individual user characteristics, our work points towards other applications that can benefit from reusable embeddings of gaze behaviour.
@article{strohm24_etra,
title = {Learning User Embeddings from Human Gaze for Personalised Saliency Prediction},
author = {Florian Strohm and Mihai Bâce and Andreas Bulling},
year = {2024},
journal = {Proc. ACM on Human-Computer Interaction (PACM HCI)},
volume = {8},
number = {ETRA},
pages = {1--18},
doi = {10.1145/3655603},
}

VisRecall++: Analysing and Predicting Visualisation Recallability from Gaze Behaviour
Yao Wang, Yue Jiang, Zhiming Hu, Constantin Ruhdorfer, Mihai Bâce, Andreas Bulling
Proc. ACM on Human-Computer Interaction (PACM HCI), 8 (ETRA), pp. 1--18, 2024.
AbstractLinksBibTeXProject
Question answering has recently been proposed as a promising means to assess the recallability of information visualisations. However, prior works are yet to study the link between visually encoding a visualisation in memory and recall performance. To fill this gap, we propose VisRecall++ – a novel 40-participant recallability dataset that contains gaze data on 200 visualisations and five question types, such as identifying the title, and finding extreme values.We measured recallability by asking participants questions after they observed the visualisation for 10 seconds.Our analyses reveal several insights, such as saccade amplitude, number of fixations, and fixation duration significantly differ between high and low recallability groups.Finally, we propose GazeRecallNet – a novel computational method to predict recallability from gaze behaviour that outperforms several baselines on this task.Taken together, our results shed light on assessing recallability from gaze behaviour and inform future work on recallability-based visualisation optimisation.
@article{wang24_etra,
title = {VisRecall++: Analysing and Predicting Visualisation Recallability from Gaze Behaviour},
author = {Yao Wang and Yue Jiang and Zhiming Hu and Constantin Ruhdorfer and Mihai Bâce and Andreas Bulling},
year = {2024},
journal = {Proc. ACM on Human-Computer Interaction (PACM HCI)},
volume = {8},
number = {ETRA},
pages = {1--18},
doi = {10.1145/3655613},
}

SeFFeC: Semantic Facial Feature Control for Fine-grained Face Editing
Florian Strohm, Mihai Bâce, Markus Kaltenecker, Andreas Bulling
arXiv:2403.13972, pp. 1--18, 2024.
AbstractLinksBibTeXProject
We propose Semantic Facial Feature Control (SeFFeC) - a novel method for fine-grained face shape editing. Our method enables the manipulation of human-understandable, semantic face features, such as nose length or mouth width, which are defined by different groups of facial landmarks. In contrast to existing methods, the use of facial landmarks enables precise measurement of the facial features, which then enables training SeFFeC without any manually annotated labels. SeFFeC consists of a transformer-based encoder network that takes a latent vector of a pre-trained generative model and a facial feature embedding as input, and learns to modify the latent vector to perform the desired face edit operation. To ensure that the desired feature measurement is changed towards the target value without altering uncorrelated features, we introduced a novel semantic face feature loss. Qualitative and quantitative results show that SeFFeC enables precise and fine-grained control of 23 facial features, some of which could not previously be controlled by other methods, without requiring manual annotations. Unlike existing methods, SeFFeC also provides deterministic control over the exact values of the facial features and more localised and disentangled face edits.
@techreport{strohm24_arxiv_2,
title = {SeFFeC: Semantic Facial Feature Control for Fine-grained Face Editing},
author = {Florian Strohm and Mihai Bâce and Markus Kaltenecker and Andreas Bulling},
year = {2024},
pages = {1--18},
url = {https://arxiv.org/abs/2403.13972},
}

DiffGaze: A Diffusion Model for Continuous Gaze Sequence Generation on 360° Images
Chuhan Jiao, Yao Wang, Guanhua Zhang, Mihai Bâce, Zhiming Hu, Andreas Bulling
arXiv:2403.17477, pp. 1--13, 2024.
AbstractLinksBibTeXProject
We present DiffGaze, a novel method for generating realistic and diverse continuous human gaze sequences on 360° images based on a conditional score-based denoising diffusion model. Generating human gaze on 360° images is important for various human-computer interaction and computer graphics applications, e.g. for creating large-scale eye tracking datasets or for realistic animation of virtual humans. However, existing methods are limited to predicting discrete fixation sequences or aggregated saliency maps, thereby neglecting crucial parts of natural gaze behaviour. Our method uses features extracted from 360° images as condition and uses two transformers to model the temporal and spatial dependencies of continuous human gaze. We evaluate DiffGaze on two 360° image benchmarks for gaze sequence generation as well as scanpath prediction and saliency prediction. Our evaluations show that DiffGaze outperforms state-of-the-art methods on all tasks on both benchmarks. We also report a 21-participant user study showing that our method generates gaze sequences that are indistinguishable from real human sequences. Taken together, our evaluations not only demonstrate the effectiveness of DiffGaze but also point towards a new generation of methods that faithfully model the rich spatial and temporal nature of natural human gaze behaviour.
@techreport{jiao24_arxiv,
title = {DiffGaze: A Diffusion Model for Continuous Gaze Sequence Generation on 360° Images},
author = {Chuhan Jiao and Yao Wang and Guanhua Zhang and Mihai B{\^a}ce and Zhiming Hu and Andreas Bulling},
year = {2024},
pages = {1--13},
url = {https://arxiv.org/abs/2403.17477},
}
2023

Scanpath Prediction on Information Visualisations
Yao Wang, Mihai Bâce, Andreas Bulling
IEEE Transactions on Visualization and Computer Graphics (TVCG), 30 (7), pp. 3902--3914, 2023.
AbstractLinksBibTeXProject
We propose Unified Model of Saliency and Scanpaths (UMSS) – a model that learns to predict multi-duration saliency and scanpaths (i.e. sequences of eye fixations) on information visualisations. Although scanpaths provide rich information about the importance of different visualisation elements during the visual exploration process, prior work has been limited to predicting aggregated attention statistics, such as visual saliency. We present in-depth analyses of gaze behaviour for different information visualisation elements (e.g. Title, Label, Data) on the popular MASSVIS dataset. We show that while, overall, gaze patterns are surprisingly consistent across visualisations and viewers, there are also structural differences in gaze dynamics for different elements. Informed by our analyses, UMSS first predicts multi-duration element-level saliency maps, then probabilistically samples scanpaths from them. Extensive experiments on MASSVIS show that our method consistently outperforms state-of-the-art methods with respect tto several, widely used scanpath and saliency evaluation metrics. Our method achieves a relative improvement in sequence score of 11.5 % for scanpath prediction, and a relative improvement in Pearson correlation coefficient of up to 23.6 % for saliency prediction. These results are auspicious and point towards richer user models and simulations of visual attention on visualisations without the need for any eye tracking equipment.
@article{wang23_tvcg,
title = {Scanpath Prediction on Information Visualisations},
author = {Yao Wang and Mihai Bâce and Andreas Bulling},
year = {2023},
journal = {IEEE Transactions on Visualization and Computer Graphics (TVCG)},
volume = {30},
number = {7},
pages = {3902--3914},
doi = {10.1109/TVCG.2023.3242293},
}

Exploring Natural Language Processing Methods for Interactive Behaviour Modelling
Guanhua Zhang, Matteo Bortoletto, Zhiming Hu, Lei Shi, Mihai Bâce, Andreas Bulling
Proc. IFIP TC13 Conference on Human-Computer Interaction (INTERACT), pp. 1--22, 2023.
AbstractLinksBibTeXProject Best Student Paper Nomination
Analysing and modelling interactive behaviour is an important topic in human-computer interaction (HCI) and a key requirement for the development of intelligent interactive systems. Interactive behaviour has a sequential (actions happen one after another) and hierarchical (a sequence of actions forms an activity driven by interaction goals) structure, which may be similar to the structure of natural language. Designed based on such a structure, natural language processing (NLP) methods have achieved groundbreaking success in various downstream tasks. However, few works linked interactive behaviour with natural language. In this paper, we explore the similarity between interactive behaviour and natural language by applying an NLP method, byte pair encoding (BPE), to encode mouse and keyboard behaviour. We then analyse the vocabulary, i.e., the set of action sequences, learnt by BPE, as well as use the vocabulary to encode the input behaviour for interactive task recognition. An existing dataset collected in constrained lab settings and our novel out-of-the-lab dataset were used for evaluation. Results show that this natural language-inspired approach not only learns action sequences that reflect specific interaction goals, but also achieves higher F1 scores on task recognition than other methods. Our work reveals the similarity between interactive behaviour and natural language, and presents the potential of applying the new pack of methods that leverage insights from NLP to model interactive behaviour in HCI.
@inproceedings{zhang23_interact,
title = {Exploring Natural Language Processing Methods for Interactive Behaviour Modelling},
author = {Zhang, Guanhua and Bortoletto, Matteo and Hu, Zhiming and Shi, Lei and B{\^a}ce, Mihai and Bulling, Andreas},
year = {2023},
booktitle = {Proc. IFIP TC13 Conference on Human-Computer Interaction (INTERACT)},
pages = {1--22},
publisher = {Springer},
}

Multimodal Integration of Human-Like Attention in Visual Question Answering
Ekta Sood, Fabian Kögel, Philipp Müller, Dominike Thomas, Mihai Bâce, Andreas Bulling
Proc. Workshop on Gaze Estimation and Prediction in the Wild (GAZE), CVPRW, pp. 2647--2657, 2023.
AbstractLinksBibTeXProject Tobii Sponsor Award, Oral Presentation
Human-like attention as a supervisory signal to guide neural attention has shown significant promise but is currently limited to uni-modal integration – even for inherently multi-modal tasks such as visual question answering (VQA). We present the Multimodal Human-like Attention Network (MULAN) – the first method for multimodal integration of human-like attention on image and text during training of VQA models. MULAN integrates attention predictions from two state-of-the-art text and image saliency models into neural self-attention layers of a recent transformer-based VQA model. Through evaluations on the challenging VQAv2 dataset, we show that MULAN achieves a new state-of-the-art performance of 73.98% accuracy on test-std and 73.72% on test-dev and, at the same time, has approximately 80% fewer trainable parameters than prior work. Overall, our work underlines the potential of integrating multimodal human-like and neural attention for VQA.
@inproceedings{sood23_gaze,
title = {Multimodal Integration of Human-Like Attention in Visual Question Answering},
author = {Sood, Ekta and Fabian Kögel and Philipp Müller and Dominike Thomas and Mihai Bâce and Andreas Bulling},
year = {2023},
booktitle = {Proc. Workshop on Gaze Estimation and Prediction in the Wild (GAZE), CVPRW},
pages = {2647--2657},
url = {https://openaccess.thecvf.com/content/CVPR2023W/GAZE/papers/Sood_Multimodal_Integration_of_Human-Like_Attention_in_Visual_Question_Answering_CVPRW_2023_paper.pdf},
}

Int-HRL: Towards Intention-based Hierarchical Reinforcement Learning
Anna Penzkofer, Simon Schaefer, Florian Strohm, Mihai Bâce, Stefan Leutenegger, Andreas Bulling
Proc. Adaptive and Learning Agents Workshop (ALA), pp. 1--7, 2023.
AbstractLinksBibTeXProject
While deep reinforcement learning (RL) agents outperform humans on an increasing number of tasks, training them requires data equivalent to decades of human gameplay. Recent hierarchical RL methods have increased sample efficiency by incorporating information inherent to the structure of the decision problem but at the cost of having to discover or use human-annotated sub-goals that guide the learning process. We show that intentions of human players, i.e. the precursor of goal-oriented decisions, can be robustly predicted from eye gaze even for the long-horizon sparse rewards task of Montezuma’s Revenge – one of the most challenging RL tasks in the Atari2600 game suite. We propose Int-HRL: Hierarchical RL with intention-based sub-goals that are inferred from human eye gaze. Our novel sub-goal extraction pipeline is fully automatic and replaces the need for manual sub-goal annotation by human experts. Our evaluations show that replacing hand-crafted sub-goals with automatically extracted intentions leads to a HRL agent that is significantly more sample efficient than previous methods.
@inproceedings{penzkofer23_ala,
title = {Int-HRL: Towards Intention-based Hierarchical Reinforcement Learning},
author = {Anna Penzkofer and Simon Schaefer and Florian Strohm and Mihai Bâce and Stefan Leutenegger and Andreas Bulling},
year = {2023},
booktitle = {Proc. Adaptive and Learning Agents Workshop (ALA)},
pages = {1--7},
}

Facial Composite Generation with Iterative Human Feedback
Florian Strohm, Ekta Sood, Dominike Thomas, Mihai Bâce, Andreas Bulling
Proc. The 1st Gaze Meets ML workshop, PMLR, pp. 165--183, 2023.
AbstractLinksBibTeXProject
We propose the first method in which human and AI collaborate to iteratively reconstruct the human’s mental image of another person’s face only from their eye gaze. Current tools for generating digital human faces involve a tedious and time-consuming manual design process. While gaze-based mental image reconstruction represents a promising alternative, previous methods still assumed prior knowledge about the target face, thereby severely limiting their practical usefulness. The key novelty of our method is a collaborative, it- erative query engine: Based on the user’s gaze behaviour in each iteration, our method predicts which images to show to the user in the next iteration. Results from two human studies (N=12 and N=22) show that our method can visually reconstruct digital faces that are more similar to the mental image, and is more usable compared to other methods. As such, our findings point at the significant potential of human-AI collaboration for recon- structing mental images, potentially also beyond faces, and of human gaze as a rich source of information and a powerful mediator in said collaboration.
@inproceedings{strohm23_gmml,
title = {Facial Composite Generation with Iterative Human Feedback},
author = {Strohm, Florian and Sood, Ekta and Thomas, Dominike and B{\^a}ce, Mihai and Bulling, Andreas},
editor = {Lourentzou, Ismini and Wu, Joy and Kashyap, Satyananda and Karargyris, Alexandros and Celi, Leo Anthony and Kawas, Ban and Talathi, Sachin},
year = {2023},
booktitle = {Proc. The 1st Gaze Meets ML workshop, PMLR},
volume = {210},
pages = {165--183},
url = {https://proceedings.mlr.press/v210/strohm23a.html},
publisher = {PMLR},
series = {Proceedings of Machine Learning Research},
pdf = {https://proceedings.mlr.press/v210/strohm23a/strohm23a.pdf},
}

SUPREYES: SUPer Resolution for EYES Using Implicit Neural Representation Learning
Chuhan Jiao, Zhiming Hu, Mihai Bâce, Andreas Bulling
Proc. ACM Symposium on User Interface Software and Technology (UIST), pp. 1--13, 2023.
AbstractLinksBibTeXProject
We introduce SUPREYES – a novel self-supervised method to increase the spatio-temporal resolution of gaze data recorded using low(er)-resolution eye trackers. Despite continuing advances in eye tracking technology, the vast majority of current eye trackers – particularly mobile ones and those integrated into mobile devices – suffer from low-resolution gaze data, thus fundamentally limiting their practical usefulness. SUPREYES learns a continuous implicit neural representation from low-resolution gaze data to up-sample the gaze data to arbitrary resolutions. We compare our method with commonly used interpolation methods on arbitrary scale super-resolution and demonstrate that SUPREYES outperforms these baselines by a significant margin. We also test on
the sample downstream task of gaze-based user identification and show that our method improves the performance of original low-resolution gaze data and outperforms other baselines. These results are promising as they open up a new direction for increasing eye tracking fidelity as well as enabling new gaze-based applications without the need for new eye tracking equipment.
@inproceedings{jiao23_uist,
title = {SUPREYES: SUPer Resolution for EYES Using Implicit Neural Representation Learning},
author = {Jiao, Chuhan and Hu, Zhiming and B{\^a}ce, Mihai and Bulling, Andreas},
year = {2023},
booktitle = {Proc. ACM Symposium on User Interface Software and Technology (UIST)},
pages = {1--13},
doi = {10.1145/3586183.3606780},
}

Usable and Fast Interactive Mental Face Reconstruction
Florian Strohm, Mihai Bâce, Andreas Bulling
Proc. ACM Symposium on User Interface Software and Technology (UIST), pp. 1--15, 2023.
AbstractLinksBibTeXProject
We introduce an end-to-end interactive system for mental face reconstruction – the challenging task of visually reconstructing a face image a person only has in their mind. In contrast to existing
methods that suffer from low usability and high mental load, our approach only requires the user to rank images over multiple iterations according to the perceived similarity with their mental image.
Based on these rankings, our mental face reconstruction system extracts image features in each iteration, combines them into a joint feature vector, and then uses a generative model to visually reconstruct the mental image. To avoid the need for collecting large
amounts of human training data, we further propose a computational user model that can simulate human ranking behaviour using data from an online crowd-sourcing study (N=215). Results from a 12-participant user study show that our method can reconstruct mental images that are visually similar to existing approaches but has significantly higher usability, lower perceived workload, and
is 40% faster. In addition, results from a third 22-participant lineup study in which we validated our reconstructions on a face ranking task show a identification rate of 55.3%, which is in line with prior work. These results represent an important step towards new interactive intelligent systems that can robustly and effortlessly reconstruct a user’s mental image.
@inproceedings{strohm23_uist,
title = {Usable and Fast Interactive Mental Face Reconstruction},
author = {Strohm, Florian and B{\^a}ce, Mihai and Bulling, Andreas},
year = {2023},
booktitle = {Proc. ACM Symposium on User Interface Software and Technology (UIST)},
pages = {1--15},
doi = {https://doi.org/10.1145/3586183.3606795},
}
2022

Designing for Noticeability: The Impact of Visual Importance on Desktop Notifications
Philipp Müller, Sander Staal, Mihai Bâce, Andreas Bulling
Proc. ACM SIGCHI Conference on Human Factors in Computing Systems (CHI), pp. 1--13, 2022.
AbstractLinksBibTeXProject
Desktop notifications should be noticeable but are also subject to a number of design choices, e.g. concerning their size, placement, or opacity. It is currently unknown, however, how these choices interact with the desktop background and their influence on noticeability. To address this limitation, we introduce a software tool to automatically synthesize realistically looking desktop images for major operating systems and applications. Using these images, we present a user study (N=34) to investigate the noticeability of notifications during a primary task. We are first to show that visual importance of the background at the notification location significantly impacts whether users detect notifications. We analyse the utility of visual importance to compensate for suboptimal design choices with respect to noticeability, e.g. small notification size. Finally, we introduce noticeability maps - 2D maps encoding the predicted noticeability across the desktop and inform designers how to trade-off notification design and noticeability.
@inproceedings{mueller22_chi,
title = {Designing for Noticeability: The Impact of Visual Importance on Desktop Notifications},
author = {Philipp Müller and Sander Staal and Mihai B{\^a}ce and Andreas Bulling},
year = {2022},
booktitle = {Proc. ACM SIGCHI Conference on Human Factors in Computing Systems (CHI)},
pages = {1--13},
doi = {10.1145/3491102.3501954},
}

PrivacyScout: Assessing Vulnerability to Shoulder Surfing on Mobile Devices
Mihai Bâce, Alia Saad, Mohamed Khamis, Stefan Schneegass, Andreas Bulling
Proc. on Privacy Enhancing Technologies (PETs), pp. 650--669, 2022.
AbstractLinksBibTeXProject
One approach to mitigate shoulder surfing attacks on mobile devices is to detect the presence of a bystander using the phone’s front-facing camera. However, a person’s face in the camera’s field of view does not always indicate an attack. To overcome this limitation, in a novel data collection study (N=16), we analysed the influence of three viewing angles and four distances on the success of shoulder surfing attacks. In contrast to prior works that mainly focused on user authentication, we investigated three common types of content susceptible to shoulder surfing: text, photos, and PIN authentications. We show that the vulnerability of text and photos depends on the observer’s location relative to the device, while PIN authentications are vulnerable independent of the observation location. We then present PrivacyScout - a novel method that predicts the shoulder-surfing risk based on visual features extracted from the observer’s face as captured by the front-facing camera. Finally, evaluations from our data collection study demonstrate our method’s feasibility to assess the risk of a shoulder surfing attack more accurately.
@inproceedings{bace22_pets,
title = {PrivacyScout: Assessing Vulnerability to Shoulder Surfing on Mobile Devices},
author = {Mihai B{\^a}ce and Alia Saad and Mohamed Khamis and Stefan Schneegass and Andreas Bulling},
year = {2022},
booktitle = {Proc. on Privacy Enhancing Technologies (PETs)},
pages = {650--669},
doi = {10.56553/popets-2022-0090},
issue = {3},
}

Predicting Next Actions and Latent Intents during Text Formatting
Guanhua Zhang, Susanne Hindennach, Jan Leusmann, Felix Bühler, Benedict Steuerlein, Sven Mayer, Mihai Bâce, Andreas Bulling
Proc. the CHI Workshop Computational Approaches for Understanding, Generating, and Adapting User Interfaces, pp. 1--6, 2022.
AbstractLinksBibTeXProject
In this work we investigate the challenging task of predicting user intents from mouse and keyboard input as well as gaze behaviour. In contrast to prior work we study intent prediction at two different resolutions on the behavioural timeline: predicting future input actions as well as latent intents to achieve a high-level interaction goal. Results from a user study (N=15) on a sample text formatting task show that the sequence of prior actions is more informative for intent prediction than gaze. Only using the action sequence, we can predict the next action and the high-level intent with an accuracy of 66% and 96%, respectively. In contrast, accuracy when using features extracted from gaze behaviour was significantly lower, at 41% and 46%. This finding is important for the development of future anticipatory user interfaces that aim to proactively adapt to user intents and interaction goals.
@inproceedings{zhang22_caugaui,
title = {Predicting Next Actions and Latent Intents during Text Formatting},
author = {Guanhua Zhang and Susanne Hindennach and Jan Leusmann and Felix Bühler and Benedict Steuerlein and Sven Mayer and Mihai Bâce and Andreas Bulling},
year = {2022},
booktitle = {Proc. the CHI Workshop Computational Approaches for Understanding, Generating, and Adapting User Interfaces},
pages = {1--6},
}

VisRecall: Quantifying Information Visualisation Recallability via Question Answering
Yao Wang, Chuhan Jiao, Mihai Bâce, Andreas Bulling
IEEE Transactions on Visualization and Computer Graphics (TVCG), 28 (12), pp. 4995-5005, 2022.
AbstractLinksBibTeXProject
Despite its importance for assessing the effectiveness of communicating information visually, fine-grained recallability of information visualisations has not been studied quantitatively so far. In this work, we propose a question-answering paradigm to study visualisation recallability and present VisRecall - a novel dataset consisting of 200 visualisations that are annotated with crowd-sourced human (N = 305) recallability scores obtained from 1,000 questions of five question types. Furthermore, we present the first computational method to predict recallability of different visualisation elements, such as the title or specific data values. We report detailed analyses of our method on VisRecall and demonstrate that it outperforms several baselines in overall recallability and FE-, F-, RV-, and U-question recallability. Our work makes fundamental contributions towards a new generation of methods to assist designers in optimising visualisations.
@article{wang22_tvcg,
title = {VisRecall: Quantifying Information Visualisation Recallability via Question Answering},
author = {Yao Wang and Chuhan Jiao and Mihai Bâce and Andreas Bulling},
year = {2022},
journal = {IEEE Transactions on Visualization and Computer Graphics (TVCG)},
volume = {28},
number = {12},
pages = {4995-5005},
doi = {10.1109/TVCG.2022.3198163},
}

Neuro-Symbolic Visual Dialog
Adnen Abdessaied, Mihai Bâce, Andreas Bulling
Proc. 29th Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING), pp. 1--11, 2022.
AbstractLinksBibTeXProject
We propose Neuro-Symbolic Visual Dialog (NSVD) —the first method to combine deep learning and symbolic program execution for multi-round visually-grounded reasoning. NSVD significantly outperforms existing purely-connectionist methods on two key challenges inherent to visual dialog: long-distance co-reference resolution as well as vanishing question-answering performance. We demonstrate the latter by proposing a more realistic and stricter evaluation scheme in which we use predicted answers for the full dialog history when calculating accuracy. We describe two variants of our model and show that using this new scheme, our best model achieves an accuracy of 99.72% on CLEVR-Dialog —a relative improvement of more than 10% over the state of the art —while only requiring a fraction of training data. Moreover, we demonstrate that our neuro-symbolic models have a higher mean first failure round, are more robust against incomplete dialog histories, and generalise better not only to dialogs that are up to three times longer than those seen during training but also to unseen question types and scenes.
@inproceedings{abdessaied22_coling,
title = {Neuro-Symbolic Visual Dialog},
author = {Adnen Abdessaied and Mihai Bâce and Andreas Bulling},
year = {2022},
booktitle = {Proc. 29th Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING)},
pages = {1--11},
}

Impact of Gaze Uncertainty on AOIs in Information Visualisations
Yao Wang, Maurice Koch, Mihai Bâce, Daniel Weiskopf, Andreas Bulling
ETRA Workshop on Eye Tracking and Visualization (ETVIS), pp. 1--6, 2022.
AbstractLinksBibTeXProject
Gaze-based analysis of areas of interest (AOIs) is widely used in information visualisation research to understand how people explore visualisations or assess the quality of visualisations concerning key characteristics such as memorability. However, nearby AOIs in visualisations amplify the uncertainty caused by the gaze estimation error, which strongly influences the mapping between gaze samples or fixations and different AOIs. We contribute a novel investigation into gaze uncertainty and quantify its impact on AOI-based analysis on visualisations using two novel metrics: the Flipping Candidate Rate (FCR) and Hit Any AOI Rate (HAAR). Our analysis of 40 real-world visualisations, including human gaze and AOI annotations, shows that gaze uncertainty frequently and significantly impacts the analysis conducted in AOI-based studies. Moreover, we analysed four visualisation types and found that bar and scatter plots are usually designed in a way that causes more uncertainty than line and pie plots in gaze-based analysis.
@inproceedings{wang22_etvis,
title = {Impact of Gaze Uncertainty on AOIs in Information Visualisations},
author = {Yao Wang and Maurice Koch and Mihai B{\^a}ce and Daniel Weiskopf and Andreas Bulling},
year = {2022},
booktitle = {ETRA Workshop on Eye Tracking and Visualization (ETVIS)},
pages = {1--6},
doi = {10.1145/3517031.3531166},
}
2021

Neural Photofit: Gaze-based Mental Image Reconstruction
Florian Strohm, Ekta Sood, Sven Mayer, Philipp Müller, Mihai Bâce, Andreas Bulling
Proc. IEEE International Conference on Computer Vision (ICCV), pp. 245-254, 2021.
AbstractLinksBibTeXProject
We propose a novel method that leverages human fixations to visually decode the image a person has in mind into a photofit (facial composite). Our method combines three neural networks: An encoder, a scoring network, and a decoder. The encoder extracts image features and predicts a neural activation map for each face looked at by a human observer. A neural scoring network compares the human and neural attention and predicts a relevance score for each extracted image feature. Finally, image features are aggregated into a single feature vector as a linear combination of all features weighted by relevance which a decoder decodes into the final photofit. We train the neural scoring network on a novel dataset containing gaze data of 19 participants looking at collages of synthetic faces. We show that our method significantly outperforms a mean baseline predictor and report on a human study that shows that we can decode photofits that are visually plausible and close to the observer's mental image. Code and dataset available upon request.
Code: Available upon request.
Dataset: Available upon request.
@inproceedings{strohm21_iccv,
title = {Neural Photofit: Gaze-based Mental Image Reconstruction},
author = {Florian Strohm and Ekta Sood and Sven Mayer and Philipp Müller and Mihai Bâce and Andreas Bulling},
year = {2021},
booktitle = {Proc. IEEE International Conference on Computer Vision (ICCV)},
pages = {245-254},
doi = {10.1109/ICCV48922.2021.00031},
}
2020

Combining Gaze Estimation and Optical Flow for Pursuits Interaction
Mihai Bâce, Vincent Becker, Chenyang Wang, Andreas Bulling
Proc. ACM International Symposium on Eye Tracking Research and Applications (ETRA), pp. 1-10, 2020.
AbstractLinksBibTeXProject Best Paper Award
Pursuit eye movements have become widely popular because they enable spontaneous eye-based interaction. However, existing methods to detect smooth pursuits require special-purpose eye trackers. We propose the first method to detect pursuits using a single off-the-shelf RGB camera in unconstrained remote settings. The key novelty of our method is that it combines appearance-based gaze estimation with optical flow in the eye region to jointly analyse eye movement dynamics in a single pipeline. We evaluate the performance and robustness of our method for different numbers of targets and trajectories in a 13-participant user study. We show that our method not only outperforms the current state of the art but also achieves competitive performance to a consumer eye tracker for a small number of targets. As such, our work points towards a new family of methods for pursuit interaction directly applicable to an ever-increasing number of devices readily equipped with cameras.
@inproceedings{bace20_etra,
title = {Combining Gaze Estimation and Optical Flow for Pursuits Interaction},
author = {Mihai B{\^a}ce and Vincent Becker and Chenyang Wang and Andreas Bulling},
year = {2020},
booktitle = {Proc. ACM International Symposium on Eye Tracking Research and Applications (ETRA)},
pages = {1-10},
doi = {10.1145/3379155.3391315},
}

How far are we from quantifying visual attention in mobile HCI?
Mihai Bâce, Sander Staal, Andreas Bulling
IEEE Pervasive Computing, 19 (2), pp. 46-55, 2020.
AbstractLinksBibTeXProject
With an ever-increasing number of mobile devices competing for attention, quantifying when, how often, or for how long users look at their devices has emerged as a key challenge in mobile human-computer interaction. Encouraged by recent advances in automatic eye contact detection using machine learning and device-integrated cameras, we provide a fundamental investigation into the feasibility of quantifying overt visual attention during everyday mobile interactions. We discuss the main challenges and sources of error associated with sensing visual attention on mobile devices in the wild, including the impact of face and eye visibility, the importance of robust head pose estimation, and the need for accurate gaze estimation. Our analysis informs future research on this emerging topic and underlines the potential of eye contact detection for exciting new applications towards next-generation pervasive attentive user interfaces.
@article{bace20_pcm,
title = {How far are we from quantifying visual attention in mobile HCI?},
author = {Mihai B{\^a}ce and Sander Staal and Andreas Bulling},
year = {2020},
journal = {IEEE Pervasive Computing},
volume = {19},
number = {2},
pages = {46-55},
doi = {10.1109/MPRV.2020.2967736},
}

Quantification of Users' Visual Attention During Everyday Mobile Device Interactions
Mihai Bâce, Sander Staal, Andreas Bulling
Proc. ACM SIGCHI Conference on Human Factors in Computing Systems (CHI), pp. 1--14, 2020.
AbstractLinksBibTeXProject
We present the first real-world dataset and quantitative evaluation of visual attention of mobile device users \textitin-situ, i.e. while using their devices during everyday routine. Understanding user attention is a core research challenge in mobile HCI but previous approaches relied on usage logs or self-reports that are only proxies and consequently do neither reflect attention completely nor accurately. Our evaluations are based on \textitEveryday Mobile Visual Attention (EMVA) – a new 32-participant dataset containing around 472 hours of video snippets recorded over more than two weeks in real life using the front-facing camera as well as associated usage logs, interaction events, and sensor data. Using an eye contact detection method, we are first to quantify the highly dynamic nature of everyday visual attention across users, mobile applications, and usage contexts. We discuss key insights from our analyses that highlight the potential and inform the design of future mobile attentive user interfaces.
@inproceedings{bace20_chi,
title = {Quantification of Users' Visual Attention During Everyday Mobile Device Interactions},
author = {Mihai B{\^a}ce and Sander Staal and Andreas Bulling},
year = {2020},
booktitle = {Proc. ACM SIGCHI Conference on Human Factors in Computing Systems (CHI)},
pages = {1--14},
doi = {10.1145/3313831.3376449},
video = {https://www.youtube.com/watch?v=SzLn3LujIqw},
}
2018
Wearable Eye Tracker Calibration at Your Fingertips
Mihai Bâce, Sander Staal, Gábor Sörös
ACM Symposium on Eye Tracking Research & Applications, 2018.
LinksBibTeXProject
@inproceedings{bace18_etra,
title = {Wearable Eye Tracker Calibration at Your Fingertips},
author = {Mihai B{\^a}ce and Sander Staal and G{\'a}bor S{\"o}r{\"o}s},
year = {2018},
booktitle = {ACM Symposium on Eye Tracking Research & Applications},
doi = {10.1145/3204493.3204592},
publisher = {ACM},
address = {Warsaw, Poland},
series = {ETRA '18},
}
2017
Augmenting Human Interaction Capabilities with Proximity, Natural Gestures, and Eye Gaze
Mihai Bâce
Proceedings of the 19th ACM International Conference on Human-Computer Interaction with Mobile Devices and Services (MobileHCI 2017), 2017.
LinksBibTeXProject
@inproceedings{bace17_mobilehci,
title = {Augmenting Human Interaction Capabilities with Proximity, Natural Gestures, and Eye Gaze},
author = {Mihai B{\^a}ce},
year = {2017},
booktitle = {Proceedings of the 19th ACM International Conference on Human-Computer Interaction with Mobile Devices and Services (MobileHCI 2017)},
doi = {10.1145/3098279.3119924},
publisher = {ACM},
address = {Vienna, Austria},
series = {MobileHCI '17},
isbn = {978-1-4503-4835-5},
}
Facilitating Object Detection and Recognition through Eye Gaze
Mihai Bâce, Philippe Schlattner, Vincent Becker, Gábor Sörös
Proceedings of the Workshop on Object Recognition for Input and Mobile Interaction at the 19th ACM International Conference on Human-Computer Interaction with Mobile Devices and Services (MobileHCI 2017), 2017.
LinksBibTeXProject
@inproceedings{bace17_mobilehci_2,
title = {Facilitating Object Detection and Recognition through Eye Gaze},
author = {Mihai B{\^a}ce and Philippe Schlattner and Vincent Becker and G{\'a}bor S{\"o}r{\"o}s},
year = {2017},
booktitle = {Proceedings of the Workshop on Object Recognition for Input and Mobile Interaction at the 19th ACM International Conference on Human-Computer Interaction with Mobile Devices and Services (MobileHCI 2017)},
doi = {10.3929/ethz-b-000221545},
publisher = {ACM},
address = {Vienna, Austria},
series = {MobileHCI '17},
isbn = {978-1-4503-4835-5},
}
Collocated Multi-user Gestural Interactions with Unmodified Wearable Devices
Mihai Bâce, Sander Staal, Gábor Sörös, Giorgio Corbellini
Augmented Human Research, 2, 2017.
LinksBibTeXProject
@article{bace17_ahr,
title = {Collocated Multi-user Gestural Interactions with Unmodified Wearable Devices},
author = {Mihai B{\^a}ce and Sander Staal and G{\'a}bor S{\"o}r{\"o}s and Giorgio Corbellini},
year = {2017},
journal = {Augmented Human Research},
volume = {2},
doi = {10.1007/s41133-017-0009-z},
}
HandshakAR: Wearable Augmented Reality System for Effortless Information Sharing
Mihai Bâce, Gábor Sörös, Sander Staal, Giorgio Corbellini
Proceedings of the Augmented Human 2017 Conference (AH 2017), 2017.
LinksBibTeXProject
@inproceedings{bace17_ah,
title = {HandshakAR: Wearable Augmented Reality System for Effortless Information Sharing},
author = {Mihai B{\^a}ce and G{\'a}bor S{\"o}r{\"o}s and Sander Staal and Giorgio Corbellini},
year = {2017},
booktitle = {Proceedings of the Augmented Human 2017 Conference (AH 2017)},
doi = {10.1145/3041164.3041203},
publisher = {ACM},
address = {Mountain View, CA, USA},
series = {AH '17},
isbn = {978-1-4503-4835-5},
}
2016
ubiGaze: Ubiquitous Augmented Reality Messaging Using Gaze Gestures
Mihai Bâce, Teemu Leppänen, Argenis Ramirez Gomez, David Gil de Gomez
SIGGRAPH ASIA 2016 Mobile Graphics and Interactive Applications, pp. 11:1--11:5, 2016.
LinksBibTeXProject
@inproceedings{bace16_mgia,
title = {ubiGaze: Ubiquitous Augmented Reality Messaging Using Gaze Gestures},
author = {Mihai B{\^a}ce and Teemu Lepp{\"a}nen and Argenis Ramirez Gomez and David Gil de Gomez},
year = {2016},
booktitle = {SIGGRAPH ASIA 2016 Mobile Graphics and Interactive Applications},
pages = {11:1--11:5},
doi = {10.1145/2999508.2999530},
publisher = {ACM},
address = {Macau},
series = {SA '16},
isbn = {978-1-4503-4551-4},
}
2015
Lightweight Indoor Localization System
Mihai Bâce, Yvonne-Anne Pignolet
Proceedings of the 8th IFIP Wireless and Mobile Networking Conference (WMNC 2015). Munich, Germany., pp. 160--167, 2015.
LinksBibTeXProject
@inproceedings{bace15_wmnc,
title = {Lightweight Indoor Localization System},
author = {Mihai B{\^a}ce and Yvonne-Anne Pignolet},
year = {2015},
booktitle = {Proceedings of the 8th IFIP Wireless and Mobile Networking Conference (WMNC 2015). Munich, Germany.},
pages = {160--167},
doi = {10.1109/WMNC.2015.22},
}
2011
Lane identification and ego-vehicle accurate global positioning in intersections
Voichita Popescu, Mihai Bâce, Sergiu Nedevschi
Intelligent Vehicles Symposium (IV 2011). Baden-Baden, Germany., pp. 870 - 875, 2011.
LinksBibTeXProject
@inproceedings{bace11_iv,
title = {Lane identification and ego-vehicle accurate global positioning in intersections},
author = {Voichita Popescu and Mihai B{\^a}ce and Sergiu Nedevschi},
year = {2011},
booktitle = {Intelligent Vehicles Symposium (IV 2011). Baden-Baden, Germany.},
pages = {870 - 875},
doi = {10.1109/IVS.2011.5940523},
publisher = {IEEE},
isbn = {978-1-4577-0890-9},
}
Probabilistic Approach for Automated Reasoning for Lane Identification in Intelligent Vehicles
Voichita Popescu, Mihai Bâce, Sergiu Nedevschi
International Symposium on Symbolic and Numeric Algorithms for Scientific Computing (SYNASC 2011). Timisoara, Romania., pp. 255 - 258, 2011.
LinksBibTeXProject
@inproceedings{bace11_synasc,
title = {Probabilistic Approach for Automated Reasoning for Lane Identification in Intelligent Vehicles},
author = {Voichita Popescu and Mihai B{\^a}ce and Sergiu Nedevschi},
year = {2011},
booktitle = {International Symposium on Symbolic and Numeric Algorithms for Scientific Computing (SYNASC 2011). Timisoara, Romania.},
pages = {255 - 258},
doi = {10.1109/SYNASC.2011.10},
publisher = {IEEE},
isbn = {978-1-4673-0207-4},
}