peyes 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. peyes/_DataModels/DatasetLoader.py +637 -0
  2. peyes/_DataModels/Detector.py +1609 -0
  3. peyes/_DataModels/Event.py +538 -0
  4. peyes/_DataModels/EventLabelEnum.py +14 -0
  5. peyes/_DataModels/EventMatcher.py +302 -0
  6. peyes/_DataModels/UnparsedEventLabel.py +9 -0
  7. peyes/_DataModels/config.py +57 -0
  8. peyes/__init__.py +18 -0
  9. peyes/_base/create.py +257 -0
  10. peyes/_base/match.py +127 -0
  11. peyes/_base/parse.py +52 -0
  12. peyes/_base/postprocess_events.py +53 -0
  13. peyes/_base/set_config.py +119 -0
  14. peyes/_utils/constants.py +117 -0
  15. peyes/_utils/event_utils.py +104 -0
  16. peyes/_utils/metric_utils.py +111 -0
  17. peyes/_utils/pixel_utils.py +206 -0
  18. peyes/_utils/vector_utils.py +101 -0
  19. peyes/_utils/visualization_utils.py +173 -0
  20. peyes/alignment_metrics/__init__.py +3 -0
  21. peyes/alignment_metrics/_signal_detection_metrics.py +178 -0
  22. peyes/alignment_metrics/_timing_differences.py +95 -0
  23. peyes/datasets/__init__.py +52 -0
  24. peyes/datasets/get_metadata.py +45 -0
  25. peyes/datasets/load_dataset.py +33 -0
  26. peyes/event_metrics/__init__.py +45 -0
  27. peyes/event_metrics/_get_features.py +71 -0
  28. peyes/event_metrics/_rates_and_transitions.py +91 -0
  29. peyes/match_metrics/__init__.py +42 -0
  30. peyes/match_metrics/_get_features.py +63 -0
  31. peyes/match_metrics/_match_evaluation.py +154 -0
  32. peyes/sample_metrics/__init__.py +125 -0
  33. peyes/sample_metrics/_calculate_metrics.py +127 -0
  34. peyes/sample_metrics/_counts_and_matrices.py +77 -0
  35. peyes/visualize/__init__.py +6 -0
  36. peyes/visualize/_event_summary.py +283 -0
  37. peyes/visualize/_features.py +213 -0
  38. peyes/visualize/_gaze.py +293 -0
  39. peyes/visualize/_scarfplot.py +105 -0
  40. peyes/visualize/_video.py +139 -0
  41. peyes-0.2.0.dist-info/METADATA +123 -0
  42. peyes-0.2.0.dist-info/RECORD +43 -0
  43. peyes-0.2.0.dist-info/WHEEL +4 -0
@@ -0,0 +1,637 @@
1
+ import os
2
+ import io
3
+ import itertools
4
+ import warnings
5
+ import zipfile as zp
6
+ import posixpath as psx
7
+ from typing import final, List, Tuple, Dict, Union
8
+ from abc import ABC, abstractmethod
9
+
10
+
11
+ import numpy as np
12
+ import pandas as pd
13
+ import requests as req
14
+ from tqdm import tqdm
15
+ from scipy.io import loadmat
16
+ from scipy.interpolate import interp1d
17
+ import arff
18
+
19
+ import peyes._utils.constants as cnst
20
+ from peyes._utils.pixel_utils import calculate_pixel_size, visual_angle_to_pixels
21
+ from peyes._utils.event_utils import parse_label
22
+ from peyes._DataModels.EventLabelEnum import EventLabelEnum
23
+
24
+
25
+ class BaseDatasetLoader(ABC):
26
+ _NAME: str
27
+ _URL: str
28
+ _ARTICLES: List[str]
29
+ _LICENSE: str
30
+ _INDEXERS: List[str] = [
31
+ cnst.TRIAL_ID_STR, cnst.SUBJECT_ID_STR, cnst.STIMULUS_TYPE_STR, cnst.STIMULUS_NAME_STR
32
+ ]
33
+ _DOWNLOAD_TIMEOUT_SEC: float = 60.0 # without this, a stalled connection hangs indefinitely
34
+
35
+ @classmethod
36
+ @final
37
+ def load(cls, directory: str, save: bool = False, verbose: bool = False) -> pd.DataFrame:
38
+ """
39
+ Loads the dataset from the specified directory. If the dataset is not found, it is downloaded from the internet.
40
+ If `save` is True and the dataset was downloaded, it is saved to the specified directory.
41
+ if `verbose` is True, a progress bar is displayed while parsing the downloaded dataset.
42
+ :return: a DataFrame containing the dataset
43
+ :raises ValueError: if `save` is True and `directory` is not specified
44
+ """
45
+ if save and not directory:
46
+ raise ValueError("Directory must be specified to save the dataset")
47
+ # S-6: check the cache file exists explicitly, rather than using exception flow (catching
48
+ # FileNotFoundError/TypeError) for the cache-miss path - a truncated/unpickleable cache file used
49
+ # to propagate through pd.read_pickle as an opaque, uncaught error; now the download path only
50
+ # triggers when `directory`/the cache file genuinely aren't there.
51
+ cache_path = os.path.join(directory, f"{cls.name()}.pkl") if directory else None
52
+ if cache_path and os.path.isfile(cache_path):
53
+ dataset = pd.read_pickle(cache_path)
54
+ else:
55
+ if verbose:
56
+ if directory:
57
+ print(f"Dataset {cls.name()} not found in directory {directory}.")
58
+ print("Downloading...")
59
+ dataset = cls.download(verbose)
60
+ if save:
61
+ os.makedirs(directory, exist_ok=True)
62
+ file_path = os.path.join(directory, f"{cls.name()}.pkl")
63
+ if verbose:
64
+ print(f"Saved dataset to {file_path}")
65
+ dataset.to_pickle(file_path)
66
+ return dataset
67
+
68
+ @classmethod
69
+ @final
70
+ def download(cls, verbose: bool = False) -> pd.DataFrame:
71
+ """ Downloads the dataset from the internet, parses it and returns a DataFrame with cleaned data """
72
+ url = cls.url()
73
+ response = req.get(url, timeout=cls._DOWNLOAD_TIMEOUT_SEC)
74
+ code = response.status_code
75
+ if code != 200:
76
+ raise ConnectionError(
77
+ f"HTTP status code {code} when attempting to download dataset {cls.name()} from {url}"
78
+ )
79
+ df = cls._parse_response(response, verbose)
80
+ return cls._reorder_columns(df)
81
+
82
+ @classmethod
83
+ @abstractmethod
84
+ def _parse_response(cls, response: req.Response, verbose: bool = False) -> pd.DataFrame:
85
+ """ Parses the downloaded response and returns a DataFrame containing the raw dataset """
86
+ raise NotImplementedError
87
+
88
+ @classmethod
89
+ @final
90
+ def _reorder_columns(cls, df: pd.DataFrame) -> pd.DataFrame:
91
+ ordered_columns = sorted(df.columns, key=lambda col: cls.column_order().get(col, 10))
92
+ return df[ordered_columns]
93
+
94
+ @classmethod
95
+ @final
96
+ def name(cls) -> str:
97
+ """ Name of the dataset """
98
+ if not cls._NAME:
99
+ raise AttributeError(f"Class {cls.__name__} must implement class attribute `_NAME`")
100
+ return cls._NAME
101
+
102
+ @classmethod
103
+ @final
104
+ def url(cls) -> str:
105
+ """ URL of the dataset """
106
+ if not cls._URL:
107
+ raise AttributeError(f"Class {cls.__name__} must implement class attribute `_URL`")
108
+ return cls._URL
109
+
110
+ @classmethod
111
+ @final
112
+ def articles(cls) -> List[str]:
113
+ """ List of articles to cite when using this dataset """
114
+ if not cls._ARTICLES:
115
+ raise AttributeError(f"Class {cls.__name__} must implement class attribute `_{cnst.ARTICLES_STR}`")
116
+ return cls._ARTICLES
117
+
118
+ @classmethod
119
+ @final
120
+ def license(cls) -> str:
121
+ """ License of the dataset """
122
+ if not cls._LICENSE:
123
+ raise AttributeError(f"Class {cls.__name__} must implement class attribute `_LICENSE`")
124
+ return cls._LICENSE
125
+
126
+ @classmethod
127
+ @final
128
+ def documentation(cls) -> str:
129
+ """ Returns a string with information about the dataset """
130
+ title = f"Dataset:\t{cls.name().replace('_', ' ').title()}"
131
+ lcns = f"License:\t{cls.license()}"
132
+ urls = f"URL:\t{cls.url()}"
133
+ articles = "Articles:\n" + "\n".join([f"- {a}" for a in cls.articles()])
134
+ docstring = cls.__doc__ if cls.__doc__ else ""
135
+ return f"{title}\n{lcns}\n{urls}\n{articles}\n\n{docstring}"
136
+
137
+ @staticmethod
138
+ def column_order() -> Dict[str, float]:
139
+ return {
140
+ cnst.TRIAL_ID_STR: 0.1, cnst.SUBJECT_ID_STR: 0.2, cnst.STIMULUS_TYPE_STR: 0.3, cnst.STIMULUS_NAME_STR: 0.4,
141
+ cnst.T: 1.0, cnst.X: 1.1, cnst.Y: 1.2, cnst.PUPIL: 1.3,
142
+ cnst.LEFT_X: 2.1, cnst.LEFT_Y: 2.2, cnst.LEFT_PUPIL: 2.3,
143
+ cnst.RIGHT_X: 3.1, cnst.RIGHT_Y: 3.2, cnst.RIGHT_PUPIL: 3.3,
144
+ cnst.PIXEL_SIZE_STR: 4.2, cnst.VIEWER_DISTANCE_STR: 4.3
145
+ }
146
+
147
+ @staticmethod
148
+ @final
149
+ def _extract_filename_and_extension(full_path: str) -> (str, str, str):
150
+ """ Splits a full path into its components: path, filename and extension """
151
+ path, extension = os.path.splitext(full_path)
152
+ path, filename = os.path.split(path)
153
+ return path, filename, extension
154
+
155
+
156
+ class Lund2013DatasetLoader(BaseDatasetLoader):
157
+ """
158
+ Loads the dataset from article: "One algorithm to rule them all? An evaluation and discussion of ten eye movement
159
+ event-detection algorithms.", Andersson et al. (2017).
160
+
161
+ This loader is based on a previous implementation, see article:
162
+ Startsev, M., Zemblys, R. Evaluating Eye Movement Event Detection: A Review of the State of the Art. Behav Res 55, 1653–1714 (2023)
163
+ See their implementation: https://github.com/r-zemblys/EM-event-detection-evaluation/blob/main/misc/data_parsers/lund2013.py
164
+
165
+ ** Important Notes: **
166
+ (1) The dataset used in Andersson et al. (2017) is a subset of the complete dataset, available through this loader.
167
+ The subset of the dataset used in the article is available at: 'EyeMovementDetectorEvaluation-master/annotated_data/data used in the article/'
168
+ (2) Two files had to be replaced due to errors in the original dataset. The corrected files are included in the
169
+ dataset and used instead of the erroneous ones.
170
+ """
171
+
172
+ _NAME = "Lund2013"
173
+ _URL = 'https://github.com/richardandersson/EyeMovementDetectorEvaluation/archive/refs/heads/master.zip'
174
+ _LICENSE = "GNU GPL-3.0"
175
+ _ARTICLES = [
176
+ "Andersson, R., Larsson, L., Holmqvist, K., Stridh, M., & Nyström, M. (2017): One algorithm to rule them " +
177
+ "all? An evaluation and discussion of ten eye movement event-detection algorithms. Behavior Research " +
178
+ "Methods, 49(2), 616-637.",
179
+ ]
180
+
181
+ __PREFIX = 'EyeMovementDetectorEvaluation-master/annotated_data/originally uploaded data/'
182
+ # note: the article contained only a subset of the full dataset. article data is in directory 'annotated_data/data used in the article/'
183
+ __ERRONEOUS_FILES = ['UH29_img_Europe_labelled_MN.mat']
184
+ __CORRECTION_FILES = [
185
+ 'EyeMovementDetectorEvaluation-master/annotated_data/fix_by_Zemblys2018/UH29_img_Europe_labelled_FIX_MN.mat'
186
+ ]
187
+
188
+ @classmethod
189
+ def _parse_response(cls, response: req.Response, verbose: bool = False) -> pd.DataFrame:
190
+
191
+ def is_valid_filename(name: str) -> bool:
192
+ # check if the file is a valid mat file containing gaze data
193
+ if not name.startswith(cls.__PREFIX):
194
+ return False
195
+ if not name.endswith('.mat'):
196
+ return False
197
+ if any(name.endswith(errs) for errs in cls.__ERRONEOUS_FILES):
198
+ # skip erroneous files, see readme.md for more info
199
+ return False
200
+ return True
201
+
202
+ # extract the zip file and list all files that contain gaze data
203
+ zip_file = zp.ZipFile(io.BytesIO(response.content))
204
+ file_names = [f for f in zip_file.namelist() if is_valid_filename(f)]
205
+ file_names.extend(cls.__CORRECTION_FILES)
206
+
207
+ # read all files into a list of dataframes
208
+ trial_id = 0
209
+ dataframes = {}
210
+ for f in tqdm(file_names, desc="Processing Files", disable=not verbose):
211
+ file = zip_file.open(f)
212
+ gaze_data = cls.__read_eyetracker_data(file)
213
+ subject_id, stimulus_type, stimulus_name, rater = cls.__extract_metadata(file)
214
+ gaze_data.rename(columns={cnst.LABEL_STR: rater}, inplace=True)
215
+
216
+ # write the DF to a dict based on the subject id, stimulus type, stimulus name, or add to existing DF
217
+ existing_df = dataframes.get((subject_id, stimulus_type, stimulus_name), None)
218
+ if existing_df is None:
219
+ trial_id += 1
220
+ gaze_data[cnst.TRIAL_ID_STR] = trial_id
221
+ gaze_data[cnst.SUBJECT_ID_STR] = subject_id
222
+ gaze_data[cnst.STIMULUS_TYPE_STR] = stimulus_type
223
+ gaze_data[cnst.STIMULUS_NAME_STR] = stimulus_name
224
+ dataframes[(subject_id, stimulus_type, stimulus_name)] = gaze_data
225
+ else:
226
+ if len(existing_df) != len(gaze_data):
227
+ # index-alignment below would otherwise silently fill NaN for the length difference
228
+ # (S-5), which can mask a truncated/corrupted rater file. Warn rather than raise: this
229
+ # loader feeds the article's Lund2013 data, and unlike a crash, the *old* behavior here
230
+ # was already silent, so a mismatch (if any exist in the real files) can't be ruled out
231
+ # without risking breaking a currently-working load.
232
+ warnings.warn(
233
+ f"Rater '{rater}' has {len(gaze_data)} samples for trial "
234
+ f"{(subject_id, stimulus_type, stimulus_name)}, expected {len(existing_df)} - "
235
+ f"the mismatched samples will be NaN-filled by index alignment.",
236
+ stacklevel=2,
237
+ )
238
+ existing_df.loc[:, rater] = gaze_data.loc[:, rater]
239
+ return pd.concat(dataframes.values(), ignore_index=True, axis=0)
240
+
241
+ @staticmethod
242
+ def __read_eyetracker_data(file) -> pd.DataFrame:
243
+ mat = loadmat(file)
244
+ eyetracking_data = mat["ETdata"]
245
+ eyetracking_data_dict = {name: eyetracking_data[name][0, 0] for name in eyetracking_data.dtype.names}
246
+
247
+ # extract singleton values and convert from meters to cm:
248
+ sampling_rate = eyetracking_data_dict['sampFreq'][0, 0]
249
+ view_dist = eyetracking_data_dict['viewDist'][0, 0] * 100
250
+ screen_width, screen_height = eyetracking_data_dict['screenDim'][0] * 100
251
+ screen_res = eyetracking_data_dict['screenRes'][0] # (1024, 768)
252
+ pixel_size = calculate_pixel_size(screen_width, screen_height, screen_res)
253
+
254
+ # extract gaze data:
255
+ samples_data = eyetracking_data_dict['pos']
256
+ right_x, right_y = samples_data[:, 3:5].T # only recording right eye
257
+ is_missing = (right_x == 0) & (right_y == 0) # missing samples are marked with (0, 0) coordinates
258
+ right_x[is_missing] = np.nan
259
+ right_y[is_missing] = np.nan
260
+ labels = pd.Series(samples_data[:, 5]).apply(lambda x: parse_label(x, safe=True))
261
+ if np.isnan(samples_data[:, 0]).any():
262
+ # if timestamps are NaN, re-populate them
263
+ timestamps = np.arange(len(right_x)) * cnst.MILLISECONDS_PER_SECOND / sampling_rate
264
+ else:
265
+ # timestamps are available but in microseconds
266
+ timestamps = samples_data[:, 0] - np.nanmin(samples_data[:, 0]) # start timestamps from 0
267
+ timestamps /= cnst.MICROSECONDS_PER_MILLISECOND
268
+ return pd.DataFrame(data={
269
+ cnst.T: timestamps, cnst.X: right_x, cnst.Y: right_y, cnst.PUPIL: np.nan,
270
+ cnst.LABEL_STR: labels, cnst.VIEWER_DISTANCE_STR: view_dist, cnst.PIXEL_SIZE_STR: pixel_size
271
+ })
272
+
273
+ @staticmethod
274
+ def __extract_metadata(file) -> Tuple[str, str, Union[str, int], str]:
275
+ file_name = os.path.basename(file.name) # remove path
276
+ if not file_name.endswith(".mat"):
277
+ raise ValueError(f"Expected a `.mat` file, got: {file_name}")
278
+ # file_name fmt: `<subject_id>_<stimulus_type>_<stimulus_name_1>_ ... _<stimulus_name_N>_labelled_<rater_name>`
279
+ # moving-dot trials don't contain stimulus names
280
+ file_name = file_name.replace(".mat", "") # remove extension
281
+ split_name = file_name.split("_")
282
+ subject_id = split_name[0] # subject id is always 1st in the file name
283
+ rater = split_name[-1].upper() # rater is always last in the file name
284
+ stimulus_type = split_name[1] # stimulus type is always 2nd in the file name
285
+ if stimulus_type == cnst.VIDEO_STR:
286
+ stimulus_name = "_".join(split_name[2:-2]).removesuffix("_labelled") # between stim type and rater
287
+ return subject_id, stimulus_type, stimulus_name, rater
288
+ if stimulus_type == "img":
289
+ stimulus_name = "_".join(split_name[2:-2]).removesuffix("_labelled") # between stim type and rater
290
+ stimulus_type = cnst.IMAGE_STR # rename stimulus type to match peyes constants
291
+ return subject_id, stimulus_type, stimulus_name, rater
292
+ if stimulus_type.startswith(cnst.TRIAL_STR):
293
+ # moving-dot stimulus is labelled as "trial1" or "trial17"
294
+ stimulus_name = int(stimulus_type.removeprefix(cnst.TRIAL_STR))
295
+ stimulus_type = cnst.MOVING_DOT_STR
296
+ return subject_id, stimulus_type, stimulus_name, rater
297
+ raise ValueError(f"Unknown stimulus type: {stimulus_type}")
298
+
299
+
300
+ class IRFDatasetLoader(BaseDatasetLoader):
301
+ """
302
+ Loads the dataset from a replication study of the article:
303
+ Using machine learning to detect events in eye-tracking data. Zemblys et al. (2018).
304
+ See also about the repro study: https://github.com/r-zemblys/irf/blob/master/doc/IRF_replication_report.pdf
305
+
306
+ Note: binocular data was recorded but only one pair of (x, y) coordinates is provided.
307
+
308
+ This loader is based on a previous implementation, see article:
309
+ Startsev, M., Zemblys, R. Evaluating Eye Movement Event Detection: A Review of the State of the Art. Behav Res 55, 1653–1714 (2023)
310
+ See their implementation: https://github.com/r-zemblys/EM-event-detection-evaluation/blob/main/misc/data_parsers/humanFixationClassification.py
311
+ """
312
+
313
+ _NAME = "IRF"
314
+ _URL = r'https://github.com/r-zemblys/irf/archive/refs/heads/master.zip'
315
+ _LICENSE = "MIT"
316
+ _ARTICLES = [
317
+ "Zemblys, Raimondas and Niehorster, Diederick C and Komogortsev, Oleg and Holmqvist, Kenneth. Using machine " +
318
+ "learning to detect events in eye-tracking data. Behavior Research Methods, 50(1), 160–181 (2018)."
319
+ ]
320
+
321
+ __PREFIX = 'irf-master/etdata/lookAtPoint_EL'
322
+ __STIMULUS_TYPE_VAL = "moving_dot" # all subjects were shown the same 13-point moving dot stimulus
323
+ __VIEWER_DISTANCE_CM_VAL = 56.5
324
+ __MONITOR_WIDTH_CM_VAL, __MONITOR_HEIGHT_CM_VAL = 37.5, 30.2
325
+ __MONITOR_RESOLUTION_VAL = (1280, 1024)
326
+ __PIXEL_SIZE_CM_VAL = calculate_pixel_size(
327
+ __MONITOR_WIDTH_CM_VAL, __MONITOR_HEIGHT_CM_VAL, __MONITOR_RESOLUTION_VAL
328
+ )
329
+ __RATER_NAME = "RZ"
330
+
331
+ @classmethod
332
+ def _parse_response(cls, response: req.Response, verbose: bool = False) -> pd.DataFrame:
333
+ zip_file = zp.ZipFile(io.BytesIO(response.content))
334
+ gaze_file_names = [
335
+ f for f in zip_file.namelist() if
336
+ (f.startswith(psx.join(cls.__PREFIX, "lookAtPoint_EL_")) and f.endswith('.npy'))
337
+ ]
338
+ gaze_dfs = []
339
+ for i, f in enumerate(tqdm(gaze_file_names, desc="Processing Files", disable=not verbose)):
340
+ file = zip_file.open(f)
341
+ _, file_name, _ = cls._extract_filename_and_extension(f)
342
+ gaze_data = pd.DataFrame(np.load(file))
343
+ gaze_data['evt'] = gaze_data['evt'].apply(lambda x: parse_label(x, safe=True)) # convert labels
344
+ gaze_data[cnst.SUBJECT_ID_STR] = file_name.split('_')[-1] # format: "lookAtPoint_EL_S<subject_num>"
345
+ gaze_data[cnst.TRIAL_ID_STR] = i + 1
346
+ gaze_data[cnst.PUPIL] = np.nan
347
+ gaze_dfs.append(gaze_data)
348
+ df = pd.concat(gaze_dfs, ignore_index=True, axis=0)
349
+
350
+ # add metadata columns:
351
+ df[cnst.STIMULUS_TYPE_STR] = cls.__STIMULUS_TYPE_VAL
352
+ df[cnst.VIEWER_DISTANCE_STR] = cls.__VIEWER_DISTANCE_CM_VAL
353
+ df[cnst.PIXEL_SIZE_STR] = cls.__PIXEL_SIZE_CM_VAL
354
+
355
+ # remap columns to correct values
356
+ df.rename(
357
+ columns={"t": cnst.T, "evt": cls.__RATER_NAME, "x": cnst.X, "y": cnst.Y}, inplace=True
358
+ )
359
+ df[cnst.T] = df[cnst.T] * cnst.MILLISECONDS_PER_SECOND # convert seconds to milliseconds
360
+ df = cls.__correct_coordinates(df)
361
+ return df
362
+
363
+ @classmethod
364
+ def __correct_coordinates(cls, df: pd.DataFrame) -> pd.DataFrame:
365
+ new_df = df.copy()
366
+ nan_idxs = new_df[~new_df[cnst.STATUS_STR]].index
367
+ new_df.loc[nan_idxs, cnst.X] = np.nan
368
+ new_df.loc[nan_idxs, cnst.Y] = np.nan
369
+ new_df.drop(columns=[cnst.STATUS_STR], inplace=True)
370
+
371
+ pixel_width_cm = cls.__MONITOR_WIDTH_CM_VAL / cls.__MONITOR_RESOLUTION_VAL[0]
372
+ x = new_df[cnst.X].apply(
373
+ lambda ang: visual_angle_to_pixels(
374
+ angle=ang, d=cls.__VIEWER_DISTANCE_CM_VAL, pixel_size=pixel_width_cm, use_radians=False, keep_sign=True
375
+ )
376
+ )
377
+ x += cls.__MONITOR_RESOLUTION_VAL[0] // 2 # move x=0 coordinate to the left of the screen
378
+ pixel_height = cls.__MONITOR_HEIGHT_CM_VAL / cls.__MONITOR_RESOLUTION_VAL[1]
379
+ y = new_df[cnst.Y].apply(
380
+ lambda ang: visual_angle_to_pixels(
381
+ angle=ang, d=cls.__VIEWER_DISTANCE_CM_VAL, pixel_size=pixel_height, use_radians=False, keep_sign=True
382
+ )
383
+ )
384
+ y += cls.__MONITOR_RESOLUTION_VAL[1] // 2 # move y=0 coordinate to the top of the screen
385
+ new_df.loc[:, cnst.X] = x.astype("float32")
386
+ new_df.loc[:, cnst.Y] = y.astype("float32")
387
+ return new_df
388
+
389
+
390
+ class HFCDatasetLoader(BaseDatasetLoader):
391
+ """
392
+ Loads the two datasets presented in articles:
393
+ - (adults) Is human classification by experienced untrained observers a gold standard in fixation detection?
394
+ Hooge et al. (2018)
395
+ - (infants) An in-depth look at saccadic search in infancy.
396
+ Hessels et al. (2016)
397
+
398
+ This loader is based on a previous implementation, see article:
399
+ Startsev, M., Zemblys, R. Evaluating Eye Movement Event Detection: A Review of the State of the Art. Behav Res 55, 1653–1714 (2023)
400
+ See their implementation: https://github.com/r-zemblys/EM-event-detection-evaluation/blob/main/misc/data_parsers/humanFixationClassification.py
401
+
402
+ Note the original publication did not specify the viewer distance used when recording the data. Here, we use the
403
+ value from the Tobii TX300 eye-tracker's specifications: 65.0 cm (https://www.spectratech.gr/Web/Tobii/pdf/TX300.pdf).
404
+ They used their eye-tracker's monitor with a resolution of 1920x1080 pixels, an aspect ratio of 16:9 and a diagonal
405
+ of 23". Based on this site (https://max.pm/posts/screensize/) we calculate the height and width of the monitor to be
406
+ 28.64cm and 50.92cm, respectively. The pixel size is calculated based on these values.
407
+ """
408
+
409
+ _NAME: str = "HFC"
410
+ _URL = r'https://github.com/dcnieho/humanFixationClassification/archive/refs/heads/master.zip'
411
+ _LICENSE = "CC NC-BY-SA 4.0"
412
+ _ARTICLES = [
413
+ "Hooge, I.T.C., Niehorster, D.C., Nyström, M., Andersson, R. & Hessels, R.S. (2018). Is human classification " +
414
+ "by experienced untrained observers a gold standard in fixation detection?",
415
+ ]
416
+
417
+ __PREFIX = 'humanFixationClassification-master/data'
418
+ __SUBJECT_GROUP_STR = "subject_group"
419
+ __INFANT_STR, __ADULT_STR = "infant", "adult"
420
+ __SEARCH_TASK_STR, __FREE_VIEWING_STR = "search_task", "free_viewing"
421
+
422
+ __VIEWER_DISTANCE_CM_VAL = 65
423
+ __MONITOR_WIDTH_CM_VAL, __MONITOR_HEIGHT_CM_VAL = 50.92, 28.64
424
+ __MONITOR_RESOLUTION_VAL = (1920, 1080)
425
+ __PIXEL_SIZE_CM_VAL = calculate_pixel_size(
426
+ __MONITOR_WIDTH_CM_VAL, __MONITOR_HEIGHT_CM_VAL, __MONITOR_RESOLUTION_VAL
427
+ )
428
+
429
+ @staticmethod
430
+ def column_order() -> Dict[str, float]:
431
+ return {
432
+ **super(HFCDatasetLoader, HFCDatasetLoader).column_order(), HFCDatasetLoader.__SUBJECT_GROUP_STR: 6.1
433
+ }
434
+
435
+ @classmethod
436
+ def _parse_response(cls, response: req.Response, verbose: bool = False) -> pd.DataFrame:
437
+ zip_file = zp.ZipFile(io.BytesIO(response.content))
438
+ # extract gaze data:
439
+ gaze_file_names = [
440
+ f for f in zip_file.namelist() if
441
+ (f.startswith(psx.join(cls.__PREFIX, "ETdata")) and f.endswith('.txt'))
442
+ ]
443
+ gaze_dfs = {}
444
+ for i, f in enumerate(tqdm(gaze_file_names, desc="Processing Files", disable=not verbose)):
445
+ file = zip_file.open(f)
446
+ gaze_data = pd.read_csv(file, sep='\t')
447
+ gaze_data[cnst.PUPIL] = np.nan # no pupil data available
448
+ _, file_name, _ = cls._extract_filename_and_extension(file.name)
449
+ subject_group, subject_id = file_name.split('_') # format: "<subject_type>_<subject_id>"
450
+ gaze_data[cnst.SUBJECT_ID_STR] = subject_id
451
+ gaze_data[cls.__SUBJECT_GROUP_STR] = subject_group
452
+ gaze_data[cnst.STIMULUS_TYPE_STR] = cls.__FREE_VIEWING_STR if subject_group == cls.__ADULT_STR else cls.__SEARCH_TASK_STR
453
+ gaze_data[cnst.TRIAL_ID_STR] = i + 1
454
+ gaze_dfs[file_name] = gaze_data
455
+
456
+ # extract annotations:
457
+ coder_file_names = [
458
+ f for f in zip_file.namelist() if
459
+ (f.startswith(psx.join(cls.__PREFIX, "coderSettings")) and f.endswith('.txt'))
460
+ ]
461
+ annotation_dfs = {}
462
+ for f in coder_file_names:
463
+ with zip_file.open(f) as open_file:
464
+ rater_data = pd.read_csv(open_file, sep='\t')
465
+ _, rater_name, _ = cls._extract_filename_and_extension(open_file.name)
466
+ annotation_dfs[rater_name.upper()] = rater_data
467
+
468
+ # merge annotations with gaze data:
469
+ merged_dfs = []
470
+ for key, data in gaze_dfs.items(): # noqa: B007 # false positive: `key` is read by the `@key` in .query() below
471
+ if data is None or len(data) == 0 or data.empty:
472
+ continue
473
+ l = len(data)
474
+ for rater_name in annotation_dfs.keys():
475
+ annotations = annotation_dfs.get(rater_name).query("Trial==@key")
476
+ if annotations is None or len(annotations) == 0 or annotations.empty:
477
+ labels = np.zeros(l, dtype=int)
478
+ else:
479
+ # reached here if there are annotations from this rater for this trial
480
+ labels = np.zeros(l, dtype=int)
481
+ # S-1: hoisted out of a per-row loop that recomputed the exact same thing len(annotations)
482
+ # times (the loop never used its own row); this now runs once.
483
+ f = interp1d(
484
+ data["time"], range(l), kind="nearest", bounds_error=False, fill_value="extrapolate"
485
+ )
486
+ # `f` extrapolates (bounds_error=False), so an annotation outside the trial's
487
+ # time range yields an index beyond [0, l). A negative index would silently
488
+ # wrap to the far end of the trial, so clip rather than index blindly.
489
+ fixation_samples = itertools.chain(
490
+ *[range(max(0, int(s)), min(l, int(e) + 1)) for s, e in zip(
491
+ f(annotations["FixStart"]), f(annotations["FixEnd"])
492
+ )]
493
+ )
494
+ labels[list(fixation_samples)] = 1
495
+ data[rater_name] = labels
496
+ data[rater_name] = data[rater_name].apply(lambda x: parse_label(x, safe=True))
497
+ merged_dfs.append(data)
498
+
499
+ # concatenate all dataframes into a single one and add metadata columns
500
+ full_dataset = pd.concat(merged_dfs, ignore_index=True, axis=0)
501
+ full_dataset.rename(columns={"time": cnst.T, "x": cnst.X, "y": cnst.Y}, inplace=True)
502
+ full_dataset[cnst.VIEWER_DISTANCE_STR] = cls.__VIEWER_DISTANCE_CM_VAL
503
+ full_dataset[cnst.PIXEL_SIZE_STR] = cls.__PIXEL_SIZE_CM_VAL
504
+ return full_dataset
505
+
506
+
507
+ class GazeComDatasetLoader(BaseDatasetLoader):
508
+ """
509
+ Loads a labelled subset of the dataset presented in the article:
510
+ Michael Dorr, Thomas Martinetz, Karl Gegenfurtner, and Erhardt Barth. Variability of eye movements when viewing
511
+ dynamic natural scenes. Journal of Vision, 10(10):1-17, 2010.
512
+
513
+ Labels are from the article:
514
+ Agtzidis, I., Startsev, M., & Dorr, M. (2016a). In the pursuit of (ground) truth: A hand-labelling tool for eye
515
+ movements recorded during dynamic scene viewing. In 2016 IEEE second workshop on eye tracking and visualization
516
+ (ETVIS) (pp. 65–68).
517
+
518
+ Note 1: This dataset is extremely large and may take a long time to download. It is recommended to save it to local
519
+ storage after downloading it for faster access in the future. The method `load_zipfile` can be used to load the
520
+ dataset from a raw zip file stored in a local directory.
521
+ Note 2: This is only a subset of the full GazeCom Dataset, containing hand-labelled samples. The full dataset with
522
+ documentation can be found in https://www.inb.uni-luebeck.de/index.php?id=515.
523
+ Note 3: binocular data was recorded but only one pair of (x, y) coordinates is provided.
524
+
525
+ This loader is based on a previous implementation, see article:
526
+ Startsev, M., Zemblys, R. Evaluating Eye Movement Event Detection: A Review of the State of the Art. Behav Res 55, 1653–1714 (2023)
527
+ See their implementation: https://github.com/r-zemblys/EM-event-detection-evaluation/blob/main/misc/data_parsers/tum.py
528
+ """
529
+
530
+ _NAME: str = "GazeCom"
531
+ _URL = r'https://gin.g-node.org/ioannis.agtzidis/gazecom_annotations/archive/master.zip'
532
+ _LICENSE = "GNU GPL-3.0"
533
+ _ARTICLES = [
534
+ "Agtzidis, I., Startsev, M., & Dorr, M. (2016a). In the pursuit of (ground) truth: A hand-labelling tool for " +
535
+ "eye movements recorded during dynamic scene viewing. In 2016 IEEE second workshop on eye tracking and" +
536
+ "visualization (ETVIS) (pp. 65–68).",
537
+
538
+ "Michael Dorr, Thomas Martinetz, Karl Gegenfurtner, and Erhardt Barth. Variability of eye movements when " +
539
+ "viewing dynamic natural scenes. Journal of Vision, 10(10):1-17, 2010.",
540
+ "Startsev, M., Agtzidis, I., & Dorr, M. (2016). Smooth pursuit. http://michaeldorr.de/smoothpursuit/",
541
+ ]
542
+
543
+ __PREFIX = psx.join('gazecom_annotations', 'ground_truth')
544
+ __ZIPFILE_NAME = "gazecom_annotations-master.zip"
545
+ __HANDLABELLER_PREFIX = "HL"
546
+ __VIEWER_DISTANCE_CM_VAL = 56.5
547
+ __MONITOR_WIDTH_CM_VAL, __MONITOR_HEIGHT_CM_VAL = 40, 22.5
548
+ __MONITOR_RESOLUTION_VAL = (1280, 720)
549
+ __PIXEL_SIZE_CM_VAL = calculate_pixel_size(
550
+ __MONITOR_WIDTH_CM_VAL, __MONITOR_HEIGHT_CM_VAL, __MONITOR_RESOLUTION_VAL
551
+ )
552
+
553
+ __LABEL_MAP = {
554
+ 0: EventLabelEnum.UNDEFINED,
555
+ 1: EventLabelEnum.FIXATION,
556
+ 2: EventLabelEnum.SACCADE,
557
+ 3: EventLabelEnum.SMOOTH_PURSUIT,
558
+ 4: EventLabelEnum.UNDEFINED # noise
559
+ }
560
+ __COLUMN_MAP = {
561
+ "time": cnst.T, "x": cnst.X, "y": cnst.Y, "handlabeller1": f"{__HANDLABELLER_PREFIX}1",
562
+ "handlabeller2": f"{__HANDLABELLER_PREFIX}2", "handlabeller_final": f"{__HANDLABELLER_PREFIX}_FINAL"
563
+ }
564
+
565
+ @classmethod
566
+ def load_zipfile(cls, root: str = None, verbose: bool = False) -> pd.DataFrame:
567
+ """
568
+ Loads the dataset from a zip file stored in a local directory (so it doesn't need to be downloaded again).
569
+ :param root: path to the directory containing the zip file.
570
+ :param verbose: whether to display progress bars.
571
+ :return: DataFrame with the annotated gaze data.
572
+ """
573
+ if not root or not psx.isdir(root):
574
+ raise NotADirectoryError(f"Invalid directory: {root}")
575
+ zip_file = psx.join(root, cls.__ZIPFILE_NAME)
576
+ if not psx.isfile(zip_file):
577
+ raise FileNotFoundError(f"File not found: {zip_file}")
578
+ with zp.ZipFile(zip_file, 'r') as zip_ref:
579
+ df = cls.__read_zipfile(zf=zip_ref, verbose=verbose)
580
+ return cls._reorder_columns(df)
581
+
582
+ @staticmethod
583
+ def column_order() -> Dict[str, float]:
584
+ handlabeller_scores = {
585
+ f"{GazeComDatasetLoader.__HANDLABELLER_PREFIX}1": 5.1,
586
+ f"{GazeComDatasetLoader.__HANDLABELLER_PREFIX}2": 5.2,
587
+ f"{GazeComDatasetLoader.__HANDLABELLER_PREFIX}_FINAL": 5.3
588
+ }
589
+ return {**super(GazeComDatasetLoader, GazeComDatasetLoader).column_order(), **handlabeller_scores}
590
+
591
+ @classmethod
592
+ def _parse_response(cls, response: req.Response, verbose: bool = False) -> pd.DataFrame:
593
+ zip_file = zp.ZipFile(io.BytesIO(response.content))
594
+ return cls.__read_zipfile(zf=zip_file, verbose=verbose)
595
+
596
+ @classmethod
597
+ def __read_zipfile(cls, zf: zp.ZipFile, verbose: bool = False) -> pd.DataFrame:
598
+ """
599
+ Reads the contents of a zip file and returns a DataFrame with the annotated data.
600
+ :param zf: ZipFile object
601
+ :param verbose: whether to display progress bars
602
+ :return: DataFrame with annotated gaze data
603
+ """
604
+ annotated_file_names = [f for f in zf.namelist() if (f.endswith('.arff') and cls.__PREFIX in f)]
605
+ gaze_dfs = []
606
+ for i, f in enumerate(tqdm(annotated_file_names, desc="Processing Files", disable=not verbose)):
607
+ file = zf.open(f)
608
+ data = arff.loads(file.read().decode('utf-8'))
609
+
610
+ # parse gaze data:
611
+ df = pd.DataFrame(data['data'], columns=[attr[0] for attr in data['attributes']])
612
+ df[cnst.PUPIL] = np.nan # no pupil data available
613
+ invalid_idxs = np.where(np.all(df[["x", "y"]] == 0, axis=1) | (df["confidence"] < 0.5))[0]
614
+ df.iloc[invalid_idxs, df.columns.get_indexer(["x", "y"])] = np.nan
615
+ df['time'] = df['time'] / cnst.MILLISECONDS_PER_SECOND
616
+ df['time'] = df['time'] - df['time'].min() # start timestamps from 0
617
+ df.drop(columns=['confidence'], inplace=True)
618
+ df.rename(columns=cls.__COLUMN_MAP, inplace=True)
619
+ for col in df.columns:
620
+ if col.startswith(cls.__HANDLABELLER_PREFIX):
621
+ df[col] = df[col].map(cls.__LABEL_MAP)
622
+
623
+ # add metadata columns:
624
+ _, file_name, _ = cls._extract_filename_and_extension(f)
625
+ subj_id = file_name.split('_')[0] # file_name: <subject_id>_<stimulus>_<name>_<with>_<underscores>.arff
626
+ stimulus = '_'.join(file_name.split('_')[1:])
627
+ df[cnst.SUBJECT_ID_STR] = subj_id
628
+ df[cnst.STIMULUS_NAME_STR] = stimulus
629
+ df[cnst.TRIAL_ID_STR] = i + 1
630
+ gaze_dfs.append(df)
631
+
632
+ # merge and add common metadata columns:
633
+ full_dataset = pd.concat(gaze_dfs, ignore_index=True, axis=0)
634
+ full_dataset[cnst.STIMULUS_TYPE_STR] = cnst.VIDEO_STR
635
+ full_dataset[cnst.VIEWER_DISTANCE_STR] = cls.__VIEWER_DISTANCE_CM_VAL
636
+ full_dataset[cnst.PIXEL_SIZE_STR] = cls.__PIXEL_SIZE_CM_VAL
637
+ return full_dataset