Coverage for scanpath_studio/eyegenbench.py: 90%

87 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""Loader for a local bundle of harmonised reading corpora. 

2 

3EyeGenBench (https://github.com/EyeBench/EyeGenBench) harmonises many public 

4eye-tracking-while-reading corpora into one schema, but discards screen 

5geometry. `scripts/prepare_eyegenbench.py` runs their pipeline, recovers the 

6geometry, and writes the bundle this module reads. See 

7`docs/benchmark-corpora.md` and `plans/data-27-eyegenbench-datasets.md`. 

8 

9The pipeline can *load* 39 corpora; a bundle holds however many the user 

10prepared, which is fewer -- some publishers require manual acquisition. Read 

11the manifest for what is actually there rather than assuming a count. 

12 

13Bundle contract for `fixations.parquet`: it must NOT carry a column literally 

14named `unique_trial_id`. `data.normalize_fixations` used to key `trial_id` on 

15that exact column name whenever it was present, overriding 

16`EYEGENBENCH_FIX_SCHEMA`'s `trial` mapping below and breaking the words 

17broadcast (paragraph-keyed stimulus-level words vs. reading-keyed fixations 

18never matched, silently broadcasting zero word boxes). BUG-58 made the mapping 

19authoritative, but the name stays reserved: the normalized frame's own 

20`unique_trial_id` is the mapped trial id, so the raw values would not survive 

21under it. If the prep script carries EyeGenBench's own finer-grained 

22(per-reading) trial identity through at all, it must use the column name 

23`eyegenbench_trial_id` instead -- registered as an opaque passthrough in Task 7. It must also give repeated 

24readings of the same paragraph by the same participant distinct 

25`unique_paragraph_id` values: this loader keys `trial_id` on that column 

26directly and does not disambiguate repeats itself. 

27""" 

28 

29from __future__ import annotations 

30 

31import json 

32from pathlib import Path 

33 

34import pandas as pd 

35 

36from . import progress 

37 

38MANIFEST_NAME = "manifest.json" 

39_TABLES = ("words", "fixations", "participants") 

40 

41# Stimulus-level: `participant=None` marks one shared layout rather than one row 

42# per reader, so `data.broadcast_stimulus_words` expands it (as for PoTeC). 

43EYEGENBENCH_WORD_SCHEMA = dict( 

44 participant=None, 

45 trial="unique_paragraph_id", 

46 word_id="ia_index", 

47 text="ia_label", 

48 line="line", 

49 left="start_x", 

50 right="end_x", 

51 top="start_y", 

52 bottom="end_y", 

53 # Declared absent, not merely omitted. A prepared corpus is single-screen by 

54 # contract -- the prep script writes one coordinate space per paragraph -- 

55 # but the frames carry the publisher's leftover columns through, and some 

56 # corpora (Provo, SBSAT) keep a `page`. Auto-detection finds it on the 

57 # fixations and not here, and `multipart.validate_matching_parts` then 

58 # rejects the pair outright ("Multipart identity is present in only one 

59 # report"). Saying `None` on BOTH schemas is what stops a leftover column 

60 # being read as a screen identity that this bundle does not have. 

61 screen_id=None, 

62) 

63 

64# `trial` is `unique_paragraph_id`, matching the word schema above -- not 

65# EyeGenBench's own (finer-grained, per-reading) `unique_trial_id`. See the 

66# module docstring: a raw `unique_trial_id` column used to silently override 

67# this mapping and break the stimulus-level words broadcast. Keying on 

68# unique_paragraph_id is what makes that broadcast join work; repeated 

69# readings of the same paragraph by the same participant are NOT separated 

70# here (`data._disambiguate_repeated_readings` no-ops without a raw 

71# TRIAL_INDEX column, which EyeGenBench frames don't carry) -- the prep 

72# script is responsible for giving each reading its own paragraph key 

73# upstream (Task 8, R17). 

74EYEGENBENCH_FIX_SCHEMA = dict( 

75 participant="unique_participant_id", 

76 trial="unique_paragraph_id", 

77 duration="fix_duration", 

78 x="x", 

79 y="y", 

80 fixation_id="fix_index", 

81 word_id="ia_index", 

82 screen_id=None, # see EYEGENBENCH_WORD_SCHEMA 

83) 

84 

85 

86def _manifest_path(root) -> Path: 

87 return Path(root) / MANIFEST_NAME 

88 

89 

90def eyegenbench_manifest(root) -> dict: 

91 """The bundle manifest, or a `FileNotFoundError` naming the fix.""" 

92 path = _manifest_path(root) 

93 if not path.is_file(): 

94 raise FileNotFoundError( 

95 f"No EyeGenBench bundle at {root!s} (missing {MANIFEST_NAME}). " 

96 "Build one with: python scripts/prepare_eyegenbench.py --all" 

97 ) 

98 return json.loads(path.read_text(encoding="utf-8")) 

99 

100 

101def eyegenbench_datasets(root) -> list: 

102 """Manifest entries, one per prepared dataset. Cheap -- no Parquet is read.""" 

103 return list(eyegenbench_manifest(root).get("datasets", [])) 

104 

105 

106def entry_name(entry) -> str: 

107 """A manifest row's dataset name, or ``""`` when it hasn't got one. 

108 

109 A row is data from a file on disk, so it can be malformed. Reading the name 

110 as ``entry["name"]`` raised `KeyError` — outside the `(FileNotFoundError, 

111 ValueError, OSError)` triple every caller guards with, so **one** nameless 

112 row ordered before a valid one took the whole app down through whichever 

113 surface looked at the manifest next (M7 / I2). Nameless rows are unusable 

114 by definition — nothing can address them — so every reader skips them here 

115 rather than each remembering to catch a third exception type. 

116 """ 

117 if not isinstance(entry, dict): 

118 return "" 

119 return str(entry.get("name") or "").strip() 

120 

121 

122def entry_count(entry, key: str) -> int | None: 

123 """A manifest row's integer count field: the number, ``0``, or ``None``. 

124 

125 The same rule as `entry_name`, for the count fields (`n_texts`, 

126 `n_readers`, `n_fixations`, `paragraphs_without_real_boxes`): a manifest is 

127 data from a file on disk, and a bare ``int(entry.get(key))`` raises 

128 `ValueError`/`TypeError` on ``"many"``, ``[1]`` or any other shape a hand 

129 edit can produce — outside the catch every caller guards with, and now on 

130 the path that builds a picker entry for every added corpus (N1). 

131 

132 ``0`` when the field is absent or blank — *not recorded* is a known 

133 quantity for a count, and every reader treats it as none. ``None`` when the 

134 value is there but isn't a number, so a caller can tell "no missing texts" 

135 from "the missing-text count is unreadable" and word its claim accordingly 

136 rather than asserting the confident one. 

137 

138 **Never write ``entry_count(...) or 0``.** It collapses `None` into `0` and 

139 so reads an unreadable count as *nothing missing* — which is precisely how 

140 a corpus with unknown coverage gets badged a confident "Real", the 

141 overclaim R34 exists to prevent. Test the two cases apart (``if count := 

142 entry_count(...)`` is fine — it drops both, which is right when the value is 

143 only being formatted), or handle `None` explicitly. 

144 """ 

145 if not isinstance(entry, dict): 

146 return None 

147 raw = entry.get(key) 

148 if raw is None or (isinstance(raw, str) and not raw.strip()): 

149 return 0 

150 try: 

151 return int(raw) 

152 except (TypeError, ValueError): 

153 return None 

154 

155 

156def _find_entry(root, dataset: str) -> dict | None: 

157 """The manifest entry named ``dataset`` (case-insensitive), or ``None``. 

158 

159 Shared by every function that resolves a dataset name against the 

160 manifest, so the case-insensitive comparison lives in exactly one place. 

161 """ 

162 for entry in eyegenbench_datasets(root): 

163 if (name := entry_name(entry)) and name.lower() == str(dataset).lower(): 

164 return entry 

165 return None 

166 

167 

168def eyegenbench_present(root, dataset: str | None = None) -> bool: 

169 """True when the bundle holds everything a load needs. Path stats only. 

170 

171 Strict on purpose: a lenient check passes a partial tree and then crashes 

172 mid-load, whereas a strict one lets the app offer the fix. That includes 

173 resolving ``dataset`` against the manifest first -- a directory holding 

174 all three Parquet files under a name absent from the manifest must not 

175 read as present, since loading that same name would still raise. 

176 """ 

177 root = Path(root) 

178 if not _manifest_path(root).is_file(): 

179 return False 

180 try: 

181 if dataset is None: 

182 names = [n for e in eyegenbench_datasets(root) if (n := entry_name(e))] 

183 else: 

184 entry = _find_entry(root, dataset) 

185 names = [] if entry is None else [entry_name(entry)] 

186 except (OSError, ValueError): 

187 return False 

188 if not names: 

189 return False 

190 return all( 

191 all((root / name / f"{table}.parquet").is_file() for table in _TABLES) 

192 for name in names 

193 ) 

194 

195 

196def declared_monitor(entry) -> tuple[int, int] | None: 

197 """A manifest row's screen when the corpus actually documents one (I3). 

198 

199 ``monitor_source: "default"`` marks `eyegenbench_geometry.py`'s generic 

200 guess for a corpus that documents no screen at all -- 1920x1080, invented. 

201 Snapping a canvas to an invented screen presents a made-up geometry as the 

202 corpus', so both surfaces decline it: ``None`` means "no declared screen", 

203 and the caller falls back to the data's own extents. 

204 

205 **The rule lives here, once.** The app reads it building each corpus' 

206 registry entry and the CLI reads it resolving ``--eyegenbench``'s canvas; 

207 duplicating the condition is how the same corpus came to render at two 

208 different scales depending on which surface asked. 

209 """ 

210 if not isinstance(entry, dict): 

211 return None 

212 monitor = entry.get("monitor") 

213 if not monitor or entry.get("monitor_source") == "default": 

214 return None 

215 try: 

216 return int(monitor[0]), int(monitor[1]) 

217 except (IndexError, KeyError, TypeError, ValueError): 

218 return None 

219 

220 

221def eyegenbench_monitor(root, dataset: str) -> tuple[int, int] | None: 

222 """The corpus' documented screen in pixels, or ``None`` (I3). 

223 

224 ``None`` when the manifest records no screen for this corpus, or only the 

225 invented default one -- see `declared_monitor`. A `ValueError` still means 

226 the *dataset* isn't in the bundle, which is a different failure and stays 

227 loud. 

228 """ 

229 entry = _find_entry(root, dataset) 

230 if entry is None: 

231 raise ValueError(f"{dataset!r} is not in the bundle at {root!s}") 

232 return declared_monitor(entry) 

233 

234 

235def _dataset_dir(root, dataset: str) -> Path: 

236 root = Path(root) 

237 eyegenbench_manifest(root) # raises FileNotFoundError with the fix 

238 entry = _find_entry(root, dataset) 

239 if entry is None: 

240 raise ValueError(f"{dataset!r} is not in the bundle at {root!s}") 

241 return root / entry["name"] 

242 

243 

244def eyegenbench_raw_frames(root, *, dataset: str) -> tuple[pd.DataFrame, pd.DataFrame]: 

245 """Raw (pre-normalization) ``(words, fixations)`` frames for ``dataset``.""" 

246 directory = _dataset_dir(root, dataset) 

247 # UX-166: "0 of 2" first, so a gated card is armed before the first table. 

248 progress.report(0, 2, unit="tables") 

249 words = pd.read_parquet(directory / "words.parquet") 

250 progress.report(1, 2, unit="tables") 

251 fixations = pd.read_parquet(directory / "fixations.parquet") 

252 progress.report(2, 2, unit="tables") 

253 return words, fixations 

254 

255 

256def load_eyegenbench( 

257 root, *, dataset: str, names: str = "source" 

258) -> tuple[pd.DataFrame, pd.DataFrame]: 

259 """Load an EyeGenBench corpus as normalized ``(words, fixations)``, under the 

260 bundle's own column names (``names="canonical"`` for the internal ones).""" 

261 from .api import load_scanpath_data 

262 

263 words, fixations = eyegenbench_raw_frames(root, dataset=dataset) 

264 return load_scanpath_data( 

265 words, 

266 fixations, 

267 word_schema=EYEGENBENCH_WORD_SCHEMA, 

268 fix_schema=EYEGENBENCH_FIX_SCHEMA, 

269 names=names, 

270 ) 

271 

272 

273def load_eyegenbench_participants(root, dataset: str) -> pd.DataFrame: 

274 """Per-reader metadata -- the DATA-20 participant-metadata table. 

275 

276 Never broadcast onto the word/fixation frames. 

277 """ 

278 return pd.read_parquet(_dataset_dir(root, dataset) / "participants.parquet")