Coverage for scanpath_studio/eyegenbench.py: 90%
87 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Loader for a local bundle of harmonised reading corpora.
3EyeGenBench (https://github.com/EyeBench/EyeGenBench) harmonises many public
4eye-tracking-while-reading corpora into one schema, but discards screen
5geometry. `scripts/prepare_eyegenbench.py` runs their pipeline, recovers the
6geometry, and writes the bundle this module reads. See
7`docs/benchmark-corpora.md` and `plans/data-27-eyegenbench-datasets.md`.
9The pipeline can *load* 39 corpora; a bundle holds however many the user
10prepared, which is fewer -- some publishers require manual acquisition. Read
11the manifest for what is actually there rather than assuming a count.
13Bundle contract for `fixations.parquet`: it must NOT carry a column literally
14named `unique_trial_id`. `data.normalize_fixations` used to key `trial_id` on
15that exact column name whenever it was present, overriding
16`EYEGENBENCH_FIX_SCHEMA`'s `trial` mapping below and breaking the words
17broadcast (paragraph-keyed stimulus-level words vs. reading-keyed fixations
18never matched, silently broadcasting zero word boxes). BUG-58 made the mapping
19authoritative, but the name stays reserved: the normalized frame's own
20`unique_trial_id` is the mapped trial id, so the raw values would not survive
21under it. If the prep script carries EyeGenBench's own finer-grained
22(per-reading) trial identity through at all, it must use the column name
23`eyegenbench_trial_id` instead -- registered as an opaque passthrough in Task 7. It must also give repeated
24readings of the same paragraph by the same participant distinct
25`unique_paragraph_id` values: this loader keys `trial_id` on that column
26directly and does not disambiguate repeats itself.
27"""
29from __future__ import annotations
31import json
32from pathlib import Path
34import pandas as pd
36from . import progress
38MANIFEST_NAME = "manifest.json"
39_TABLES = ("words", "fixations", "participants")
41# Stimulus-level: `participant=None` marks one shared layout rather than one row
42# per reader, so `data.broadcast_stimulus_words` expands it (as for PoTeC).
43EYEGENBENCH_WORD_SCHEMA = dict(
44 participant=None,
45 trial="unique_paragraph_id",
46 word_id="ia_index",
47 text="ia_label",
48 line="line",
49 left="start_x",
50 right="end_x",
51 top="start_y",
52 bottom="end_y",
53 # Declared absent, not merely omitted. A prepared corpus is single-screen by
54 # contract -- the prep script writes one coordinate space per paragraph --
55 # but the frames carry the publisher's leftover columns through, and some
56 # corpora (Provo, SBSAT) keep a `page`. Auto-detection finds it on the
57 # fixations and not here, and `multipart.validate_matching_parts` then
58 # rejects the pair outright ("Multipart identity is present in only one
59 # report"). Saying `None` on BOTH schemas is what stops a leftover column
60 # being read as a screen identity that this bundle does not have.
61 screen_id=None,
62)
64# `trial` is `unique_paragraph_id`, matching the word schema above -- not
65# EyeGenBench's own (finer-grained, per-reading) `unique_trial_id`. See the
66# module docstring: a raw `unique_trial_id` column used to silently override
67# this mapping and break the stimulus-level words broadcast. Keying on
68# unique_paragraph_id is what makes that broadcast join work; repeated
69# readings of the same paragraph by the same participant are NOT separated
70# here (`data._disambiguate_repeated_readings` no-ops without a raw
71# TRIAL_INDEX column, which EyeGenBench frames don't carry) -- the prep
72# script is responsible for giving each reading its own paragraph key
73# upstream (Task 8, R17).
74EYEGENBENCH_FIX_SCHEMA = dict(
75 participant="unique_participant_id",
76 trial="unique_paragraph_id",
77 duration="fix_duration",
78 x="x",
79 y="y",
80 fixation_id="fix_index",
81 word_id="ia_index",
82 screen_id=None, # see EYEGENBENCH_WORD_SCHEMA
83)
86def _manifest_path(root) -> Path:
87 return Path(root) / MANIFEST_NAME
90def eyegenbench_manifest(root) -> dict:
91 """The bundle manifest, or a `FileNotFoundError` naming the fix."""
92 path = _manifest_path(root)
93 if not path.is_file():
94 raise FileNotFoundError(
95 f"No EyeGenBench bundle at {root!s} (missing {MANIFEST_NAME}). "
96 "Build one with: python scripts/prepare_eyegenbench.py --all"
97 )
98 return json.loads(path.read_text(encoding="utf-8"))
101def eyegenbench_datasets(root) -> list:
102 """Manifest entries, one per prepared dataset. Cheap -- no Parquet is read."""
103 return list(eyegenbench_manifest(root).get("datasets", []))
106def entry_name(entry) -> str:
107 """A manifest row's dataset name, or ``""`` when it hasn't got one.
109 A row is data from a file on disk, so it can be malformed. Reading the name
110 as ``entry["name"]`` raised `KeyError` — outside the `(FileNotFoundError,
111 ValueError, OSError)` triple every caller guards with, so **one** nameless
112 row ordered before a valid one took the whole app down through whichever
113 surface looked at the manifest next (M7 / I2). Nameless rows are unusable
114 by definition — nothing can address them — so every reader skips them here
115 rather than each remembering to catch a third exception type.
116 """
117 if not isinstance(entry, dict):
118 return ""
119 return str(entry.get("name") or "").strip()
122def entry_count(entry, key: str) -> int | None:
123 """A manifest row's integer count field: the number, ``0``, or ``None``.
125 The same rule as `entry_name`, for the count fields (`n_texts`,
126 `n_readers`, `n_fixations`, `paragraphs_without_real_boxes`): a manifest is
127 data from a file on disk, and a bare ``int(entry.get(key))`` raises
128 `ValueError`/`TypeError` on ``"many"``, ``[1]`` or any other shape a hand
129 edit can produce — outside the catch every caller guards with, and now on
130 the path that builds a picker entry for every added corpus (N1).
132 ``0`` when the field is absent or blank — *not recorded* is a known
133 quantity for a count, and every reader treats it as none. ``None`` when the
134 value is there but isn't a number, so a caller can tell "no missing texts"
135 from "the missing-text count is unreadable" and word its claim accordingly
136 rather than asserting the confident one.
138 **Never write ``entry_count(...) or 0``.** It collapses `None` into `0` and
139 so reads an unreadable count as *nothing missing* — which is precisely how
140 a corpus with unknown coverage gets badged a confident "Real", the
141 overclaim R34 exists to prevent. Test the two cases apart (``if count :=
142 entry_count(...)`` is fine — it drops both, which is right when the value is
143 only being formatted), or handle `None` explicitly.
144 """
145 if not isinstance(entry, dict):
146 return None
147 raw = entry.get(key)
148 if raw is None or (isinstance(raw, str) and not raw.strip()):
149 return 0
150 try:
151 return int(raw)
152 except (TypeError, ValueError):
153 return None
156def _find_entry(root, dataset: str) -> dict | None:
157 """The manifest entry named ``dataset`` (case-insensitive), or ``None``.
159 Shared by every function that resolves a dataset name against the
160 manifest, so the case-insensitive comparison lives in exactly one place.
161 """
162 for entry in eyegenbench_datasets(root):
163 if (name := entry_name(entry)) and name.lower() == str(dataset).lower():
164 return entry
165 return None
168def eyegenbench_present(root, dataset: str | None = None) -> bool:
169 """True when the bundle holds everything a load needs. Path stats only.
171 Strict on purpose: a lenient check passes a partial tree and then crashes
172 mid-load, whereas a strict one lets the app offer the fix. That includes
173 resolving ``dataset`` against the manifest first -- a directory holding
174 all three Parquet files under a name absent from the manifest must not
175 read as present, since loading that same name would still raise.
176 """
177 root = Path(root)
178 if not _manifest_path(root).is_file():
179 return False
180 try:
181 if dataset is None:
182 names = [n for e in eyegenbench_datasets(root) if (n := entry_name(e))]
183 else:
184 entry = _find_entry(root, dataset)
185 names = [] if entry is None else [entry_name(entry)]
186 except (OSError, ValueError):
187 return False
188 if not names:
189 return False
190 return all(
191 all((root / name / f"{table}.parquet").is_file() for table in _TABLES)
192 for name in names
193 )
196def declared_monitor(entry) -> tuple[int, int] | None:
197 """A manifest row's screen when the corpus actually documents one (I3).
199 ``monitor_source: "default"`` marks `eyegenbench_geometry.py`'s generic
200 guess for a corpus that documents no screen at all -- 1920x1080, invented.
201 Snapping a canvas to an invented screen presents a made-up geometry as the
202 corpus', so both surfaces decline it: ``None`` means "no declared screen",
203 and the caller falls back to the data's own extents.
205 **The rule lives here, once.** The app reads it building each corpus'
206 registry entry and the CLI reads it resolving ``--eyegenbench``'s canvas;
207 duplicating the condition is how the same corpus came to render at two
208 different scales depending on which surface asked.
209 """
210 if not isinstance(entry, dict):
211 return None
212 monitor = entry.get("monitor")
213 if not monitor or entry.get("monitor_source") == "default":
214 return None
215 try:
216 return int(monitor[0]), int(monitor[1])
217 except (IndexError, KeyError, TypeError, ValueError):
218 return None
221def eyegenbench_monitor(root, dataset: str) -> tuple[int, int] | None:
222 """The corpus' documented screen in pixels, or ``None`` (I3).
224 ``None`` when the manifest records no screen for this corpus, or only the
225 invented default one -- see `declared_monitor`. A `ValueError` still means
226 the *dataset* isn't in the bundle, which is a different failure and stays
227 loud.
228 """
229 entry = _find_entry(root, dataset)
230 if entry is None:
231 raise ValueError(f"{dataset!r} is not in the bundle at {root!s}")
232 return declared_monitor(entry)
235def _dataset_dir(root, dataset: str) -> Path:
236 root = Path(root)
237 eyegenbench_manifest(root) # raises FileNotFoundError with the fix
238 entry = _find_entry(root, dataset)
239 if entry is None:
240 raise ValueError(f"{dataset!r} is not in the bundle at {root!s}")
241 return root / entry["name"]
244def eyegenbench_raw_frames(root, *, dataset: str) -> tuple[pd.DataFrame, pd.DataFrame]:
245 """Raw (pre-normalization) ``(words, fixations)`` frames for ``dataset``."""
246 directory = _dataset_dir(root, dataset)
247 # UX-166: "0 of 2" first, so a gated card is armed before the first table.
248 progress.report(0, 2, unit="tables")
249 words = pd.read_parquet(directory / "words.parquet")
250 progress.report(1, 2, unit="tables")
251 fixations = pd.read_parquet(directory / "fixations.parquet")
252 progress.report(2, 2, unit="tables")
253 return words, fixations
256def load_eyegenbench(
257 root, *, dataset: str, names: str = "source"
258) -> tuple[pd.DataFrame, pd.DataFrame]:
259 """Load an EyeGenBench corpus as normalized ``(words, fixations)``, under the
260 bundle's own column names (``names="canonical"`` for the internal ones)."""
261 from .api import load_scanpath_data
263 words, fixations = eyegenbench_raw_frames(root, dataset=dataset)
264 return load_scanpath_data(
265 words,
266 fixations,
267 word_schema=EYEGENBENCH_WORD_SCHEMA,
268 fix_schema=EYEGENBENCH_FIX_SCHEMA,
269 names=names,
270 )
273def load_eyegenbench_participants(root, dataset: str) -> pd.DataFrame:
274 """Per-reader metadata -- the DATA-20 participant-metadata table.
276 Never broadcast onto the word/fixation frames.
277 """
278 return pd.read_parquet(_dataset_dir(root, dataset) / "participants.parquet")