Coverage for scanpath_studio/eyegenbench_geometry.py: 97%
203 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Screen geometry for EyeGenBench corpora.
3EyeGenBench's harmonised output records *which* interest area each fixation
4landed on and *where within it* (a 0-1 offset), but no pixel coordinates and no
5word boxes. Scanpath Studio needs boxes. This module recovers them at the best
6fidelity available -- see `resolve_geometry` for the four tiers.
7"""
9from __future__ import annotations
11from dataclasses import dataclass
13import pandas as pd
16@dataclass(frozen=True)
17class DisplaySpec:
18 """How one corpus presented its text, in pixels.
20 ``source`` cites where the numbers came from (e.g. ``"pymovements:potec"``),
21 so a reconstructed layout can always be traced back to its evidence.
22 """
24 width_px: int
25 height_px: int
26 font_px: float
27 char_width_px: float
28 line_pitch_px: float
29 monospaced: bool
30 margin_px: int
31 source: str
34# Provisional -- DATA-27 open decision. A synthesized layout needs *some*
35# screen; this is the one used when nothing at all is published for a corpus.
36DEFAULT_SPEC = DisplaySpec(
37 width_px=1920,
38 height_px=1080,
39 font_px=20,
40 char_width_px=12.0, # Courier-family advance width at 20px.
41 line_pitch_px=60.0, # Double-spaced, the reading-study norm.
42 monospaced=True,
43 margin_px=100,
44 source="default",
45)
48def layout_words(ia_list: list[str], spec: DisplaySpec) -> pd.DataFrame:
49 """Lay ``ia_list`` out as word boxes on ``spec``'s screen.
51 Greedy left-to-right wrapping with one space between words. Shared by the
52 reconstructed and synthesized tiers -- same code, different ``spec``.
53 """
54 if not len(ia_list):
55 raise ValueError("layout_words needs at least one interest area")
57 usable = spec.width_px - 2 * spec.margin_px
58 space = spec.char_width_px
59 rows = []
60 x, line = spec.margin_px, 0
61 for word_id, text in enumerate(ia_list):
62 width = max(len(str(text)), 1) * spec.char_width_px
63 if x > spec.margin_px and (x + width) > (spec.margin_px + usable):
64 x, line = spec.margin_px, line + 1
65 top = spec.margin_px + line * spec.line_pitch_px
66 rows.append(
67 {
68 "word_id": word_id,
69 "text": str(text),
70 "line": line,
71 "start_x": x,
72 "end_x": x + width,
73 "start_y": top,
74 "end_y": top + spec.font_px,
75 }
76 )
77 x += width + space
78 return pd.DataFrame(rows)
81def _spec(
82 width_px,
83 height_px,
84 font_px,
85 *,
86 source,
87 mono=True,
88 chars_per_deg=None,
89 double_spaced=False,
90 distance_cm=None,
91 width_cm=None,
92 margin_px=100,
93) -> DisplaySpec:
94 """Build a DisplaySpec from what a corpus actually reported.
96 ``chars_per_deg`` + ``distance_cm`` + ``width_cm`` gives a measured
97 character width; otherwise fall back to the monospace advance ratio.
98 """
99 if chars_per_deg and distance_cm and width_cm:
100 px_per_cm = width_px / width_cm
101 cm_per_deg = 2 * distance_cm * 0.00872686779 # tan(0.5 deg)
102 char_width = (cm_per_deg * px_per_cm) / chars_per_deg
103 else:
104 char_width = font_px * 0.6
105 return DisplaySpec(
106 width_px=width_px,
107 height_px=height_px,
108 font_px=font_px,
109 char_width_px=char_width,
110 line_pitch_px=font_px * (2.0 if double_spaced else 1.5),
111 monospaced=mono,
112 margin_px=margin_px,
113 source=source,
114 )
117# Published display parameters, best source first: pymovements dataset YAMLs ->
118# the UZH dataset review table -> the corpus' own paper. Anything not listed
119# here falls back to DEFAULT_SPEC and is stamped `synthesized`.
120#
121# R49/R50 -- the entry bar is a PUBLISHED parameter, not a plausible one. An
122# entry here upgrades a corpus from `synthesized` to `reconstructed`, i.e. from
123# "this geometry is invented" to "this geometry follows the corpus' own
124# reported screen"; a plausible-but-unsourced entry makes an invented layout
125# indistinguishable from a sourced one, which is the single thing the tier
126# system exists to prevent. So: no modal/mean screen for a multi-site corpus
127# (each site's readers would be laid out on another site's screen), no
128# borrowing a sibling corpus' apparatus, no filling a missing pixel resolution
129# from the hardware of the era. Where a value comes from somewhere other than
130# the corpus' own paper (e.g. a monitor datasheet for a model the paper names),
131# `source` says so rather than presenting it as a paper value.
132DISPLAY_SPECS: dict[str, DisplaySpec] = {
133 "potec": _spec(
134 1680, 1050, 20, source="pymovements:potec", width_cm=47.5, distance_cm=65
135 ),
136 "copco": _spec(
137 1920,
138 1080,
139 14,
140 source="pymovements:copco + paper:Hollenstein2022",
141 double_spaced=True,
142 width_cm=59.0,
143 distance_cm=85,
144 ),
145 "emtec": _spec(
146 1280,
147 1024,
148 14,
149 source="pymovements:emtec + uzh",
150 chars_per_deg=2.86,
151 width_cm=38.2,
152 distance_cm=60,
153 ),
154 "colagaze": _spec(
155 1280,
156 1024,
157 17,
158 source="pymovements:colagaze + uzh",
159 chars_per_deg=2.0,
160 width_cm=54.37,
161 distance_cm=60,
162 ),
163 "interead": _spec(
164 1920,
165 1080,
166 16,
167 source="pymovements:interead + uzh",
168 width_cm=52.8,
169 distance_cm=57,
170 ),
171 "ggtg": _spec(
172 1100,
173 900,
174 20,
175 source="pymovements:ggtg + uzh",
176 mono=False,
177 double_spaced=True,
178 width_cm=31.2,
179 distance_cm=66,
180 ),
181 "etdd70": _spec(
182 1680,
183 1050,
184 20,
185 source="pymovements:etdd70 + uzh",
186 mono=False,
187 distance_cm=65,
188 ),
189 "gaze4hate": _spec(
190 2560,
191 1440,
192 20,
193 source="pymovements:gaze4hate",
194 width_cm=59.8,
195 distance_cm=78.0,
196 ),
197 "raccoons": _spec(
198 1920,
199 1080,
200 20,
201 source="pymovements:raccoons",
202 width_cm=56.8,
203 distance_cm=105.5,
204 ),
205 "sbsat": _spec(
206 1024,
207 768,
208 20,
209 source="pymovements:sb_sat",
210 width_cm=44.5,
211 distance_cm=70,
212 ),
213 "provo": _spec(
214 1600,
215 900,
216 20,
217 source="paper:LukeChristianson2018",
218 chars_per_deg=3.0,
219 width_cm=40.0,
220 distance_cm=60,
221 ),
222 "psr": _spec(
223 1024,
224 768,
225 18,
226 source="uzh",
227 chars_per_deg=2.38,
228 distance_cm=73,
229 width_cm=47.0,
230 ),
231 "eyevoicespan": _spec(
232 1280,
233 960,
234 24,
235 source="uzh",
236 chars_per_deg=2.22,
237 distance_cm=60,
238 width_cm=40.0,
239 ),
240 "iitbhgc": _spec(
241 1920,
242 1080,
243 20,
244 source="uzh",
245 mono=False,
246 distance_cm=70,
247 ),
248 "bsc": _spec(
249 1024,
250 768,
251 20,
252 source="uzh",
253 chars_per_deg=0.75,
254 distance_cm=43,
255 width_cm=36.0,
256 ),
257 "chinesereading": _spec(1024, 768, 20, source="uzh", distance_cm=58),
258 "cuentos": _spec(1920, 1080, 24, source="uzh", distance_cm=55),
259 "zuco1": _spec(1920, 1080, 20, source="paper:Hollenstein2018", mono=False),
260 "zuco2": _spec(1920, 1080, 20, source="paper:Hollenstein2020", mono=False),
261 # R49 -- OneStop publishes more of its layout than any other corpus here:
262 # monitor, display area, both viewing distances, the letter cell in px and
263 # the line pitch in px (Berzak et al. 2025, Sci Data 12:1995,
264 # doi:10.1038/s41597-025-06272-2, Methods -> Apparatus). Two judgement
265 # calls, both choices *between* published numbers rather than inventions:
266 #
267 # * `distance_cm=75.0`. The paper gives eye level as 750 mm from the top
268 # of the display area and 795 mm from its bottom, so a scalar has to
269 # pick an end. 75.0 is the right one twice over: the text starts at
270 # y=186 px, near the top, and it is the only end at which the paper's
271 # other two published figures agree -- 0.34 deg per letter at 75 cm
272 # reproduces the published 19 px letter cell (19.1 px), where 79.5 cm
273 # would give 20.2 px, 6% wide.
274 # * `margin_px=368` reproduces the published *text column* (2560 - 2*368 =
275 # 1824 px = 96 characters) rather than the published *left edge* (300
276 # px). `_spec`'s margin model is symmetric and OneStop's layout is not
277 # (left 300, column 1824, right 436), so only one of the two survives.
278 # Column width is what `layout_words` actually consumes -- it decides
279 # where lines wrap -- so matching it puts every word on the right line,
280 # at the cost of shifting the whole block 68 px right.
281 #
282 # `double_spaced=True` is not an assumption here: 38 x 2 = 76 px is exactly
283 # the "triple spacing (76 px)" the paper reports. This entry is also the
284 # single sourced home for the 2560x1440 that app.py / cli.py / datasets.py
285 # pin for the *native* OneStop corpus; those sites now cite it.
286 "onestop": _spec(
287 2560,
288 1440,
289 38, # published letter cell 19 px x 38 px (25 pt at this screen's 109 ppi)
290 source=(
291 "paper:Berzak2025-OneStop"
292 " (distance_cm=75, the published eye-to-top-of-display;"
293 " margin fits the published 1824 px text column)"
294 ),
295 chars_per_deg=2.94, # 1 / 0.34 deg per letter, stated
296 distance_cm=75.0,
297 width_cm=59.7, # 597 mm display area, stated
298 double_spaced=True,
299 margin_px=368,
300 ),
301 # R49 -- Laurinavichyute et al. 2019, Behav Res Methods 51(3):1161-1178,
302 # doi:10.3758/s13428-018-1051-6, Method -> Procedure states the monitor and
303 # resolution, 22 pt Courier New, 90 cm, and 0.29 deg per character.
304 # `font_px=28` converts the published 22 pt at this screen's 92 ppi -- it is
305 # NOT the point size copied into a pixel field. Cross-check: 22 pt Courier
306 # New has a 16.8 px advance at 92 ppi and the published 0.29 deg/char at
307 # 90 cm gives 16.5 px, two independent published routes agreeing to 2%,
308 # which is also what justifies pairing them with the datasheet width.
309 # Single-sentence trials, so nothing about line spacing is published (there
310 # are no line breaks) and `line_pitch_px` only sets box height.
311 "rsc": _spec(
312 1920,
313 1080,
314 28, # 22 pt at 1920 px / 53.1 cm = 92 ppi
315 source=(
316 "paper:Laurinavichyute2019-RSC"
317 " (width_cm from the ASUS VG248QE datasheet, the model the paper names)"
318 ),
319 chars_per_deg=3.45, # 1 / 0.29 deg per character, stated
320 distance_cm=90,
321 width_cm=53.1,
322 ),
323 # R49 -- Yan, Pan & Kliegl 2025, Behav Res Methods 57(2):60,
324 # doi:10.3758/s13428-024-02523-z, Apparatus. BSC-II's apparatus does NOT
325 # carry over from `bsc` above and the asymmetry between the two entries is
326 # real, not a gap: BSC was a 19" CRT at 1024x768 / 43 cm / 0.75 chars per
327 # degree, BSC-II a 24.5" LCD at 1920x1080 / 70 cm / 0.909. Inheriting BSC's
328 # numbers would have been wrong by ~2x in px per degree.
329 # A Song-font CJK glyph cell is square, so the published 1.1 deg per
330 # character IS the font size (~48 px); the harmonised interest areas are
331 # words, and `layout_words` multiplies per-character, so the per-character
332 # width is the correct thing to hand it.
333 "bscii": _spec(
334 1920,
335 1080,
336 48, # square CJK cell = the published character width
337 source=(
338 "paper:YanPanKliegl2025-BSCII"
339 " (width_cm from the BenQ ZOWIE XL2546K datasheet, the model the"
340 " paper names)"
341 ),
342 chars_per_deg=0.909, # 1 / 1.1 deg per character, stated
343 distance_cm=70,
344 width_cm=53.8,
345 ),
346 # R50 -- `mecol2w2` USED to be here as
347 # `_spec(1920, 1080, 21, source="uzh + paper:MECO-L2-W2")`. It was removed
348 # after both of its cited sources were re-read, and neither contains it:
349 #
350 # * Kuperman et al. 2025 (SSLA, doi:10.1017/S0272263125000105), Procedure:
351 # "A mono-spaced font (Consolas) was used, with a size generally ranging
352 # from 20 to 22 points (given variation in screen size and resolution at
353 # different testing sites) and 1.5 line spacing. In accordance with
354 # their local experimental setup, the German site in Zurich used a
355 # smaller font size of 10 with a lower resolution of 1280 x 1024. ... For
356 # further specifications of the screen, font size, presentation settings,
357 # and apparatus at each participating site, see supplementary material
358 # S2." The string "1920" does not occur in the paper at all; the only
359 # resolution it prints is one site's exception.
360 # * The UZH review row for "MECO L2 2nd Wave" has `Resolution: null` and
361 # puts a LINK to those same supplementary materials in its `Monitor` and
362 # eye-to-screen-distance cells -- the curators' way of writing "per
363 # site".
364 #
365 # So 1920x1080 was corpus-level for no site in particular, and `font_px=21`
366 # was the midpoint of a 20-22 **pt** range used as **px** -- a unit error
367 # on its own terms, and one that ignored both the range and the 10 pt site.
368 # Falling back to `synthesized` is a user-visible downgrade and the correct
369 # direction: an unsourced `reconstructed` is worse than an honest
370 # `synthesized`, because the user cannot tell it is a guess.
371 #
372 # Decided the same way for both MECO L2 waves, which share sites, protocol
373 # and materials -- `mecol2w1` gets no entry either, and neither do
374 # `mecol1w1` / `mecol1w2`. Wave 1 of L1 states outright that "maintaining
375 # an identical font size, distance from the screen, and screen resolution
376 # was unfeasible"; Wave 2 of L1 tabulates 16 different screens (Table 10,
377 # Siegelman et al. 2025). These corpora have no corpus-level screen to
378 # publish. Per-lab geometry would need a per-site table, not a per-corpus
379 # one; the site code is recoverable from the data if that is ever built.
380 #
381 # `psc2` is deliberately absent too, and it is the easiest one to "fix"
382 # wrongly. A complete apparatus IS published for the PSC2 sentences
383 # (1280x960, Courier New 24 pt, 14 px/letter, 60 cm; Laubrock & Kliegl
384 # 2015, Front Psychol 6:1432) -- but for the 32-reader ORAL reading
385 # experiment, which is already in this table above as `eyevoicespan`. The
386 # corpus shipped as `psc2` is a 149-reader SILENT reading collection whose
387 # only cited document (Heister, Wuerzner & Kliegl 2012) has no methods
388 # section at all. Same lab, same materials, probably the same room -- which
389 # is exactly why copying it across would be invisibly wrong: the two
390 # corpora would become indistinguishable here, and nothing in the data
391 # would reveal the error. Pinned by
392 # `tests/test_eyegenbench_geometry.py::test_psc2_stays_synthesized_...`.
393}
396def display_spec_for(dataset: str) -> DisplaySpec:
397 """The published screen for ``dataset``, or ``DEFAULT_SPEC`` if none is known."""
398 return DISPLAY_SPECS.get(str(dataset).lower(), DEFAULT_SPEC)
401IA_DATA_COLUMN = "CURRENT_FIX_INTEREST_AREA_DATA"
402_BOX_COLUMNS = ["start_x", "start_y", "end_x", "end_y"]
405def parse_ia_data(series: pd.Series) -> pd.DataFrame:
406 """EyeLink ``[STATIC, RECTANGLE, left, top, right, bottom]`` -> box columns.
408 This is the corpus' *real* on-screen word box. EyeGenBench reads it, divides
409 it into a normalised landing position, and discards it; we keep it.
410 """
411 parts = series.astype(str).str.strip("[]").str.split(",", expand=True)
412 if parts.shape[1] < 6:
413 return pd.DataFrame(
414 {c: [float("nan")] * len(series) for c in _BOX_COLUMNS}, index=series.index
415 )
416 out = pd.DataFrame(index=series.index)
417 for name, idx in zip(_BOX_COLUMNS, (2, 3, 4, 5)):
418 out[name] = pd.to_numeric(parts[idx].str.strip(), errors="coerce")
419 return out
422def extract_eyelink_boxes(
423 frame: pd.DataFrame,
424 *,
425 paragraph_col: str = "unique_paragraph_id",
426 ia_col: str = "ia_index",
427 data_col: str = IA_DATA_COLUMN,
428) -> pd.DataFrame:
429 """One real box per ``(paragraph, interest area)`` found in ``frame``.
431 Only interest areas that were *fixated* appear -- the column rides on
432 fixations. `fill_missing_boxes` completes the rest.
434 R28: screen coordinates are non-negative by definition, so a box whose
435 ``start_x``/``start_y`` is negative is malformed data, not a real
436 on-screen position -- it is dropped here rather than earning the `real`
437 stamp. This is the same discipline `extract_text_df_boxes` applies to its
438 own real-box source; both feed the same stamp, so both must guarantee it.
440 R33: the raw-EyeLink tier is an optional upgrade (tier 1 of 4). All three
441 columns this function indexes -- ``paragraph_col``, ``ia_col`` and
442 ``data_col`` -- must be present before any of them is read; when one is
443 missing this returns the empty, correctly-columned frame so
444 `resolve_geometry` falls through to the next tier instead of raising.
445 Verified against a real corpus: onestop's raw EyeLink export carries
446 ``CURRENT_FIX_INTEREST_AREA_DATA`` but neither ``unique_paragraph_id`` nor
447 ``ia_index`` (an un-prefixed ``paragraph_id`` and no interest-area index
448 at all) -- a hard skip here used to drop the entire corpus from the
449 catalogue behind a one-line log entry, when the bundle the reconstructed/
450 synthesized tiers would have produced is perfectly good.
451 """
452 if not {paragraph_col, ia_col, data_col} <= set(frame.columns):
453 return pd.DataFrame(columns=[paragraph_col, ia_col, *_BOX_COLUMNS])
454 boxes = pd.concat(
455 [
456 frame[[paragraph_col, ia_col]].reset_index(drop=True),
457 parse_ia_data(frame[data_col]).reset_index(drop=True),
458 ],
459 axis=1,
460 )
461 boxes = boxes.dropna(subset=_BOX_COLUMNS)
462 boxes = boxes[(boxes["start_x"] >= 0) & (boxes["start_y"] >= 0)]
463 boxes = boxes.drop_duplicates(subset=[paragraph_col, ia_col], keep="first")
464 return boxes.sort_values([paragraph_col, ia_col]).reset_index(drop=True)
467_TEXT_DF_BOX_COLUMNS = ["start_x", "start_y", "end_x", "end_y"]
470def extract_text_df_boxes(
471 text_df: pd.DataFrame,
472 *,
473 paragraph_col: str = "unique_paragraph_id",
474 ia_col: str = "ia_index",
475) -> pd.DataFrame:
476 """One real box per ``(paragraph, interest area)`` straight off ``text_df``.
478 Some EyeGenBench corpora (verified: PoTeC) carry EyeLink's own genuinely
479 measured on-screen coordinates directly on the harmonised ``text_df`` --
480 one row per ``(paragraph, interest area)``, each repeating that
481 paragraph's whole ``ia_list``. This is **not** part of EyeGenBench's
482 documented schema (``start_x`` appears in no ``.py``/``.yaml`` in the
483 EyeGenBench source -- it survives only as source-column pass-through), so
484 it is detected here at runtime, per dataset, and never assumed to exist.
486 Presence is not trust: a row only contributes a box when its ``ia_index``
487 is a valid finite non-negative whole number (`_valid_ia_index`, the same
488 discipline `extract_eyelink_boxes` uses) *and* the box itself is finite,
489 non-negative (``start_x >= 0`` and ``start_y >= 0`` -- R28: a screen
490 coordinate is non-negative by definition, and since ``start < end`` is
491 already required, a non-negative start also bounds ``end`` -- a box
492 straddling the origin, e.g. ``start_x=-5, end_x=40``, is rejected too, not
493 representable on a 0-origin canvas), with ``start_x < end_x`` and
494 ``start_y < end_y``. A row failing any check contributes nothing -- never
495 a guessed or truncated position, and never a `real` stamp for coordinates
496 that might turn out to be some other space in a corpus not yet verified.
497 """
498 if not all(c in text_df.columns for c in [ia_col, *_TEXT_DF_BOX_COLUMNS]):
499 # The four box columns alone don't guarantee ia_index is there too --
500 # the brief's own TEXTS fixture is exactly that shape (no ia_index).
501 # Without this check that row indexing below raised KeyError, which
502 # dropped the whole dataset from the manifest instead of falling
503 # through to the reconstructed/synthesized tiers as it must.
504 return pd.DataFrame(columns=[paragraph_col, ia_col, *_TEXT_DF_BOX_COLUMNS])
506 ia_index = _valid_ia_index(text_df[ia_col])
507 numeric_box = text_df[_TEXT_DF_BOX_COLUMNS].apply(pd.to_numeric, errors="coerce")
508 is_finite = numeric_box.apply(
509 lambda s: (s > float("-inf")) & (s < float("inf"))
510 ).all(axis=1)
511 valid = (
512 ia_index.notna()
513 & is_finite
514 & (numeric_box["start_x"] >= 0)
515 & (numeric_box["start_y"] >= 0)
516 & (numeric_box["start_x"] < numeric_box["end_x"])
517 & (numeric_box["start_y"] < numeric_box["end_y"])
518 )
520 out = pd.DataFrame(
521 {
522 paragraph_col: text_df[paragraph_col],
523 ia_col: ia_index,
524 **{c: numeric_box[c] for c in _TEXT_DF_BOX_COLUMNS},
525 }
526 )
527 out = out.loc[valid]
528 out = out.drop_duplicates(subset=[paragraph_col, ia_col], keep="first")
529 return out.sort_values([paragraph_col, ia_col]).reset_index(drop=True)
532def _valid_ia_index(series: pd.Series) -> pd.Series:
533 """``ia_index`` as validated numbers, ``NaN`` for anything invalid.
535 An interest-area index is a finite whole number ``>= 0`` -- nothing
536 fractional, negative, infinite or NaN is ever a valid EyeLink/EyeGenBench
537 index, whatever `pd.to_numeric` manages to parse out of a row. Parses
538 first (handling "." and NaN, EyeGenBench's own documented "landed on no
539 interest area" sentinels, and string-typed columns), then keeps only
540 finite, non-negative whole numbers: a fractional value like ``0.9`` or
541 ``-0.5`` is exactly as invalid as ``"abc"``, and ``inf`` is exactly as
542 invalid as either -- never truncated, overflowed, or guessed at.
544 Deliberately returned as plain (float) numbers, not cast to a
545 fixed-width integer type -- nothing here bounds the magnitude of an
546 otherwise-valid-shaped index; an oversized one (say, larger than any
547 real paragraph could have) is rejected the same way a merely
548 past-the-end one already is, by the existing ``< count`` checks in
549 `fill_missing_boxes` and `_paragraphs_with_real_boxes`. A fixed-width
550 cast over the whole column would instead raise the moment one row
551 exceeded its range, poisoning every genuine box beside it rather than
552 rejecting the one bad row. Only a value already validated here is ever
553 handed to Python's own arbitrary-precision ``int()`` (in
554 `fill_missing_boxes`), where it can no longer overflow anything.
556 Shared by `fill_missing_boxes` and `_paragraphs_with_real_boxes` so the
557 two accept and reject exactly the same set of rows, by construction
558 rather than by convention -- neither one raises on a row this returns
559 ``NaN`` for, and neither one guesses a position for it.
560 """
561 numeric = pd.to_numeric(series, errors="coerce")
562 is_finite = (numeric > float("-inf")) & (numeric < float("inf"))
563 is_valid_index = is_finite & (numeric >= 0) & (numeric == numeric.round())
564 return numeric.where(is_valid_index)
567def fill_missing_boxes(
568 boxes: pd.DataFrame, ia_counts: dict[str, int], spec: DisplaySpec
569) -> tuple[pd.DataFrame, float]:
570 """Complete ``boxes`` so every interest area of every paragraph has one.
572 Real boxes only cover *fixated* areas. A row whose ``ia_index`` isn't a
573 valid finite non-negative whole number (`_valid_ia_index`; EyeLink's own
574 sentinel for "landed on no interest area" is ``.`` or ``NaN``, and a
575 fractional, negative, or infinite value is exactly as invalid) is
576 treated exactly the same as a genuinely unfixated index -- simply absent
577 from the lookup, never a raised error and never a guessed or truncated
578 position. Because `_valid_ia_index` already excludes anything not
579 finite, converting a *validated* value with plain ``int()`` below can
580 never overflow, however large -- it is Python's own arbitrary-precision
581 conversion, not a fixed-width one. Consecutive gaps (including those)
582 are filled as runs (maximal consecutive stretches of missing indices):
584 - Bracketed on both sides, same line: divide the space evenly between them
585 - Bracketed across a line break: continue rightward from L on L's line
586 - Trailing (L exists, no R): advance rightward from L
587 - Leading (no L, R exists): the same bracketed formula, treating
588 ``margin_px`` as a virtual left anchor -- divide the space between it and
589 R evenly (zero-width boxes at ``margin_px`` when there is no room, i.e.
590 ``available <= 0``)
591 - No anchor (neither L nor R): distribute evenly across the screen
593 Returns ``(filled, interpolated_fraction)`` so the manifest can report
594 how much of the geometry is inferred rather than measured.
595 """
596 width = spec.char_width_px * 4
597 out = []
598 n_filled = 0
599 n_total = 0
601 for paragraph, count in ia_counts.items():
602 present = boxes[boxes["unique_paragraph_id"] == paragraph]
603 valid_index = _valid_ia_index(present["ia_index"])
604 by_index = {
605 int(ia): row
606 for ia, row in zip(valid_index, present.itertuples())
607 if pd.notna(ia)
608 }
609 n_total += count
611 # Process all indices, identifying runs of consecutive missing indices
612 idx = 0
613 while idx < count:
614 if idx in by_index:
615 # Real box: add it as-is
616 row = by_index[idx]
617 out.append(
618 {
619 "unique_paragraph_id": paragraph,
620 "ia_index": idx,
621 "start_x": row.start_x,
622 "start_y": row.start_y,
623 "end_x": row.end_x,
624 "end_y": row.end_y,
625 }
626 )
627 idx += 1
628 else:
629 # Start of a run of consecutive missing indices
630 run_start = idx
631 while idx < count and idx not in by_index:
632 idx += 1
633 run_end = idx - 1
634 k = run_end - run_start + 1
636 # Find L (nearest real box before the run)
637 L = None
638 for i in range(run_start - 1, -1, -1):
639 if i in by_index:
640 L = by_index[i]
641 break
643 # Find R (nearest real box after the run)
644 R = None
645 for i in range(run_end + 1, count):
646 if i in by_index:
647 R = by_index[i]
648 break
650 # Fill the run based on anchor availability and line position
651 if L is not None and R is not None:
652 if L.start_y == R.start_y:
653 # Bracketed on both sides, same line
654 span = R.start_x - L.end_x
655 if span > 0:
656 slot = span / k
657 for i in range(k):
658 out.append(
659 {
660 "unique_paragraph_id": paragraph,
661 "ia_index": run_start + i,
662 "start_x": L.end_x + i * slot,
663 "start_y": L.start_y,
664 "end_x": L.end_x + (i + 1) * slot,
665 "end_y": L.end_y,
666 }
667 )
668 else:
669 # Zero or negative span, use zero-width boxes at L.end_x
670 for i in range(k):
671 out.append(
672 {
673 "unique_paragraph_id": paragraph,
674 "ia_index": run_start + i,
675 "start_x": L.end_x,
676 "start_y": L.start_y,
677 "end_x": L.end_x,
678 "end_y": L.end_y,
679 }
680 )
681 n_filled += k
682 else:
683 # Bracketed but across line break, treat as trailing
684 for i in range(k):
685 out.append(
686 {
687 "unique_paragraph_id": paragraph,
688 "ia_index": run_start + i,
689 "start_x": L.end_x
690 + i * (width + spec.char_width_px)
691 + spec.char_width_px,
692 "start_y": L.start_y,
693 "end_x": L.end_x
694 + i * (width + spec.char_width_px)
695 + spec.char_width_px
696 + width,
697 "end_y": L.end_y,
698 }
699 )
700 n_filled += k
701 elif L is not None:
702 # Trailing run: advance rightward from L
703 for i in range(k):
704 out.append(
705 {
706 "unique_paragraph_id": paragraph,
707 "ia_index": run_start + i,
708 "start_x": L.end_x
709 + i * (width + spec.char_width_px)
710 + spec.char_width_px,
711 "start_y": L.start_y,
712 "end_x": L.end_x
713 + i * (width + spec.char_width_px)
714 + spec.char_width_px
715 + width,
716 "end_y": L.end_y,
717 }
718 )
719 n_filled += k
720 elif R is not None:
721 # Leading run: divide space from margin_px to R using bracketed formula
722 available = R.start_x - spec.margin_px
723 if available > 0:
724 slot = available / k
725 for i in range(k):
726 out.append(
727 {
728 "unique_paragraph_id": paragraph,
729 "ia_index": run_start + i,
730 "start_x": spec.margin_px + i * slot,
731 "start_y": R.start_y,
732 "end_x": spec.margin_px + (i + 1) * slot,
733 "end_y": R.end_y,
734 }
735 )
736 else:
737 # No room before R, use zero-width boxes at margin_px
738 for i in range(k):
739 out.append(
740 {
741 "unique_paragraph_id": paragraph,
742 "ia_index": run_start + i,
743 "start_x": spec.margin_px,
744 "start_y": R.start_y,
745 "end_x": spec.margin_px,
746 "end_y": R.end_y,
747 }
748 )
749 n_filled += k
750 else:
751 # No anchor: distribute evenly across usable screen width
752 usable_width = spec.width_px - 2 * spec.margin_px
753 slot = usable_width / k if k > 0 else 0
754 for i in range(k):
755 out.append(
756 {
757 "unique_paragraph_id": paragraph,
758 "ia_index": run_start + i,
759 "start_x": spec.margin_px + i * slot,
760 "start_y": spec.margin_px,
761 "end_x": spec.margin_px + (i + 1) * slot,
762 "end_y": spec.margin_px + spec.font_px,
763 }
764 )
765 n_filled += k
767 fraction = (n_filled / n_total) if n_total else 0.0
768 return pd.DataFrame(out), fraction
771GEOMETRY_REAL = "real"
772GEOMETRY_RECONSTRUCTED = "reconstructed"
773GEOMETRY_SYNTHESIZED = "synthesized"
776def _paragraphs_with_real_boxes(boxes: pd.DataFrame, ia_counts: dict) -> set:
777 """Paragraph ids that contributed at least one real, in-range box.
779 A raw ``ia_index`` only "counts" if `fill_missing_boxes` could actually
780 place it -- it only ever considers indices ``0..count-1`` for a paragraph,
781 so an index past the end (a harmonised-text/raw-export mismatch, however
782 large) is silently never used, checked here as the remaining upper bound.
783 Every other way an index can be unusable -- negative (EyeLink writes
784 ``-1`` for a fixation that landed on no interest area at all, an expected
785 raw-export shape, not a corner case), fractional, non-finite, or
786 unparseable (EyeGenBench's own documented convention for that same event
787 is ``.`` or ``NaN``) -- is already rejected by `_valid_ia_index` itself,
788 the same helper `fill_missing_boxes` uses to build its lookup. The two
789 functions accept and reject exactly the same set of rows by construction,
790 so a row neither of them can use is never counted as a contribution and
791 never raises.
792 """
793 ia_index = _valid_ia_index(boxes["ia_index"])
794 counts = boxes["unique_paragraph_id"].map(ia_counts)
795 in_range = ia_index < counts
796 return set(boxes.loc[in_range, "unique_paragraph_id"])
799def resolve_geometry(
800 dataset: str, text_df: pd.DataFrame, raw_fix_df: pd.DataFrame | None
801) -> tuple[pd.DataFrame, dict]:
802 """Word boxes for ``dataset``, at the best fidelity available.
804 Four tiers, first hit wins: real boxes read straight off ``text_df``
805 (some corpora, verified: PoTeC, carry EyeLink's own measured coordinates
806 there directly -- see `extract_text_df_boxes`); real boxes parsed out of
807 the raw files EyeGenBench downloaded (`extract_eyelink_boxes`); a layout
808 reconstructed from the corpus' published display parameters; a
809 synthesized layout on default defaults. ``text_df`` boxes win over
810 raw-EyeLink-parsed boxes when both are available -- they are complete,
811 need no parsing of ``CURRENT_FIX_INTEREST_AREA_DATA``, and do not depend
812 on the raw download still being on disk. Neither real source is
813 schema-guaranteed (this is data-dependent, not documented by EyeGenBench),
814 so both are detected at runtime and the raw-EyeLink path stays as the
815 fallback for corpora without ``text_df`` boxes.
817 ``text_df`` is one row per ``(paragraph, interest area)`` for corpora that
818 carry boxes this way, each row repeating that paragraph's whole
819 ``ia_list`` -- `dict(zip(text_df["unique_paragraph_id"], text_df["ia_list"]))`
820 below takes the last row per paragraph, which holds the same list either
821 way.
823 The tier is stamped **per paragraph**, not per dataset: a paragraph that
824 contributed at least one real box (from either real source) is ``real``;
825 a paragraph with no real box of its own has no measured geometry and
826 falls back to ``reconstructed`` (a published screen exists) or
827 ``synthesized`` (nothing is known) -- even when other paragraphs in the
828 same dataset are real. Stamping a placeholder paragraph as measured would
829 overstate the dataset's fidelity, which is the one thing this function
830 must never do.
832 ``report["geometry_source"]`` is the tier the dataset as a whole achieved
833 (``real`` if any paragraph is), ``report["display_source"]`` cites which
834 real source won (``"eyegenbench:texts"``) or the published-display
835 provenance otherwise, and ``report["paragraphs_without_real_boxes"]``
836 counts how many paragraphs had to fall back, so the manifest can carry it.
837 """
838 spec = display_spec_for(dataset)
839 fallback_source = (
840 GEOMETRY_RECONSTRUCTED if spec is not DEFAULT_SPEC else GEOMETRY_SYNTHESIZED
841 )
842 ia_lists = dict(zip(text_df["unique_paragraph_id"], text_df["ia_list"]))
843 ia_counts = {pid: len(ia) for pid, ia in ia_lists.items()}
845 text_boxes = extract_text_df_boxes(text_df)
846 if not text_boxes.empty:
847 boxes = text_boxes
848 display_source = "eyegenbench:texts"
849 else:
850 boxes = (
851 extract_eyelink_boxes(raw_fix_df)
852 if raw_fix_df is not None
853 else pd.DataFrame()
854 )
855 display_source = spec.source
857 if not boxes.empty:
858 words, interpolated_fraction = fill_missing_boxes(boxes, ia_counts, spec)
859 paragraphs_with_real_boxes = _paragraphs_with_real_boxes(boxes, ia_counts)
860 per_paragraph_source = {
861 pid: GEOMETRY_REAL if pid in paragraphs_with_real_boxes else fallback_source
862 for pid in ia_counts
863 }
864 words["geometry_source"] = words["unique_paragraph_id"].map(
865 per_paragraph_source
866 )
867 source = GEOMETRY_REAL if paragraphs_with_real_boxes else fallback_source
868 words["line"] = words.groupby("unique_paragraph_id")["start_y"].transform(
869 lambda s: s.rank(method="dense").astype(int) - 1
870 )
871 else:
872 paragraphs_with_real_boxes = set()
873 source = fallback_source
874 interpolated_fraction = 0.0
875 frames = []
876 for paragraph, ia_list in ia_lists.items():
877 laid = layout_words(list(ia_list), spec)
878 laid["unique_paragraph_id"] = paragraph
879 laid = laid.rename(columns={"word_id": "ia_index"})
880 frames.append(laid.drop(columns=["text"]))
881 words = pd.concat(frames, ignore_index=True)
882 words["geometry_source"] = source
884 labels = [
885 {"unique_paragraph_id": pid, "ia_index": i, "ia_label": str(label)}
886 for pid, ia_list in ia_lists.items()
887 for i, label in enumerate(ia_list)
888 ]
889 words = words.merge(
890 pd.DataFrame(labels), on=["unique_paragraph_id", "ia_index"], how="left"
891 )
892 report = {
893 "geometry_source": source,
894 "interpolated_fraction": round(float(interpolated_fraction), 4),
895 "display_source": display_source,
896 "paragraphs_without_real_boxes": len(ia_counts)
897 - len(paragraphs_with_real_boxes),
898 }
899 return words, report
902def place_fixations(fix_df: pd.DataFrame, words: pd.DataFrame) -> pd.DataFrame:
903 """Add ``x``/``y`` to fixations from their interest area's box.
905 Exactly inverts EyeGenBench's landing-position formula
906 (``(fix_x - left) / (right - left)``), so wherever the box is real the
907 round trip is lossless. Fixations with no matching box are dropped -- they
908 cannot be placed, and a wrong placement is worse than a missing one.
910 Some EyeGenBench corpora' ``fix_df`` already carries its own
911 ``start_x``/``start_y``/``end_x``/``end_y`` columns (verified: PoTeC --
912 passthrough source columns, not part of EyeGenBench's documented schema).
913 A plain merge would collide on those names and pandas would suffix both
914 sides (``start_x_x``/``start_x_y``), leaving no column literally named
915 ``start_x`` -- exactly the R24 crash. The box used for placement must
916 always be the **words** frame's box, resolved once and unambiguously by
917 merging the words-side columns in under private names, never whatever
918 ``fix_df`` happened to already carry; ``fix_df``'s own box columns (if
919 any) are left untouched in the output as plain passthrough data.
920 """
921 box_cols = ["start_x", "start_y", "end_x", "end_y"]
922 internal = {c: f"_words_{c}" for c in box_cols}
923 geometry_col = "_words_geometry_source"
924 words_boxes = words[["unique_paragraph_id", "ia_index", *box_cols]].copy()
925 words_boxes[geometry_col] = words.get(
926 "geometry_source", pd.Series(index=words.index, dtype="object")
927 )
928 words_boxes = words_boxes.rename(columns=internal)
929 merged = fix_df.merge(
930 words_boxes, on=["unique_paragraph_id", "ia_index"], how="inner"
931 )
932 if "fix_landing_position" in merged.columns:
933 landing = pd.to_numeric(merged["fix_landing_position"], errors="coerce").fillna(
934 0.5
935 )
936 else:
937 landing = pd.Series(0.5, index=merged.index)
938 landing = landing.clip(0.0, 1.0)
939 box_start_x = merged.pop(internal["start_x"])
940 box_end_x = merged.pop(internal["end_x"])
941 box_start_y = merged.pop(internal["start_y"])
942 box_end_y = merged.pop(internal["end_y"])
943 geometry_source = merged.pop(geometry_col)
944 merged["x"] = box_start_x + landing * (box_end_x - box_start_x)
945 inferred_y = (box_start_y + box_end_y) / 2.0
946 recorded_y = pd.to_numeric(
947 merged.get("recorded_fixation_y", pd.Series(index=merged.index, dtype=float)),
948 errors="coerce",
949 )
950 use_recorded = geometry_source.eq(GEOMETRY_REAL) & recorded_y.notna()
951 merged["y"] = recorded_y.where(use_recorded, inferred_y)
952 merged["fixation_y_source"] = use_recorded.map(
953 {True: "recorded", False: "word-box-center"}
954 )
955 return merged