Coverage for scanpath_studio/eyegenbench_geometry.py: 97%

203 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""Screen geometry for EyeGenBench corpora. 

2 

3EyeGenBench's harmonised output records *which* interest area each fixation 

4landed on and *where within it* (a 0-1 offset), but no pixel coordinates and no 

5word boxes. Scanpath Studio needs boxes. This module recovers them at the best 

6fidelity available -- see `resolve_geometry` for the four tiers. 

7""" 

8 

9from __future__ import annotations 

10 

11from dataclasses import dataclass 

12 

13import pandas as pd 

14 

15 

16@dataclass(frozen=True) 

17class DisplaySpec: 

18 """How one corpus presented its text, in pixels. 

19 

20 ``source`` cites where the numbers came from (e.g. ``"pymovements:potec"``), 

21 so a reconstructed layout can always be traced back to its evidence. 

22 """ 

23 

24 width_px: int 

25 height_px: int 

26 font_px: float 

27 char_width_px: float 

28 line_pitch_px: float 

29 monospaced: bool 

30 margin_px: int 

31 source: str 

32 

33 

34# Provisional -- DATA-27 open decision. A synthesized layout needs *some* 

35# screen; this is the one used when nothing at all is published for a corpus. 

36DEFAULT_SPEC = DisplaySpec( 

37 width_px=1920, 

38 height_px=1080, 

39 font_px=20, 

40 char_width_px=12.0, # Courier-family advance width at 20px. 

41 line_pitch_px=60.0, # Double-spaced, the reading-study norm. 

42 monospaced=True, 

43 margin_px=100, 

44 source="default", 

45) 

46 

47 

48def layout_words(ia_list: list[str], spec: DisplaySpec) -> pd.DataFrame: 

49 """Lay ``ia_list`` out as word boxes on ``spec``'s screen. 

50 

51 Greedy left-to-right wrapping with one space between words. Shared by the 

52 reconstructed and synthesized tiers -- same code, different ``spec``. 

53 """ 

54 if not len(ia_list): 

55 raise ValueError("layout_words needs at least one interest area") 

56 

57 usable = spec.width_px - 2 * spec.margin_px 

58 space = spec.char_width_px 

59 rows = [] 

60 x, line = spec.margin_px, 0 

61 for word_id, text in enumerate(ia_list): 

62 width = max(len(str(text)), 1) * spec.char_width_px 

63 if x > spec.margin_px and (x + width) > (spec.margin_px + usable): 

64 x, line = spec.margin_px, line + 1 

65 top = spec.margin_px + line * spec.line_pitch_px 

66 rows.append( 

67 { 

68 "word_id": word_id, 

69 "text": str(text), 

70 "line": line, 

71 "start_x": x, 

72 "end_x": x + width, 

73 "start_y": top, 

74 "end_y": top + spec.font_px, 

75 } 

76 ) 

77 x += width + space 

78 return pd.DataFrame(rows) 

79 

80 

81def _spec( 

82 width_px, 

83 height_px, 

84 font_px, 

85 *, 

86 source, 

87 mono=True, 

88 chars_per_deg=None, 

89 double_spaced=False, 

90 distance_cm=None, 

91 width_cm=None, 

92 margin_px=100, 

93) -> DisplaySpec: 

94 """Build a DisplaySpec from what a corpus actually reported. 

95 

96 ``chars_per_deg`` + ``distance_cm`` + ``width_cm`` gives a measured 

97 character width; otherwise fall back to the monospace advance ratio. 

98 """ 

99 if chars_per_deg and distance_cm and width_cm: 

100 px_per_cm = width_px / width_cm 

101 cm_per_deg = 2 * distance_cm * 0.00872686779 # tan(0.5 deg) 

102 char_width = (cm_per_deg * px_per_cm) / chars_per_deg 

103 else: 

104 char_width = font_px * 0.6 

105 return DisplaySpec( 

106 width_px=width_px, 

107 height_px=height_px, 

108 font_px=font_px, 

109 char_width_px=char_width, 

110 line_pitch_px=font_px * (2.0 if double_spaced else 1.5), 

111 monospaced=mono, 

112 margin_px=margin_px, 

113 source=source, 

114 ) 

115 

116 

117# Published display parameters, best source first: pymovements dataset YAMLs -> 

118# the UZH dataset review table -> the corpus' own paper. Anything not listed 

119# here falls back to DEFAULT_SPEC and is stamped `synthesized`. 

120# 

121# R49/R50 -- the entry bar is a PUBLISHED parameter, not a plausible one. An 

122# entry here upgrades a corpus from `synthesized` to `reconstructed`, i.e. from 

123# "this geometry is invented" to "this geometry follows the corpus' own 

124# reported screen"; a plausible-but-unsourced entry makes an invented layout 

125# indistinguishable from a sourced one, which is the single thing the tier 

126# system exists to prevent. So: no modal/mean screen for a multi-site corpus 

127# (each site's readers would be laid out on another site's screen), no 

128# borrowing a sibling corpus' apparatus, no filling a missing pixel resolution 

129# from the hardware of the era. Where a value comes from somewhere other than 

130# the corpus' own paper (e.g. a monitor datasheet for a model the paper names), 

131# `source` says so rather than presenting it as a paper value. 

132DISPLAY_SPECS: dict[str, DisplaySpec] = { 

133 "potec": _spec( 

134 1680, 1050, 20, source="pymovements:potec", width_cm=47.5, distance_cm=65 

135 ), 

136 "copco": _spec( 

137 1920, 

138 1080, 

139 14, 

140 source="pymovements:copco + paper:Hollenstein2022", 

141 double_spaced=True, 

142 width_cm=59.0, 

143 distance_cm=85, 

144 ), 

145 "emtec": _spec( 

146 1280, 

147 1024, 

148 14, 

149 source="pymovements:emtec + uzh", 

150 chars_per_deg=2.86, 

151 width_cm=38.2, 

152 distance_cm=60, 

153 ), 

154 "colagaze": _spec( 

155 1280, 

156 1024, 

157 17, 

158 source="pymovements:colagaze + uzh", 

159 chars_per_deg=2.0, 

160 width_cm=54.37, 

161 distance_cm=60, 

162 ), 

163 "interead": _spec( 

164 1920, 

165 1080, 

166 16, 

167 source="pymovements:interead + uzh", 

168 width_cm=52.8, 

169 distance_cm=57, 

170 ), 

171 "ggtg": _spec( 

172 1100, 

173 900, 

174 20, 

175 source="pymovements:ggtg + uzh", 

176 mono=False, 

177 double_spaced=True, 

178 width_cm=31.2, 

179 distance_cm=66, 

180 ), 

181 "etdd70": _spec( 

182 1680, 

183 1050, 

184 20, 

185 source="pymovements:etdd70 + uzh", 

186 mono=False, 

187 distance_cm=65, 

188 ), 

189 "gaze4hate": _spec( 

190 2560, 

191 1440, 

192 20, 

193 source="pymovements:gaze4hate", 

194 width_cm=59.8, 

195 distance_cm=78.0, 

196 ), 

197 "raccoons": _spec( 

198 1920, 

199 1080, 

200 20, 

201 source="pymovements:raccoons", 

202 width_cm=56.8, 

203 distance_cm=105.5, 

204 ), 

205 "sbsat": _spec( 

206 1024, 

207 768, 

208 20, 

209 source="pymovements:sb_sat", 

210 width_cm=44.5, 

211 distance_cm=70, 

212 ), 

213 "provo": _spec( 

214 1600, 

215 900, 

216 20, 

217 source="paper:LukeChristianson2018", 

218 chars_per_deg=3.0, 

219 width_cm=40.0, 

220 distance_cm=60, 

221 ), 

222 "psr": _spec( 

223 1024, 

224 768, 

225 18, 

226 source="uzh", 

227 chars_per_deg=2.38, 

228 distance_cm=73, 

229 width_cm=47.0, 

230 ), 

231 "eyevoicespan": _spec( 

232 1280, 

233 960, 

234 24, 

235 source="uzh", 

236 chars_per_deg=2.22, 

237 distance_cm=60, 

238 width_cm=40.0, 

239 ), 

240 "iitbhgc": _spec( 

241 1920, 

242 1080, 

243 20, 

244 source="uzh", 

245 mono=False, 

246 distance_cm=70, 

247 ), 

248 "bsc": _spec( 

249 1024, 

250 768, 

251 20, 

252 source="uzh", 

253 chars_per_deg=0.75, 

254 distance_cm=43, 

255 width_cm=36.0, 

256 ), 

257 "chinesereading": _spec(1024, 768, 20, source="uzh", distance_cm=58), 

258 "cuentos": _spec(1920, 1080, 24, source="uzh", distance_cm=55), 

259 "zuco1": _spec(1920, 1080, 20, source="paper:Hollenstein2018", mono=False), 

260 "zuco2": _spec(1920, 1080, 20, source="paper:Hollenstein2020", mono=False), 

261 # R49 -- OneStop publishes more of its layout than any other corpus here: 

262 # monitor, display area, both viewing distances, the letter cell in px and 

263 # the line pitch in px (Berzak et al. 2025, Sci Data 12:1995, 

264 # doi:10.1038/s41597-025-06272-2, Methods -> Apparatus). Two judgement 

265 # calls, both choices *between* published numbers rather than inventions: 

266 # 

267 # * `distance_cm=75.0`. The paper gives eye level as 750 mm from the top 

268 # of the display area and 795 mm from its bottom, so a scalar has to 

269 # pick an end. 75.0 is the right one twice over: the text starts at 

270 # y=186 px, near the top, and it is the only end at which the paper's 

271 # other two published figures agree -- 0.34 deg per letter at 75 cm 

272 # reproduces the published 19 px letter cell (19.1 px), where 79.5 cm 

273 # would give 20.2 px, 6% wide. 

274 # * `margin_px=368` reproduces the published *text column* (2560 - 2*368 = 

275 # 1824 px = 96 characters) rather than the published *left edge* (300 

276 # px). `_spec`'s margin model is symmetric and OneStop's layout is not 

277 # (left 300, column 1824, right 436), so only one of the two survives. 

278 # Column width is what `layout_words` actually consumes -- it decides 

279 # where lines wrap -- so matching it puts every word on the right line, 

280 # at the cost of shifting the whole block 68 px right. 

281 # 

282 # `double_spaced=True` is not an assumption here: 38 x 2 = 76 px is exactly 

283 # the "triple spacing (76 px)" the paper reports. This entry is also the 

284 # single sourced home for the 2560x1440 that app.py / cli.py / datasets.py 

285 # pin for the *native* OneStop corpus; those sites now cite it. 

286 "onestop": _spec( 

287 2560, 

288 1440, 

289 38, # published letter cell 19 px x 38 px (25 pt at this screen's 109 ppi) 

290 source=( 

291 "paper:Berzak2025-OneStop" 

292 " (distance_cm=75, the published eye-to-top-of-display;" 

293 " margin fits the published 1824 px text column)" 

294 ), 

295 chars_per_deg=2.94, # 1 / 0.34 deg per letter, stated 

296 distance_cm=75.0, 

297 width_cm=59.7, # 597 mm display area, stated 

298 double_spaced=True, 

299 margin_px=368, 

300 ), 

301 # R49 -- Laurinavichyute et al. 2019, Behav Res Methods 51(3):1161-1178, 

302 # doi:10.3758/s13428-018-1051-6, Method -> Procedure states the monitor and 

303 # resolution, 22 pt Courier New, 90 cm, and 0.29 deg per character. 

304 # `font_px=28` converts the published 22 pt at this screen's 92 ppi -- it is 

305 # NOT the point size copied into a pixel field. Cross-check: 22 pt Courier 

306 # New has a 16.8 px advance at 92 ppi and the published 0.29 deg/char at 

307 # 90 cm gives 16.5 px, two independent published routes agreeing to 2%, 

308 # which is also what justifies pairing them with the datasheet width. 

309 # Single-sentence trials, so nothing about line spacing is published (there 

310 # are no line breaks) and `line_pitch_px` only sets box height. 

311 "rsc": _spec( 

312 1920, 

313 1080, 

314 28, # 22 pt at 1920 px / 53.1 cm = 92 ppi 

315 source=( 

316 "paper:Laurinavichyute2019-RSC" 

317 " (width_cm from the ASUS VG248QE datasheet, the model the paper names)" 

318 ), 

319 chars_per_deg=3.45, # 1 / 0.29 deg per character, stated 

320 distance_cm=90, 

321 width_cm=53.1, 

322 ), 

323 # R49 -- Yan, Pan & Kliegl 2025, Behav Res Methods 57(2):60, 

324 # doi:10.3758/s13428-024-02523-z, Apparatus. BSC-II's apparatus does NOT 

325 # carry over from `bsc` above and the asymmetry between the two entries is 

326 # real, not a gap: BSC was a 19" CRT at 1024x768 / 43 cm / 0.75 chars per 

327 # degree, BSC-II a 24.5" LCD at 1920x1080 / 70 cm / 0.909. Inheriting BSC's 

328 # numbers would have been wrong by ~2x in px per degree. 

329 # A Song-font CJK glyph cell is square, so the published 1.1 deg per 

330 # character IS the font size (~48 px); the harmonised interest areas are 

331 # words, and `layout_words` multiplies per-character, so the per-character 

332 # width is the correct thing to hand it. 

333 "bscii": _spec( 

334 1920, 

335 1080, 

336 48, # square CJK cell = the published character width 

337 source=( 

338 "paper:YanPanKliegl2025-BSCII" 

339 " (width_cm from the BenQ ZOWIE XL2546K datasheet, the model the" 

340 " paper names)" 

341 ), 

342 chars_per_deg=0.909, # 1 / 1.1 deg per character, stated 

343 distance_cm=70, 

344 width_cm=53.8, 

345 ), 

346 # R50 -- `mecol2w2` USED to be here as 

347 # `_spec(1920, 1080, 21, source="uzh + paper:MECO-L2-W2")`. It was removed 

348 # after both of its cited sources were re-read, and neither contains it: 

349 # 

350 # * Kuperman et al. 2025 (SSLA, doi:10.1017/S0272263125000105), Procedure: 

351 # "A mono-spaced font (Consolas) was used, with a size generally ranging 

352 # from 20 to 22 points (given variation in screen size and resolution at 

353 # different testing sites) and 1.5 line spacing. In accordance with 

354 # their local experimental setup, the German site in Zurich used a 

355 # smaller font size of 10 with a lower resolution of 1280 x 1024. ... For 

356 # further specifications of the screen, font size, presentation settings, 

357 # and apparatus at each participating site, see supplementary material 

358 # S2." The string "1920" does not occur in the paper at all; the only 

359 # resolution it prints is one site's exception. 

360 # * The UZH review row for "MECO L2 2nd Wave" has `Resolution: null` and 

361 # puts a LINK to those same supplementary materials in its `Monitor` and 

362 # eye-to-screen-distance cells -- the curators' way of writing "per 

363 # site". 

364 # 

365 # So 1920x1080 was corpus-level for no site in particular, and `font_px=21` 

366 # was the midpoint of a 20-22 **pt** range used as **px** -- a unit error 

367 # on its own terms, and one that ignored both the range and the 10 pt site. 

368 # Falling back to `synthesized` is a user-visible downgrade and the correct 

369 # direction: an unsourced `reconstructed` is worse than an honest 

370 # `synthesized`, because the user cannot tell it is a guess. 

371 # 

372 # Decided the same way for both MECO L2 waves, which share sites, protocol 

373 # and materials -- `mecol2w1` gets no entry either, and neither do 

374 # `mecol1w1` / `mecol1w2`. Wave 1 of L1 states outright that "maintaining 

375 # an identical font size, distance from the screen, and screen resolution 

376 # was unfeasible"; Wave 2 of L1 tabulates 16 different screens (Table 10, 

377 # Siegelman et al. 2025). These corpora have no corpus-level screen to 

378 # publish. Per-lab geometry would need a per-site table, not a per-corpus 

379 # one; the site code is recoverable from the data if that is ever built. 

380 # 

381 # `psc2` is deliberately absent too, and it is the easiest one to "fix" 

382 # wrongly. A complete apparatus IS published for the PSC2 sentences 

383 # (1280x960, Courier New 24 pt, 14 px/letter, 60 cm; Laubrock & Kliegl 

384 # 2015, Front Psychol 6:1432) -- but for the 32-reader ORAL reading 

385 # experiment, which is already in this table above as `eyevoicespan`. The 

386 # corpus shipped as `psc2` is a 149-reader SILENT reading collection whose 

387 # only cited document (Heister, Wuerzner & Kliegl 2012) has no methods 

388 # section at all. Same lab, same materials, probably the same room -- which 

389 # is exactly why copying it across would be invisibly wrong: the two 

390 # corpora would become indistinguishable here, and nothing in the data 

391 # would reveal the error. Pinned by 

392 # `tests/test_eyegenbench_geometry.py::test_psc2_stays_synthesized_...`. 

393} 

394 

395 

396def display_spec_for(dataset: str) -> DisplaySpec: 

397 """The published screen for ``dataset``, or ``DEFAULT_SPEC`` if none is known.""" 

398 return DISPLAY_SPECS.get(str(dataset).lower(), DEFAULT_SPEC) 

399 

400 

401IA_DATA_COLUMN = "CURRENT_FIX_INTEREST_AREA_DATA" 

402_BOX_COLUMNS = ["start_x", "start_y", "end_x", "end_y"] 

403 

404 

405def parse_ia_data(series: pd.Series) -> pd.DataFrame: 

406 """EyeLink ``[STATIC, RECTANGLE, left, top, right, bottom]`` -> box columns. 

407 

408 This is the corpus' *real* on-screen word box. EyeGenBench reads it, divides 

409 it into a normalised landing position, and discards it; we keep it. 

410 """ 

411 parts = series.astype(str).str.strip("[]").str.split(",", expand=True) 

412 if parts.shape[1] < 6: 

413 return pd.DataFrame( 

414 {c: [float("nan")] * len(series) for c in _BOX_COLUMNS}, index=series.index 

415 ) 

416 out = pd.DataFrame(index=series.index) 

417 for name, idx in zip(_BOX_COLUMNS, (2, 3, 4, 5)): 

418 out[name] = pd.to_numeric(parts[idx].str.strip(), errors="coerce") 

419 return out 

420 

421 

422def extract_eyelink_boxes( 

423 frame: pd.DataFrame, 

424 *, 

425 paragraph_col: str = "unique_paragraph_id", 

426 ia_col: str = "ia_index", 

427 data_col: str = IA_DATA_COLUMN, 

428) -> pd.DataFrame: 

429 """One real box per ``(paragraph, interest area)`` found in ``frame``. 

430 

431 Only interest areas that were *fixated* appear -- the column rides on 

432 fixations. `fill_missing_boxes` completes the rest. 

433 

434 R28: screen coordinates are non-negative by definition, so a box whose 

435 ``start_x``/``start_y`` is negative is malformed data, not a real 

436 on-screen position -- it is dropped here rather than earning the `real` 

437 stamp. This is the same discipline `extract_text_df_boxes` applies to its 

438 own real-box source; both feed the same stamp, so both must guarantee it. 

439 

440 R33: the raw-EyeLink tier is an optional upgrade (tier 1 of 4). All three 

441 columns this function indexes -- ``paragraph_col``, ``ia_col`` and 

442 ``data_col`` -- must be present before any of them is read; when one is 

443 missing this returns the empty, correctly-columned frame so 

444 `resolve_geometry` falls through to the next tier instead of raising. 

445 Verified against a real corpus: onestop's raw EyeLink export carries 

446 ``CURRENT_FIX_INTEREST_AREA_DATA`` but neither ``unique_paragraph_id`` nor 

447 ``ia_index`` (an un-prefixed ``paragraph_id`` and no interest-area index 

448 at all) -- a hard skip here used to drop the entire corpus from the 

449 catalogue behind a one-line log entry, when the bundle the reconstructed/ 

450 synthesized tiers would have produced is perfectly good. 

451 """ 

452 if not {paragraph_col, ia_col, data_col} <= set(frame.columns): 

453 return pd.DataFrame(columns=[paragraph_col, ia_col, *_BOX_COLUMNS]) 

454 boxes = pd.concat( 

455 [ 

456 frame[[paragraph_col, ia_col]].reset_index(drop=True), 

457 parse_ia_data(frame[data_col]).reset_index(drop=True), 

458 ], 

459 axis=1, 

460 ) 

461 boxes = boxes.dropna(subset=_BOX_COLUMNS) 

462 boxes = boxes[(boxes["start_x"] >= 0) & (boxes["start_y"] >= 0)] 

463 boxes = boxes.drop_duplicates(subset=[paragraph_col, ia_col], keep="first") 

464 return boxes.sort_values([paragraph_col, ia_col]).reset_index(drop=True) 

465 

466 

467_TEXT_DF_BOX_COLUMNS = ["start_x", "start_y", "end_x", "end_y"] 

468 

469 

470def extract_text_df_boxes( 

471 text_df: pd.DataFrame, 

472 *, 

473 paragraph_col: str = "unique_paragraph_id", 

474 ia_col: str = "ia_index", 

475) -> pd.DataFrame: 

476 """One real box per ``(paragraph, interest area)`` straight off ``text_df``. 

477 

478 Some EyeGenBench corpora (verified: PoTeC) carry EyeLink's own genuinely 

479 measured on-screen coordinates directly on the harmonised ``text_df`` -- 

480 one row per ``(paragraph, interest area)``, each repeating that 

481 paragraph's whole ``ia_list``. This is **not** part of EyeGenBench's 

482 documented schema (``start_x`` appears in no ``.py``/``.yaml`` in the 

483 EyeGenBench source -- it survives only as source-column pass-through), so 

484 it is detected here at runtime, per dataset, and never assumed to exist. 

485 

486 Presence is not trust: a row only contributes a box when its ``ia_index`` 

487 is a valid finite non-negative whole number (`_valid_ia_index`, the same 

488 discipline `extract_eyelink_boxes` uses) *and* the box itself is finite, 

489 non-negative (``start_x >= 0`` and ``start_y >= 0`` -- R28: a screen 

490 coordinate is non-negative by definition, and since ``start < end`` is 

491 already required, a non-negative start also bounds ``end`` -- a box 

492 straddling the origin, e.g. ``start_x=-5, end_x=40``, is rejected too, not 

493 representable on a 0-origin canvas), with ``start_x < end_x`` and 

494 ``start_y < end_y``. A row failing any check contributes nothing -- never 

495 a guessed or truncated position, and never a `real` stamp for coordinates 

496 that might turn out to be some other space in a corpus not yet verified. 

497 """ 

498 if not all(c in text_df.columns for c in [ia_col, *_TEXT_DF_BOX_COLUMNS]): 

499 # The four box columns alone don't guarantee ia_index is there too -- 

500 # the brief's own TEXTS fixture is exactly that shape (no ia_index). 

501 # Without this check that row indexing below raised KeyError, which 

502 # dropped the whole dataset from the manifest instead of falling 

503 # through to the reconstructed/synthesized tiers as it must. 

504 return pd.DataFrame(columns=[paragraph_col, ia_col, *_TEXT_DF_BOX_COLUMNS]) 

505 

506 ia_index = _valid_ia_index(text_df[ia_col]) 

507 numeric_box = text_df[_TEXT_DF_BOX_COLUMNS].apply(pd.to_numeric, errors="coerce") 

508 is_finite = numeric_box.apply( 

509 lambda s: (s > float("-inf")) & (s < float("inf")) 

510 ).all(axis=1) 

511 valid = ( 

512 ia_index.notna() 

513 & is_finite 

514 & (numeric_box["start_x"] >= 0) 

515 & (numeric_box["start_y"] >= 0) 

516 & (numeric_box["start_x"] < numeric_box["end_x"]) 

517 & (numeric_box["start_y"] < numeric_box["end_y"]) 

518 ) 

519 

520 out = pd.DataFrame( 

521 { 

522 paragraph_col: text_df[paragraph_col], 

523 ia_col: ia_index, 

524 **{c: numeric_box[c] for c in _TEXT_DF_BOX_COLUMNS}, 

525 } 

526 ) 

527 out = out.loc[valid] 

528 out = out.drop_duplicates(subset=[paragraph_col, ia_col], keep="first") 

529 return out.sort_values([paragraph_col, ia_col]).reset_index(drop=True) 

530 

531 

532def _valid_ia_index(series: pd.Series) -> pd.Series: 

533 """``ia_index`` as validated numbers, ``NaN`` for anything invalid. 

534 

535 An interest-area index is a finite whole number ``>= 0`` -- nothing 

536 fractional, negative, infinite or NaN is ever a valid EyeLink/EyeGenBench 

537 index, whatever `pd.to_numeric` manages to parse out of a row. Parses 

538 first (handling "." and NaN, EyeGenBench's own documented "landed on no 

539 interest area" sentinels, and string-typed columns), then keeps only 

540 finite, non-negative whole numbers: a fractional value like ``0.9`` or 

541 ``-0.5`` is exactly as invalid as ``"abc"``, and ``inf`` is exactly as 

542 invalid as either -- never truncated, overflowed, or guessed at. 

543 

544 Deliberately returned as plain (float) numbers, not cast to a 

545 fixed-width integer type -- nothing here bounds the magnitude of an 

546 otherwise-valid-shaped index; an oversized one (say, larger than any 

547 real paragraph could have) is rejected the same way a merely 

548 past-the-end one already is, by the existing ``< count`` checks in 

549 `fill_missing_boxes` and `_paragraphs_with_real_boxes`. A fixed-width 

550 cast over the whole column would instead raise the moment one row 

551 exceeded its range, poisoning every genuine box beside it rather than 

552 rejecting the one bad row. Only a value already validated here is ever 

553 handed to Python's own arbitrary-precision ``int()`` (in 

554 `fill_missing_boxes`), where it can no longer overflow anything. 

555 

556 Shared by `fill_missing_boxes` and `_paragraphs_with_real_boxes` so the 

557 two accept and reject exactly the same set of rows, by construction 

558 rather than by convention -- neither one raises on a row this returns 

559 ``NaN`` for, and neither one guesses a position for it. 

560 """ 

561 numeric = pd.to_numeric(series, errors="coerce") 

562 is_finite = (numeric > float("-inf")) & (numeric < float("inf")) 

563 is_valid_index = is_finite & (numeric >= 0) & (numeric == numeric.round()) 

564 return numeric.where(is_valid_index) 

565 

566 

567def fill_missing_boxes( 

568 boxes: pd.DataFrame, ia_counts: dict[str, int], spec: DisplaySpec 

569) -> tuple[pd.DataFrame, float]: 

570 """Complete ``boxes`` so every interest area of every paragraph has one. 

571 

572 Real boxes only cover *fixated* areas. A row whose ``ia_index`` isn't a 

573 valid finite non-negative whole number (`_valid_ia_index`; EyeLink's own 

574 sentinel for "landed on no interest area" is ``.`` or ``NaN``, and a 

575 fractional, negative, or infinite value is exactly as invalid) is 

576 treated exactly the same as a genuinely unfixated index -- simply absent 

577 from the lookup, never a raised error and never a guessed or truncated 

578 position. Because `_valid_ia_index` already excludes anything not 

579 finite, converting a *validated* value with plain ``int()`` below can 

580 never overflow, however large -- it is Python's own arbitrary-precision 

581 conversion, not a fixed-width one. Consecutive gaps (including those) 

582 are filled as runs (maximal consecutive stretches of missing indices): 

583 

584 - Bracketed on both sides, same line: divide the space evenly between them 

585 - Bracketed across a line break: continue rightward from L on L's line 

586 - Trailing (L exists, no R): advance rightward from L 

587 - Leading (no L, R exists): the same bracketed formula, treating 

588 ``margin_px`` as a virtual left anchor -- divide the space between it and 

589 R evenly (zero-width boxes at ``margin_px`` when there is no room, i.e. 

590 ``available <= 0``) 

591 - No anchor (neither L nor R): distribute evenly across the screen 

592 

593 Returns ``(filled, interpolated_fraction)`` so the manifest can report 

594 how much of the geometry is inferred rather than measured. 

595 """ 

596 width = spec.char_width_px * 4 

597 out = [] 

598 n_filled = 0 

599 n_total = 0 

600 

601 for paragraph, count in ia_counts.items(): 

602 present = boxes[boxes["unique_paragraph_id"] == paragraph] 

603 valid_index = _valid_ia_index(present["ia_index"]) 

604 by_index = { 

605 int(ia): row 

606 for ia, row in zip(valid_index, present.itertuples()) 

607 if pd.notna(ia) 

608 } 

609 n_total += count 

610 

611 # Process all indices, identifying runs of consecutive missing indices 

612 idx = 0 

613 while idx < count: 

614 if idx in by_index: 

615 # Real box: add it as-is 

616 row = by_index[idx] 

617 out.append( 

618 { 

619 "unique_paragraph_id": paragraph, 

620 "ia_index": idx, 

621 "start_x": row.start_x, 

622 "start_y": row.start_y, 

623 "end_x": row.end_x, 

624 "end_y": row.end_y, 

625 } 

626 ) 

627 idx += 1 

628 else: 

629 # Start of a run of consecutive missing indices 

630 run_start = idx 

631 while idx < count and idx not in by_index: 

632 idx += 1 

633 run_end = idx - 1 

634 k = run_end - run_start + 1 

635 

636 # Find L (nearest real box before the run) 

637 L = None 

638 for i in range(run_start - 1, -1, -1): 

639 if i in by_index: 

640 L = by_index[i] 

641 break 

642 

643 # Find R (nearest real box after the run) 

644 R = None 

645 for i in range(run_end + 1, count): 

646 if i in by_index: 

647 R = by_index[i] 

648 break 

649 

650 # Fill the run based on anchor availability and line position 

651 if L is not None and R is not None: 

652 if L.start_y == R.start_y: 

653 # Bracketed on both sides, same line 

654 span = R.start_x - L.end_x 

655 if span > 0: 

656 slot = span / k 

657 for i in range(k): 

658 out.append( 

659 { 

660 "unique_paragraph_id": paragraph, 

661 "ia_index": run_start + i, 

662 "start_x": L.end_x + i * slot, 

663 "start_y": L.start_y, 

664 "end_x": L.end_x + (i + 1) * slot, 

665 "end_y": L.end_y, 

666 } 

667 ) 

668 else: 

669 # Zero or negative span, use zero-width boxes at L.end_x 

670 for i in range(k): 

671 out.append( 

672 { 

673 "unique_paragraph_id": paragraph, 

674 "ia_index": run_start + i, 

675 "start_x": L.end_x, 

676 "start_y": L.start_y, 

677 "end_x": L.end_x, 

678 "end_y": L.end_y, 

679 } 

680 ) 

681 n_filled += k 

682 else: 

683 # Bracketed but across line break, treat as trailing 

684 for i in range(k): 

685 out.append( 

686 { 

687 "unique_paragraph_id": paragraph, 

688 "ia_index": run_start + i, 

689 "start_x": L.end_x 

690 + i * (width + spec.char_width_px) 

691 + spec.char_width_px, 

692 "start_y": L.start_y, 

693 "end_x": L.end_x 

694 + i * (width + spec.char_width_px) 

695 + spec.char_width_px 

696 + width, 

697 "end_y": L.end_y, 

698 } 

699 ) 

700 n_filled += k 

701 elif L is not None: 

702 # Trailing run: advance rightward from L 

703 for i in range(k): 

704 out.append( 

705 { 

706 "unique_paragraph_id": paragraph, 

707 "ia_index": run_start + i, 

708 "start_x": L.end_x 

709 + i * (width + spec.char_width_px) 

710 + spec.char_width_px, 

711 "start_y": L.start_y, 

712 "end_x": L.end_x 

713 + i * (width + spec.char_width_px) 

714 + spec.char_width_px 

715 + width, 

716 "end_y": L.end_y, 

717 } 

718 ) 

719 n_filled += k 

720 elif R is not None: 

721 # Leading run: divide space from margin_px to R using bracketed formula 

722 available = R.start_x - spec.margin_px 

723 if available > 0: 

724 slot = available / k 

725 for i in range(k): 

726 out.append( 

727 { 

728 "unique_paragraph_id": paragraph, 

729 "ia_index": run_start + i, 

730 "start_x": spec.margin_px + i * slot, 

731 "start_y": R.start_y, 

732 "end_x": spec.margin_px + (i + 1) * slot, 

733 "end_y": R.end_y, 

734 } 

735 ) 

736 else: 

737 # No room before R, use zero-width boxes at margin_px 

738 for i in range(k): 

739 out.append( 

740 { 

741 "unique_paragraph_id": paragraph, 

742 "ia_index": run_start + i, 

743 "start_x": spec.margin_px, 

744 "start_y": R.start_y, 

745 "end_x": spec.margin_px, 

746 "end_y": R.end_y, 

747 } 

748 ) 

749 n_filled += k 

750 else: 

751 # No anchor: distribute evenly across usable screen width 

752 usable_width = spec.width_px - 2 * spec.margin_px 

753 slot = usable_width / k if k > 0 else 0 

754 for i in range(k): 

755 out.append( 

756 { 

757 "unique_paragraph_id": paragraph, 

758 "ia_index": run_start + i, 

759 "start_x": spec.margin_px + i * slot, 

760 "start_y": spec.margin_px, 

761 "end_x": spec.margin_px + (i + 1) * slot, 

762 "end_y": spec.margin_px + spec.font_px, 

763 } 

764 ) 

765 n_filled += k 

766 

767 fraction = (n_filled / n_total) if n_total else 0.0 

768 return pd.DataFrame(out), fraction 

769 

770 

771GEOMETRY_REAL = "real" 

772GEOMETRY_RECONSTRUCTED = "reconstructed" 

773GEOMETRY_SYNTHESIZED = "synthesized" 

774 

775 

776def _paragraphs_with_real_boxes(boxes: pd.DataFrame, ia_counts: dict) -> set: 

777 """Paragraph ids that contributed at least one real, in-range box. 

778 

779 A raw ``ia_index`` only "counts" if `fill_missing_boxes` could actually 

780 place it -- it only ever considers indices ``0..count-1`` for a paragraph, 

781 so an index past the end (a harmonised-text/raw-export mismatch, however 

782 large) is silently never used, checked here as the remaining upper bound. 

783 Every other way an index can be unusable -- negative (EyeLink writes 

784 ``-1`` for a fixation that landed on no interest area at all, an expected 

785 raw-export shape, not a corner case), fractional, non-finite, or 

786 unparseable (EyeGenBench's own documented convention for that same event 

787 is ``.`` or ``NaN``) -- is already rejected by `_valid_ia_index` itself, 

788 the same helper `fill_missing_boxes` uses to build its lookup. The two 

789 functions accept and reject exactly the same set of rows by construction, 

790 so a row neither of them can use is never counted as a contribution and 

791 never raises. 

792 """ 

793 ia_index = _valid_ia_index(boxes["ia_index"]) 

794 counts = boxes["unique_paragraph_id"].map(ia_counts) 

795 in_range = ia_index < counts 

796 return set(boxes.loc[in_range, "unique_paragraph_id"]) 

797 

798 

799def resolve_geometry( 

800 dataset: str, text_df: pd.DataFrame, raw_fix_df: pd.DataFrame | None 

801) -> tuple[pd.DataFrame, dict]: 

802 """Word boxes for ``dataset``, at the best fidelity available. 

803 

804 Four tiers, first hit wins: real boxes read straight off ``text_df`` 

805 (some corpora, verified: PoTeC, carry EyeLink's own measured coordinates 

806 there directly -- see `extract_text_df_boxes`); real boxes parsed out of 

807 the raw files EyeGenBench downloaded (`extract_eyelink_boxes`); a layout 

808 reconstructed from the corpus' published display parameters; a 

809 synthesized layout on default defaults. ``text_df`` boxes win over 

810 raw-EyeLink-parsed boxes when both are available -- they are complete, 

811 need no parsing of ``CURRENT_FIX_INTEREST_AREA_DATA``, and do not depend 

812 on the raw download still being on disk. Neither real source is 

813 schema-guaranteed (this is data-dependent, not documented by EyeGenBench), 

814 so both are detected at runtime and the raw-EyeLink path stays as the 

815 fallback for corpora without ``text_df`` boxes. 

816 

817 ``text_df`` is one row per ``(paragraph, interest area)`` for corpora that 

818 carry boxes this way, each row repeating that paragraph's whole 

819 ``ia_list`` -- `dict(zip(text_df["unique_paragraph_id"], text_df["ia_list"]))` 

820 below takes the last row per paragraph, which holds the same list either 

821 way. 

822 

823 The tier is stamped **per paragraph**, not per dataset: a paragraph that 

824 contributed at least one real box (from either real source) is ``real``; 

825 a paragraph with no real box of its own has no measured geometry and 

826 falls back to ``reconstructed`` (a published screen exists) or 

827 ``synthesized`` (nothing is known) -- even when other paragraphs in the 

828 same dataset are real. Stamping a placeholder paragraph as measured would 

829 overstate the dataset's fidelity, which is the one thing this function 

830 must never do. 

831 

832 ``report["geometry_source"]`` is the tier the dataset as a whole achieved 

833 (``real`` if any paragraph is), ``report["display_source"]`` cites which 

834 real source won (``"eyegenbench:texts"``) or the published-display 

835 provenance otherwise, and ``report["paragraphs_without_real_boxes"]`` 

836 counts how many paragraphs had to fall back, so the manifest can carry it. 

837 """ 

838 spec = display_spec_for(dataset) 

839 fallback_source = ( 

840 GEOMETRY_RECONSTRUCTED if spec is not DEFAULT_SPEC else GEOMETRY_SYNTHESIZED 

841 ) 

842 ia_lists = dict(zip(text_df["unique_paragraph_id"], text_df["ia_list"])) 

843 ia_counts = {pid: len(ia) for pid, ia in ia_lists.items()} 

844 

845 text_boxes = extract_text_df_boxes(text_df) 

846 if not text_boxes.empty: 

847 boxes = text_boxes 

848 display_source = "eyegenbench:texts" 

849 else: 

850 boxes = ( 

851 extract_eyelink_boxes(raw_fix_df) 

852 if raw_fix_df is not None 

853 else pd.DataFrame() 

854 ) 

855 display_source = spec.source 

856 

857 if not boxes.empty: 

858 words, interpolated_fraction = fill_missing_boxes(boxes, ia_counts, spec) 

859 paragraphs_with_real_boxes = _paragraphs_with_real_boxes(boxes, ia_counts) 

860 per_paragraph_source = { 

861 pid: GEOMETRY_REAL if pid in paragraphs_with_real_boxes else fallback_source 

862 for pid in ia_counts 

863 } 

864 words["geometry_source"] = words["unique_paragraph_id"].map( 

865 per_paragraph_source 

866 ) 

867 source = GEOMETRY_REAL if paragraphs_with_real_boxes else fallback_source 

868 words["line"] = words.groupby("unique_paragraph_id")["start_y"].transform( 

869 lambda s: s.rank(method="dense").astype(int) - 1 

870 ) 

871 else: 

872 paragraphs_with_real_boxes = set() 

873 source = fallback_source 

874 interpolated_fraction = 0.0 

875 frames = [] 

876 for paragraph, ia_list in ia_lists.items(): 

877 laid = layout_words(list(ia_list), spec) 

878 laid["unique_paragraph_id"] = paragraph 

879 laid = laid.rename(columns={"word_id": "ia_index"}) 

880 frames.append(laid.drop(columns=["text"])) 

881 words = pd.concat(frames, ignore_index=True) 

882 words["geometry_source"] = source 

883 

884 labels = [ 

885 {"unique_paragraph_id": pid, "ia_index": i, "ia_label": str(label)} 

886 for pid, ia_list in ia_lists.items() 

887 for i, label in enumerate(ia_list) 

888 ] 

889 words = words.merge( 

890 pd.DataFrame(labels), on=["unique_paragraph_id", "ia_index"], how="left" 

891 ) 

892 report = { 

893 "geometry_source": source, 

894 "interpolated_fraction": round(float(interpolated_fraction), 4), 

895 "display_source": display_source, 

896 "paragraphs_without_real_boxes": len(ia_counts) 

897 - len(paragraphs_with_real_boxes), 

898 } 

899 return words, report 

900 

901 

902def place_fixations(fix_df: pd.DataFrame, words: pd.DataFrame) -> pd.DataFrame: 

903 """Add ``x``/``y`` to fixations from their interest area's box. 

904 

905 Exactly inverts EyeGenBench's landing-position formula 

906 (``(fix_x - left) / (right - left)``), so wherever the box is real the 

907 round trip is lossless. Fixations with no matching box are dropped -- they 

908 cannot be placed, and a wrong placement is worse than a missing one. 

909 

910 Some EyeGenBench corpora' ``fix_df`` already carries its own 

911 ``start_x``/``start_y``/``end_x``/``end_y`` columns (verified: PoTeC -- 

912 passthrough source columns, not part of EyeGenBench's documented schema). 

913 A plain merge would collide on those names and pandas would suffix both 

914 sides (``start_x_x``/``start_x_y``), leaving no column literally named 

915 ``start_x`` -- exactly the R24 crash. The box used for placement must 

916 always be the **words** frame's box, resolved once and unambiguously by 

917 merging the words-side columns in under private names, never whatever 

918 ``fix_df`` happened to already carry; ``fix_df``'s own box columns (if 

919 any) are left untouched in the output as plain passthrough data. 

920 """ 

921 box_cols = ["start_x", "start_y", "end_x", "end_y"] 

922 internal = {c: f"_words_{c}" for c in box_cols} 

923 geometry_col = "_words_geometry_source" 

924 words_boxes = words[["unique_paragraph_id", "ia_index", *box_cols]].copy() 

925 words_boxes[geometry_col] = words.get( 

926 "geometry_source", pd.Series(index=words.index, dtype="object") 

927 ) 

928 words_boxes = words_boxes.rename(columns=internal) 

929 merged = fix_df.merge( 

930 words_boxes, on=["unique_paragraph_id", "ia_index"], how="inner" 

931 ) 

932 if "fix_landing_position" in merged.columns: 

933 landing = pd.to_numeric(merged["fix_landing_position"], errors="coerce").fillna( 

934 0.5 

935 ) 

936 else: 

937 landing = pd.Series(0.5, index=merged.index) 

938 landing = landing.clip(0.0, 1.0) 

939 box_start_x = merged.pop(internal["start_x"]) 

940 box_end_x = merged.pop(internal["end_x"]) 

941 box_start_y = merged.pop(internal["start_y"]) 

942 box_end_y = merged.pop(internal["end_y"]) 

943 geometry_source = merged.pop(geometry_col) 

944 merged["x"] = box_start_x + landing * (box_end_x - box_start_x) 

945 inferred_y = (box_start_y + box_end_y) / 2.0 

946 recorded_y = pd.to_numeric( 

947 merged.get("recorded_fixation_y", pd.Series(index=merged.index, dtype=float)), 

948 errors="coerce", 

949 ) 

950 use_recorded = geometry_source.eq(GEOMETRY_REAL) & recorded_y.notna() 

951 merged["y"] = recorded_y.where(use_recorded, inferred_y) 

952 merged["fixation_y_source"] = use_recorded.map( 

953 {True: "recorded", False: "word-box-center"} 

954 ) 

955 return merged