Coverage for scanpath_studio/measures.py: 94%

422 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""Compute reading measures from fixations and word bounding boxes. 

2 

3Definitions follow standard reading-research conventions (Rayner 1998; 

4Inhoff & Radach 1998). All measures are per (participant, trial, word). When a 

5column already exists on the words dataframe (e.g. pre-aggregated EyeLink IA 

6metrics) it is preserved; we only compute values that are not present. 

7 

8Canonical output columns added to words: 

9- first_fixation_ms — FFD: duration of the first fixation on this word 

10- first_pass_gaze_ms — FPRT / gaze duration 

11- regression_path_ms — RPD / go-past time 

12- total_fixation_ms — TFD / dwell 

13- n_fixations — fixation count 

14- skip_flag — True if no first-pass fixation 

15- regression_in_flag — True if any later fixation returned here 

16- regression_out_flag — True if a fixation here was followed by a regression 

17- first_fix_x, first_fix_y — landing position of the first-pass first fixation 

18 

19The fixations dataframe is also enriched with: 

20- word_id — the data's own when it has one, else bbox containment 

21- saccade_amplitude — pixel distance from the previous fixation in the trial. 

22 Always pixels (BUG-25): EyeLink's degree-valued 

23 NEXT_SAC_AMPLITUDE / PREVIOUS_SAC_AMPLITUDE keep their 

24 own `*_deg` names rather than sharing this one. 

25- progression — 1 if the next fixation moves to a later word, -1 if earlier, 0 otherwise 

26- is_regression — True if this fixation lands on a word earlier than the 

27 running maximum word reached in the trial 

28""" 

29 

30from __future__ import annotations 

31 

32import numpy as np 

33import pandas as pd 

34 

35from .multipart import grouping_columns 

36 

37# A recorded ``timestamp_ms`` series is trusted as real reading time only when 

38# its span covers at least this fraction of the summed fixation durations. 

39# Fixations don't overlap, so a genuine recording spans at least its total 

40# dwell; the 0,1,2,… row index ``data.normalize_fixations`` synthesises when the 

41# source has no timestamps collapses to a few ms and must NOT be read as 

42# milliseconds. Shared by the similarity time-curve and the animation clock. 

43REAL_TIMESTAMP_DWELL_FRAC = 0.5 

44 

45 

46def _assign_word_ids_single( 

47 fix_chunk: pd.DataFrame, 

48 word_chunk: pd.DataFrame, 

49) -> np.ndarray: 

50 """Vectorized fixation→word_id assignment for a single trial's frames. 

51 

52 Tests every fixation in ``fix_chunk`` against every word box in 

53 ``word_chunk`` (no participant/trial grouping — the caller is responsible 

54 for slicing to one trial, or for passing frames whose ids deliberately 

55 don't match, as the model scanpaths do). A fixation outside every box gets 

56 NaN: there is no snapping to a nearby word. 

57 

58 Returns a float array aligned to ``fix_chunk`` rows (NaN = out of text). 

59 """ 

60 wx0, wy0, wx1, wy1 = word_box_bounds(word_chunk) 

61 wids = word_chunk["word_id"].to_numpy() 

62 

63 fx = pd.to_numeric(fix_chunk["x"], errors="coerce").to_numpy(dtype=float) 

64 fy = pd.to_numeric(fix_chunk["y"], errors="coerce").to_numpy(dtype=float) 

65 

66 in_box = word_box_contains( 

67 fx[:, None], fy[:, None], wx0[None, :], wy0[None, :], wx1[None, :], wy1[None, :] 

68 ) 

69 word_idx = np.where(in_box.any(axis=1), in_box.argmax(axis=1), -1) 

70 return np.where(word_idx >= 0, wids[np.clip(word_idx, 0, None)], np.nan) 

71 

72 

73# --- Tiling layouts: boxes that carry the following space -------------------- 

74# Interest areas are defined by the *experiment*, not by the tracker — EyeLink 

75# Data Viewer just measures against whatever rectangles the IAS file gives it — 

76# and a stimulus generator that tiles a monospaced line hands every box the 

77# whole following inter-word space as trailing padding. On the bundled demo 

78# every box is exactly ``(n_chars + 1) × advance`` px wide and starts at the 

79# word's first glyph (checked against the corpus's own stimulus images: per line 

80# the ink starts within 3 px of the first box's left edge and stops ~1 advance 

81# short of the last box's right edge), so a fixation on the space after a word 

82# belongs to that word, as it does in EyeLink's own reports. 

83# 

84# **The boxes are used exactly as the experiment defined them** (BUG-83). 

85# BUG-11 used to pull every tiling boundary back half a space, to the middle of 

86# the whitespace. It reassigned 7.4% of the demo's fixations relative to 

87# EyeLink's own interest-area assignment, which is the one the corpus' IA_* 

88# measures were computed from, so it was reverted: a reading measure has to be 

89# computed against the interest areas the study reported. 

90# 

91# What the layout's shape is still needed for is *inside* a word. A tiling box 

92# is one advance wider than its glyph run, so `word_char_advance` divides by 

93# ``len(text) + 1`` there (BUG-27), and `word_glyph_span` says where the letters 

94# start, for landing positions — assumed to be the box's left edge. (OneStop's 

95# own screens centre each box on its word, half a space either side, so the 

96# drawn word label sits at the box centre — BUG-97.) `word_box_space_px` only 

97# reports a padding width when the layout really looks like "tiling boxes with 

98# one trailing space"; glyph-tight AOIs (PoTeC, MultiplEYE) read 0.0. 

99_TILING_GAP_TOL_PX = 1.0 

100# Box widths are integers in the exports, so `(n_chars + 1) × advance` is only 

101# met to within a pixel of rounding; and a stray token (a stripped-out glyph, a 

102# joined punctuation mark) shouldn't disqualify an otherwise regular layout. 

103_ADVANCE_RESIDUAL_TOL_PX = 1.5 

104_ADVANCE_MIN_AGREEMENT = 0.95 

105 

106 

107def _sample_trial_words(words: pd.DataFrame) -> pd.DataFrame: 

108 """One trial's words — the layout is a property of the corpus, not a trial. 

109 

110 Necessary as well as cheap: every trial is drawn at the same screen 

111 positions, so clustering lines across a whole corpus would merge different 

112 trials' words into one "line" and read their interleaved x as gaps. 

113 """ 

114 keys = grouping_columns(words) 

115 if not keys: 

116 return words 

117 first = words[keys].iloc[0] 

118 mask = pd.Series(True, index=words.index) 

119 for key in keys: 

120 mask &= words[key] == first[key] 

121 return words[mask] 

122 

123 

124def word_box_space_px(words: pd.DataFrame) -> float: 

125 """Width of the trailing inter-word padding baked into each box, or ``0.0``. 

126 

127 Non-zero only for a monospaced, *tiling* layout whose boxes are consistently 

128 ``(n_chars + 1)`` advances wide — the shape that carries the whole space as 

129 trailing padding. Anything else (glyph-tight AOIs, proportional fonts, 

130 missing text) reports 0.0, i.e. "every box is its glyph run". 

131 

132 It never moves a box edge (BUG-83): it only tells the within-word accessors 

133 — :func:`word_char_advance` and :func:`word_glyph_span` — how many character 

134 cells a box holds. 

135 """ 

136 needed = {"x", "width", "text"} 

137 if words is None or words.empty or not needed <= set(words.columns): 

138 return 0.0 

139 sample = _sample_trial_words(words) 

140 x = pd.to_numeric(sample["x"], errors="coerce") 

141 width = pd.to_numeric(sample["width"], errors="coerce") 

142 chars = sample["text"].astype(str).str.len() 

143 ok = x.notna() & width.notna() & (chars > 0) & (width > 0) 

144 if ok.sum() < 3: 

145 return 0.0 

146 # One advance per character plus exactly one for the trailing space. 

147 advance = float((width[ok] / (chars[ok] + 1)).median()) 

148 if advance <= 0: 

149 return 0.0 

150 residual = (width[ok] - (chars[ok] + 1) * advance).abs() 

151 if (residual <= _ADVANCE_RESIDUAL_TOL_PX).mean() < _ADVANCE_MIN_AGREEMENT: 

152 return 0.0 # not monospaced-with-one-trailing-space 

153 # And the boxes must actually tile: a layout with real gaps is glyph-tight, 

154 # so its boundaries already sit in the whitespace. 

155 lines = cluster_word_lines(sample) 

156 for _, line_words in sample[ok].groupby(lines[ok], sort=False): 

157 ordered = line_words.sort_values("x") 

158 if len(ordered) < 2: 

159 continue 

160 left = pd.to_numeric(ordered["x"], errors="coerce").to_numpy() 

161 right = left + pd.to_numeric(ordered["width"], errors="coerce").to_numpy() 

162 gaps = left[1:] - right[:-1] 

163 if len(gaps) and np.nanmax(np.abs(gaps)) > _TILING_GAP_TOL_PX: 

164 return 0.0 

165 return advance 

166 

167 

168def word_box_bounds( 

169 words: pd.DataFrame, 

170) -> tuple[np.ndarray, np.ndarray, np.ndarray, np.ndarray]: 

171 """``(x0, y0, x1, y1)`` interest-area edges, exactly as the data defines them. 

172 

173 **The** accessor for word-box geometry: anything that tests a point against a 

174 box or draws one goes through here — fixation→word assignment, the in-text 

175 flag, the drawn outlines, the word heatmaps, the critical-span frame, drift 

176 correction and the model scanpaths — so there is one definition of where a 

177 word's interest area is. It is ``x .. x + width`` by ``y .. y + height``, 

178 the experiment's own rectangles (BUG-83). On a tiling corpus that includes 

179 the space after the word, which is what EyeLink's reports assume. 

180 

181 Pure: it reads ``x``/``y``/``width``/``height`` and returns arrays. 

182 

183 **Not** the frame for a position *inside* a word: a tiling box's right-hand 

184 cell is a space, so letters are counted from ``x`` in 

185 :func:`word_char_advance` units, and :func:`word_glyph_span` says where the 

186 glyphs are. 

187 """ 

188 if words is None or words.empty: 

189 # A column-less empty frame is a legitimate "no words" input. 

190 empty = np.empty(0, dtype=float) 

191 return empty, empty.copy(), empty.copy(), empty.copy() 

192 x = pd.to_numeric(words["x"], errors="coerce").to_numpy(dtype=float) 

193 y = pd.to_numeric(words["y"], errors="coerce").to_numpy(dtype=float) 

194 w = pd.to_numeric(words["width"], errors="coerce").to_numpy(dtype=float) 

195 h = pd.to_numeric(words["height"], errors="coerce").to_numpy(dtype=float) 

196 return x, y, x + w, y + h 

197 

198 

199def word_box_contains( 

200 px: np.ndarray, 

201 py: np.ndarray, 

202 x0: np.ndarray, 

203 y0: np.ndarray, 

204 x1: np.ndarray, 

205 y1: np.ndarray, 

206) -> np.ndarray: 

207 """Is each point inside each box? **The** containment rule, broadcasting. 

208 

209 Half-open — ``x0 <= x < x1`` and ``y0 <= y < y1`` — so a box ``width`` px 

210 wide holds exactly ``width`` pixel columns, and a point on the edge two 

211 tiling boxes share belongs to the box that *starts* there: the word to the 

212 right, the line below. That is how EyeLink assigned every such fixation in 

213 the bundled demo (30 of them, all integer coordinates on a shared edge), 

214 where a closed test handed each to the earlier box instead (BUG-83). 

215 Assignment, the out-of-text flag and the word heatmap all test with this, 

216 so a fixation is counted towards exactly one word. 

217 """ 

218 return (px >= x0) & (px < x1) & (py >= y0) & (py < y1) 

219 

220 

221def word_char_advance( 

222 words: pd.DataFrame, 

223 *, 

224 layout: pd.DataFrame | None = None, 

225 chars: np.ndarray | None = None, 

226) -> np.ndarray: 

227 """Width of one character in each word, in px — the *within-word* scale. 

228 

229 **The** accessor for anything that measures a position *inside* a word in 

230 letters (initial landing position, a saccade's launch/landing letter), the 

231 way :func:`word_box_bounds` is the accessor for the boundary between words. 

232 

233 ``width / len(text)`` is wrong on a tiling corpus (BUG-27, found by VAL-5): 

234 it divides a box that is ``len(text) + 1`` advances wide by ``len(text)``, 

235 so every letter was reported ~``(n+1)/n`` too wide and every landing 

236 position that far into the word. The fix is :func:`word_box_space_px`'s 

237 detection applied to the denominator — ``width / (len(text) + 1)`` for a 

238 tiling layout, ``width / len(text)`` for a glyph-tight one. 

239 

240 The *origin* of a within-word position is the word's ``x``, which on every 

241 layout this app recognises is both where the box starts and where the first 

242 glyph does. On a tiling box the last cell, ``[x + n × advance, x + width)``, 

243 is the space after the word — letter ``n + 1`` of the interest area. 

244 

245 Pass ``layout`` — the full trial — when ``words`` is a *subset* (a 

246 highlighted span, one word): tiling is a property of the whole line, and a 

247 subset's holes read as glyph-tight gaps. 

248 

249 ``chars`` lets a caller that already counted the characters hand them in — 

250 ``.astype(str).str.len()`` is a Python-level pass over the frame, and 

251 :func:`word_glyph_span` needs the same counts for the glyph run. 

252 """ 

253 if words is None or words.empty or not {"width", "text"} <= set(words.columns): 

254 # NaN, never `np.empty` — that returns *uninitialized* floats, and a 

255 # garbage positive one passes the `isfinite and > 0` guard at the call 

256 # site and lands a silently wrong landing position in a cached frame. 

257 return np.full(0 if words is None else len(words), np.nan, dtype=float) 

258 width = pd.to_numeric(words["width"], errors="coerce").to_numpy(dtype=float) 

259 counts = word_char_counts(words) if chars is None else chars 

260 padded = word_box_space_px(words if layout is None else layout) > 0 

261 return width / (counts + (1.0 if padded else 0.0)) 

262 

263 

264def word_char_counts(words: pd.DataFrame) -> np.ndarray: 

265 """Characters per word, floor 1 — the companion of :func:`word_char_advance`.""" 

266 return words["text"].astype(str).str.len().clip(lower=1).to_numpy(dtype=float) 

267 

268 

269def word_glyph_span( 

270 words: pd.DataFrame, *, layout: pd.DataFrame | None = None 

271) -> tuple[np.ndarray, np.ndarray]: 

272 """``(start, run)`` — where each word's glyphs begin, and how wide they are. 

273 

274 Not an interest area: ``agg.landing_positions`` measures a landing (and 

275 mirrors an RTL one) across it. On a glyph-tight layout the run is the whole 

276 box; on a tiling one it is ``len(text)`` :func:`word_char_advance` units, 

277 one advance short of the box, whose last cell is the space after the word. 

278 It assumes the glyphs start at ``x``. OneStop's experiment screens instead 

279 centre each tiling box on its word (half a space either side), which is 

280 why the drawn word label is centred in the box rather than placed on this 

281 run (BUG-97). 

282 

283 Falls back to the box ``width`` for a frame with no ``text`` column, where 

284 there are no letters to count. ``layout`` is as for 

285 :func:`word_char_advance`. 

286 """ 

287 if words is None or words.empty: 

288 empty = np.empty(0, dtype=float) 

289 return empty, empty.copy() 

290 start = pd.to_numeric(words["x"], errors="coerce").to_numpy(dtype=float) 

291 width = pd.to_numeric(words["width"], errors="coerce").to_numpy(dtype=float) 

292 if "text" not in words.columns: 

293 return start, width 

294 chars = word_char_counts(words) 

295 run = word_char_advance(words, layout=layout, chars=chars) * chars 

296 return start, np.where(np.isfinite(run) & (run > 0), run, width) 

297 

298 

299def assign_fixations_to_words( 

300 fixations: pd.DataFrame, 

301 words: pd.DataFrame, 

302 *, 

303 overwrite: bool = False, 

304) -> pd.DataFrame: 

305 """Give each fixation the word it belongs to. 

306 

307 When the fixations carry a ``word_id`` (the user mapped one, e.g. EyeLink's 

308 ``CURRENT_FIX_INTEREST_AREA_ID``), it is used exactly as given — a blank 

309 stays blank, since the data's own "no word" is an answer, not a gap — and 

310 nothing is computed. Only a frame with no word ids at all (or 

311 ``overwrite=True``) is assigned from geometry: the word box the fixation 

312 falls in, else NaN. 

313 

314 Box edges come from :func:`word_box_bounds` — the experiment's own 

315 rectangles (BUG-83) — so on a tiling corpus a fixation on the space after a 

316 word is credited to that word, exactly as EyeLink's interest-area report 

317 credits it. 

318 """ 

319 if fixations.empty or words.empty: 

320 return fixations 

321 

322 out = fixations.copy() 

323 if "word_id" not in out.columns or overwrite: 

324 out["word_id"] = np.nan 

325 

326 need_idx = out["word_id"].isna() 

327 if not need_idx.all(): 

328 return out 

329 

330 # Per (participant, trial), do a fast vectorized box-test against that 

331 # trial's words. 

332 keys = grouping_columns(out) 

333 if keys != grouping_columns(words): 

334 raise ValueError( 

335 "Screen ID is set in only one table; set it in both or neither." 

336 ) 

337 groups = out[need_idx].groupby(keys, sort=False) 

338 word_groups = words.groupby(keys, sort=False) 

339 

340 assignments = pd.Series(np.nan, index=out.index[need_idx], dtype=float) 

341 for group_key, fix_chunk in groups: 

342 lookup_key = group_key if len(keys) > 1 else group_key[0] 

343 try: 

344 wchunk = word_groups.get_group(lookup_key) 

345 except KeyError: 

346 continue 

347 if wchunk.empty: 

348 continue 

349 assignments.loc[fix_chunk.index] = _assign_word_ids_single(fix_chunk, wchunk) 

350 

351 out.loc[need_idx, "word_id"] = assignments 

352 return out 

353 

354 

355def materialize_runs(fixations: pd.DataFrame) -> pd.DataFrame: 

356 """Add trial run, line run, and per-word visit/pass columns (PRE-16).""" 

357 if fixations.empty: 

358 return fixations.copy() 

359 derived = ( 

360 "run", 

361 "linerun", 

362 "word_runid", 

363 "word_run", 

364 "word_run_fix", 

365 "nrun", 

366 "reread", 

367 ) 

368 out = fixations.drop(columns=[c for c in derived if c in fixations]).copy() 

369 if "word_id" not in out: 

370 out["word_id"] = pd.Series(pd.NA, index=out.index, dtype="Float64") 

371 excluded = out.get("excluded", pd.Series(False, index=out.index)).fillna(False) 

372 active = out.loc[~excluded.astype(bool)].copy() 

373 for column in derived: 

374 out[column] = pd.NA 

375 if active.empty: 

376 return out.sort_index() 

377 keys = grouping_columns(active) 

378 order_cols = keys + (["timestamp_ms"] if "timestamp_ms" in out else []) 

379 active = active.sort_values(order_cols, kind="stable") 

380 group = active.groupby(keys, sort=False, dropna=False) if keys else [((), active)] 

381 pieces = [] 

382 for _, chunk in group: 

383 chunk = chunk.copy() 

384 word = pd.to_numeric( 

385 chunk.get("word_id", pd.Series(pd.NA, index=chunk.index)), 

386 errors="coerce", 

387 ) 

388 line_source = ( 

389 chunk["line_id"] 

390 if "line_id" in chunk 

391 else chunk.get("line_idx", pd.Series(pd.NA, index=chunk.index)) 

392 ) 

393 line = pd.to_numeric(line_source, errors="coerce") 

394 chunk["run"] = (word.diff().fillna(0) < 0).cumsum().astype("Int64") + 1 

395 chunk["linerun"] = ( 

396 (line.ne(line.shift()) & line.notna()).cumsum().astype("Int64") 

397 ) 

398 visit_start = word.ne(word.shift()) | word.isna() 

399 chunk["word_runid"] = visit_start.cumsum().astype("Int64") 

400 valid = chunk[word.notna()].copy() 

401 if not valid.empty: 

402 visits = valid[["word_id", "word_runid"]].drop_duplicates() 

403 visits["word_run"] = visits.groupby("word_id", sort=False).cumcount() + 1 

404 lookup = visits.set_index(["word_id", "word_runid"])["word_run"] 

405 visit_keys = pd.MultiIndex.from_frame(chunk[["word_id", "word_runid"]]) 

406 chunk["word_run"] = lookup.reindex(visit_keys).to_numpy() 

407 chunk["word_run_fix"] = ( 

408 chunk.groupby(["word_id", "word_runid"], dropna=False).cumcount() + 1 

409 ) 

410 else: 

411 chunk["word_run"] = pd.NA 

412 chunk["word_run_fix"] = pd.NA 

413 chunk["nrun"] = chunk.groupby("word_id", dropna=False)["word_runid"].transform( 

414 "nunique" 

415 ) 

416 chunk["reread"] = pd.to_numeric(chunk["word_run"], errors="coerce").gt(1) 

417 pieces.append(chunk) 

418 computed = pd.concat(pieces).sort_index() if pieces else active 

419 for column in derived: 

420 out.loc[computed.index, column] = computed[column] 

421 return out.sort_index() 

422 

423 

424def enrich_fixations(fixations: pd.DataFrame, words: pd.DataFrame) -> pd.DataFrame: 

425 """Add saccade_amplitude, progression, and is_regression to fixations.""" 

426 if fixations.empty: 

427 return fixations 

428 out = fixations.copy() 

429 keys = grouping_columns(out) 

430 out = out.sort_values(keys + ["timestamp_ms"]) 

431 

432 g = out.groupby(keys, sort=False) 

433 dx = g["x"].diff() 

434 dy = g["y"].diff() 

435 if "saccade_amplitude" not in out.columns: 

436 out["saccade_amplitude"] = np.sqrt(dx * dx + dy * dy) 

437 else: 

438 out["saccade_amplitude"] = out["saccade_amplitude"].where( 

439 out["saccade_amplitude"].notna(), np.sqrt(dx * dx + dy * dy) 

440 ) 

441 incoming = np.degrees(np.arctan2(-dy, dx)) 

442 out["angle_incoming"] = incoming 

443 g = out.groupby(keys, sort=False) 

444 out["angle_outgoing"] = g["angle_incoming"].shift(-1) 

445 

446 next_word = g["word_id"].shift(-1) 

447 out["progression"] = np.sign(next_word - out["word_id"]).fillna(0).astype(int) 

448 

449 running_max = g["word_id"].cummax() 

450 out["is_regression"] = (out["word_id"] < running_max).fillna(False).astype(bool) 

451 return materialize_runs(out) 

452 

453 

454def classify_saccades(fixations: pd.DataFrame, words: pd.DataFrame) -> pd.Series: 

455 """Classify each *outgoing* saccade by its reading type (VIZ-8). 

456 

457 A saccade is the segment from one fixation to the next in reading order; its 

458 class is stored on the *departing* fixation, so the last fixation of every 

459 trial (which has no outgoing saccade) is ``None``. Classes follow the classic 

460 reading schematic: 

461 

462 - ``refixation`` — lands back on the same word (``word_id`` unchanged), 

463 - ``regression`` — moves to an earlier line, or backward within a line, 

464 - ``return_sweep`` — sweeps down to a later line (forward reading), 

465 - ``forward`` — advances to the next word on the same line, 

466 - ``skip`` — advances past one or more words on the same line, 

467 - ``other`` — a real saccade whose endpoints can't be classified (an 

468 out-of-text fixation with no assigned word). 

469 

470 Line membership comes from :func:`assign_fixation_lines` (word-box geometry), 

471 word order from the fixation ``word_id`` assignment. Returns an object Series 

472 aligned to ``fixations.index`` (pure — no plotting), so both the render path 

473 (``plots._add_saccade_layer``) and any future analysis can reuse it. Works on 

474 a single already-sliced trial or a multi-trial frame (grouped by 

475 participant/trial). 

476 """ 

477 if fixations.empty: 

478 return pd.Series([], dtype=object, index=fixations.index) 

479 

480 keys = grouping_columns(fixations) 

481 sort_cols = keys + (["timestamp_ms"] if "timestamp_ms" in fixations.columns else []) 

482 order = fixations.sort_values(sort_cols).index if sort_cols else fixations.index 

483 

484 if "word_id" in fixations.columns: 

485 word = pd.to_numeric(fixations["word_id"], errors="coerce").reindex(order) 

486 else: 

487 word = pd.Series(np.nan, index=order, dtype=float) 

488 if words is not None and not words.empty: 

489 line = assign_fixation_lines(fixations, words).reindex(order) 

490 else: 

491 line = pd.Series(np.nan, index=order, dtype=float) 

492 

493 if keys: 

494 grp = [fixations[k].reindex(order) for k in keys] 

495 next_word = word.groupby(grp).shift(-1) 

496 next_line = line.groupby(grp).shift(-1) 

497 next_pos = pd.Series(np.arange(len(order)), index=order).groupby(grp).shift(-1) 

498 else: 

499 next_word = word.shift(-1) 

500 next_line = line.shift(-1) 

501 next_pos = pd.Series(np.arange(len(order)), index=order).shift(-1) 

502 

503 word_arr = word.to_numpy(dtype=float) 

504 next_word_arr = next_word.to_numpy(dtype=float) 

505 dw = next_word_arr - word_arr 

506 dl = (next_line - line).to_numpy(dtype=float) 

507 has_next = next_pos.notna().to_numpy() 

508 # A saccade is classifiable only when BOTH endpoints have an assigned word; 

509 # otherwise the word/line deltas are meaningless (an off-text fixation still 

510 # gets a *nearest* line, which would spuriously read as a line change). 

511 both_words = ~np.isnan(word_arr) & ~np.isnan(next_word_arr) 

512 same_line = np.isnan(dl) | (dl == 0) 

513 

514 # Priority order: same word first, then line moves, then within-line word moves. 

515 conds = [ 

516 dw == 0, # refixation 

517 dl < 0, # regression — up to an earlier line 

518 dl > 0, # return sweep — down to a later line 

519 same_line & (dw < 0), # regression — backward within the line 

520 same_line & (dw == 1), # forward 

521 same_line & (dw >= 2), # skip 

522 ] 

523 choices = [ 

524 "refixation", 

525 "regression", 

526 "return_sweep", 

527 "regression", 

528 "forward", 

529 "skip", 

530 ] 

531 classed = np.select(conds, choices, default="other").astype(object) 

532 classed[~both_words] = "other" # a real saccade, but an endpoint is off-text 

533 classed[~has_next] = None # last fixation of a trial: no outgoing saccade 

534 # dtype=object is load-bearing: pandas 3.0 infers a dedicated `str` dtype for 

535 # an all-string+None array and coerces the None sentinel to NaN, breaking the 

536 # "None for the last fixation" contract. Pin object so None survives. 

537 return pd.Series(classed, index=order, dtype=object).reindex(fixations.index) 

538 

539 

540def rebased_fixation_onsets(ordered_fixations: pd.DataFrame) -> np.ndarray: 

541 """Fixation onset times (ms), rebased so the first fixation is t=0. 

542 

543 ``ordered_fixations`` must already be in reading order (sorted by 

544 ``timestamp_ms``); the returned array is aligned to its rows. Uses the 

545 recorded ``timestamp_ms`` when they look like real times — their span is at 

546 least ``REAL_TIMESTAMP_DWELL_FRAC`` of the summed durations — otherwise lays 

547 fixations back-to-back by their durations, so a synthesised 0,1,2,… index 

548 doesn't crush the time axis. Fixations normalization numbered because the 

549 table had no onset (``data.TIMESTAMP_SYNTHESIZED``) always take the 

550 durations — the same estimate the reading summaries use. Shared by the similarity time-curve 

551 (:func:`scanpath_studio.similarity._rebased_onsets`) and the animation clock 

552 (:func:`scanpath_studio.plots._scanpath_anim_specs`). 

553 """ 

554 if ordered_fixations.empty: 

555 return np.array([], dtype=float) 

556 if "duration_ms" in ordered_fixations.columns: 

557 dur = ( 

558 pd.to_numeric(ordered_fixations["duration_ms"], errors="coerce") 

559 .fillna(0) 

560 .to_numpy(dtype=float) 

561 ) 

562 else: 

563 dur = np.zeros(len(ordered_fixations)) 

564 contiguous = np.concatenate(([0.0], np.cumsum(dur)[:-1])) if len(dur) else dur 

565 from .data import timestamps_synthesized 

566 

567 if timestamps_synthesized(ordered_fixations): 

568 return contiguous 

569 if "timestamp_ms" in ordered_fixations.columns: 

570 ts = pd.to_numeric(ordered_fixations["timestamp_ms"], errors="coerce").to_numpy( 

571 dtype=float 

572 ) 

573 total_dwell = float(dur.sum()) if len(dur) else 0.0 

574 if ( 

575 len(ts) 

576 and not np.isnan(ts).any() 

577 and (ts[-1] - ts[0]) >= REAL_TIMESTAMP_DWELL_FRAC * total_dwell 

578 ): 

579 return ts - ts[0] 

580 return contiguous 

581 

582 

583# --------------------------------------------------------------------------- 

584# Geometry helpers: line clustering, in-text test, fixation -> line. 

585# 

586# These power the "highlight out-of-text fixations" and "color fixations by 

587# line" plot options. They are deliberately pure (no Streamlit, no plotting) 

588# so they can be unit-tested against a known synthetic layout. 

589# --------------------------------------------------------------------------- 

590 

591 

592def _in_any_box(fix_chunk: pd.DataFrame, word_chunk: pd.DataFrame) -> np.ndarray: 

593 """Boolean array (aligned to fix_chunk order): is each fixation inside any box?""" 

594 x0, y0, x1, y1 = word_box_bounds(word_chunk) 

595 fx = pd.to_numeric(fix_chunk["x"], errors="coerce").to_numpy(dtype=float) 

596 fy = pd.to_numeric(fix_chunk["y"], errors="coerce").to_numpy(dtype=float) 

597 inside = word_box_contains( 

598 fx[:, None], fy[:, None], x0[None, :], y0[None, :], x1[None, :], y1[None, :] 

599 ) 

600 return inside.any(axis=1) 

601 

602 

603def cluster_word_lines(words: pd.DataFrame, tol_frac: float = 0.5) -> pd.Series: 

604 """Assign each word a 0-based line id by clustering on vertical position. 

605 

606 OneStop IA exports rarely carry a real per-word line number (the 

607 normalized ``line_idx`` is often a constant), so we infer visual lines from 

608 word-box geometry: words are sorted by vertical center and a new line 

609 starts whenever the center jumps by more than ``tol_frac`` of the median 

610 word height. Lines are numbered top-to-bottom starting at 0. 

611 

612 Returns an int Series aligned to ``words.index`` (empty when ``words`` is 

613 empty). Mirrors the line clustering used by ``plots.build_critical_span_overlay``. 

614 """ 

615 if words.empty: 

616 return pd.Series([], dtype="int64", index=words.index) 

617 heights = pd.to_numeric(words["height"], errors="coerce") 

618 typical_h = float(heights.median()) if heights.notna().any() else 1.0 

619 typical_h = typical_h if typical_h > 0 else 1.0 

620 y_center = pd.to_numeric(words["y"], errors="coerce") + heights.fillna(0) / 2.0 

621 order = y_center.sort_values(kind="stable") 

622 line_of_sorted = ( 

623 (order.diff().fillna(0) > typical_h * tol_frac).cumsum().astype(int) 

624 ) 

625 return line_of_sorted.reindex(words.index) 

626 

627 

628def _groupwise(fixations: pd.DataFrame, words: pd.DataFrame, fn) -> pd.Series: 

629 """Apply ``fn(fix_chunk, word_chunk)`` per (participant, trial), aligning 

630 the per-chunk result back onto a Series indexed like ``fixations``. 

631 

632 Falls back to a single group when the id columns are absent (e.g. the 

633 figure builders pass already-sliced single-trial frames).""" 

634 keys = grouping_columns(fixations) 

635 # object dtype so a chunk fn may return bools (in-text) or floats (line ids) 

636 # without triggering a dtype-incompatibility cast; callers coerce the result. 

637 out = pd.Series(np.nan, index=fixations.index, dtype=object) 

638 has_groups = bool(keys) and keys == grouping_columns(words) 

639 if not has_groups: 

640 out.loc[:] = fn(fixations, words) 

641 return out 

642 word_groups = words.groupby(keys, sort=False) 

643 for key, fix_chunk in fixations.groupby(keys, sort=False): 

644 try: 

645 word_chunk = word_groups.get_group(key) 

646 except KeyError: 

647 continue 

648 if word_chunk.empty: 

649 continue 

650 out.loc[fix_chunk.index] = fn(fix_chunk, word_chunk) 

651 return out 

652 

653 

654def fixation_in_text_mask(fixations: pd.DataFrame, words: pd.DataFrame) -> pd.Series: 

655 """Boolean Series: True where a fixation falls inside any word box. 

656 

657 "Out-of-text" fixations are simply ``~fixation_in_text_mask(...)``. Works 

658 on multi-trial frames (grouped by participant/trial) or on a single 

659 already-sliced trial. Fixations with non-finite coordinates count as 

660 out-of-text (mask = False).""" 

661 if fixations.empty: 

662 return pd.Series([], dtype=bool, index=fixations.index) 

663 if words.empty: 

664 return pd.Series(False, index=fixations.index) 

665 res = _groupwise(fixations, words, _in_any_box) 

666 return res.where(res.notna(), other=False).astype(bool) 

667 

668 

669def assign_fixation_lines(fixations: pd.DataFrame, words: pd.DataFrame) -> pd.Series: 

670 """Assign each fixation the 0-based line id of the nearest text line. 

671 

672 Lines are derived from word geometry via :func:`cluster_word_lines`; each 

673 fixation is mapped to the line whose mean vertical center is closest to the 

674 fixation's y. Returns a float Series (NaN where unmappable) aligned to 

675 ``fixations.index`` so it can be used as a categorical color field.""" 

676 if fixations.empty: 

677 return pd.Series([], dtype="float64", index=fixations.index) 

678 if words.empty: 

679 return pd.Series(np.nan, index=fixations.index, dtype="float64") 

680 

681 def _nearest_line(fix_chunk: pd.DataFrame, word_chunk: pd.DataFrame) -> np.ndarray: 

682 lines = cluster_word_lines(word_chunk) 

683 y_center = ( 

684 pd.to_numeric(word_chunk["y"], errors="coerce") 

685 + pd.to_numeric(word_chunk["height"], errors="coerce").fillna(0) / 2.0 

686 ) 

687 centers = y_center.groupby(lines).mean() 

688 line_ids = centers.index.to_numpy(dtype=float) 

689 line_cy = centers.to_numpy(dtype=float) 

690 fy = pd.to_numeric(fix_chunk["y"], errors="coerce").to_numpy(dtype=float) 

691 dist = np.abs(fy[:, None] - line_cy[None, :]) 

692 nearest = np.where(np.isnan(fy), -1, dist.argmin(axis=1)) 

693 result = np.where(nearest >= 0, line_ids[np.clip(nearest, 0, None)], np.nan) 

694 return result 

695 

696 return pd.to_numeric(_groupwise(fixations, words, _nearest_line), errors="coerce") 

697 

698 

699def compute_per_word_measures( 

700 fixations: pd.DataFrame, words: pd.DataFrame 

701) -> pd.DataFrame: 

702 """Compute canonical reading measures per word. 

703 

704 Returns a copy of `words` with computed columns added. Existing values on 

705 `words` (e.g. EyeLink IA metrics) take precedence over computed ones. 

706 """ 

707 if words.empty: 

708 return words.copy() 

709 

710 enriched = ( 

711 enrich_fixations(assign_fixations_to_words(fixations, words), words) 

712 if not fixations.empty 

713 else fixations 

714 ) 

715 

716 out = words.copy() 

717 key_cols = grouping_columns(words, include_word=True) 

718 # VAL-5: the letter scale for the landing measures, resolved once on the 

719 # whole frame (tiling is a property of the layout, so it must not be 

720 # re-detected per trial word, which is a subset with holes in it). 

721 char_advance = pd.Series(word_char_advance(words), index=words.index) 

722 

723 # Initialize defaults 

724 computed = ( 

725 words[key_cols] 

726 .drop_duplicates() 

727 .assign( 

728 _first_fixation_ms=np.nan, 

729 _first_pass_gaze_ms=np.nan, 

730 _regression_path_ms=np.nan, 

731 _total_fixation_ms=0.0, 

732 _n_fixations=0, 

733 _skip_flag=True, 

734 _regression_in_flag=False, 

735 _regression_out_flag=False, 

736 _first_fix_x=np.nan, 

737 _first_fix_y=np.nan, 

738 _initial_landing_position=np.nan, 

739 _initial_landing_distance=np.nan, 

740 _number_of_regressions_in=0, 

741 _second_pass_duration=0.0, 

742 _single_fixation_duration=np.nan, 

743 ) 

744 ) 

745 

746 # PERF-12: each trial's word rows, found once. The landing measures below 

747 # used to locate a fixated word by masking the *whole* words frame — one 

748 # full-frame comparison per fixated word, per identity column — which made 

749 # this function quadratic in corpus size (1.7 s on the demo, 22 s at 8×). 

750 word_num = pd.to_numeric(words["word_id"], errors="coerce").to_numpy(dtype=float) 

751 identity_cols = grouping_columns(enriched) if not enriched.empty else [] 

752 word_rows: dict[tuple, np.ndarray] = {} 

753 if identity_cols and set(identity_cols) <= set(words.columns): 

754 grouped = words.groupby(identity_cols, sort=False).indices 

755 word_rows = { 

756 (k if isinstance(k, tuple) else (k,)): v for k, v in grouped.items() 

757 } 

758 

759 analysis_fixations = enriched 

760 if "excluded" in enriched.columns: 

761 analysis_fixations = enriched[~enriched["excluded"].fillna(False).astype(bool)] 

762 if not analysis_fixations.empty and "word_id" in analysis_fixations.columns: 

763 per_word_rows = [] 

764 # Group fixations by trial to walk them in temporal order. 

765 trial_keys = grouping_columns(analysis_fixations) 

766 for group_key, trial_chunk in analysis_fixations.groupby( 

767 trial_keys, sort=False 

768 ): 

769 values = group_key if isinstance(group_key, tuple) else (group_key,) 

770 identity = dict(zip(trial_keys, values)) 

771 # BUG-66: the walk below sees the off-text fixations too — one lands 

772 # between two runs, so it must end the first of them, exactly as 

773 # `materialize_runs` numbers `word_run` (and so second-pass). Walking 

774 # only the in-text rows glued the two runs into one first pass, and a 

775 # word's first + second pass then added up to more than its total. 

776 trial_chunk = trial_chunk.sort_values("timestamp_ms") 

777 fix_chunk = trial_chunk.dropna(subset=["word_id"]) 

778 if fix_chunk.empty: 

779 continue 

780 # Total / n / first-fixation are per-word aggregations 

781 grp = fix_chunk.groupby("word_id") 

782 tot = grp["duration_ms"].sum() 

783 n = grp.size() 

784 ffd = grp["duration_ms"].first() 

785 ffx = grp["x"].first() 

786 ffy = grp["y"].first() 

787 second_pass = ( 

788 fix_chunk[ 

789 pd.to_numeric(fix_chunk.get("word_run"), errors="coerce").eq(2) 

790 ] 

791 .groupby("word_id")["duration_ms"] 

792 .sum() 

793 ) 

794 first_pass_counts = ( 

795 fix_chunk[ 

796 pd.to_numeric(fix_chunk.get("word_run"), errors="coerce").eq(1) 

797 ] 

798 .groupby("word_id") 

799 .size() 

800 ) 

801 

802 # One walk of the trial in time order. The definitions are EyeLink's 

803 # IA_* ones, so a computed measure means the same as an imported one 

804 # (imported values take precedence, and the two must not disagree 

805 # about what a column is): validated against the bundled OneStop IA 

806 # report, which EyeLink Data Viewer produced from these fixations. 

807 first_run: dict[float, float] = {} # IA_FIRST_RUN_DWELL_TIME 

808 first_pass: set = set() # not IA_SKIP 

809 regression_path: dict[float, float] = {} 

810 regression_in: set = set() 

811 regression_out: set = set() 

812 regression_in_count: dict[float, int] = {} 

813 

814 running_max = -np.inf 

815 run_word: float | None = None 

816 run_duration = 0.0 

817 seen: set = set() 

818 go_past_open: dict[float, float] = {} 

819 left_forward: set = set() 

820 

821 prev_word: float | None = None 

822 for row in trial_chunk.itertuples(): 

823 dur = float(row.duration_ms) 

824 if pd.isna(row.word_id): 

825 # Outside every box: ends the current run (BUG-66), but 

826 # opens, extends and closes no go-past window. 

827 if run_word is not None: 

828 first_run.setdefault(run_word, run_duration) 

829 run_word = None 

830 continue 

831 w = float(row.word_id) 

832 

833 # First run: the word's first unbroken stretch of fixations, 

834 # whenever it begins (as IA_FIRST_RUN_DWELL_TIME). 

835 if run_word == w: 

836 run_duration += dur 

837 else: 

838 if run_word is not None: 

839 first_run.setdefault(run_word, run_duration) 

840 run_word = w if w not in first_run else None 

841 run_duration = dur 

842 

843 # BUG-62: first pass means entered *before any later word was 

844 # fixated*. A word first reached by a regression was skipped, 

845 # which is what IA_SKIP says and what `skip_flag` must say. 

846 if w not in seen and w > running_max: 

847 first_pass.add(w) 

848 

849 # BUG-61: go-past (regression-path) time runs from the word's 

850 # first fixation until the first fixation on a later word, and 

851 # EVERY fixation in between counts — a first visit to a skipped 

852 # earlier word during the regression included. Adding only 

853 # revisits dropped those, so RPD always came out short. 

854 for k in go_past_open: 

855 if w <= k: 

856 go_past_open[k] += dur 

857 if w not in seen: 

858 go_past_open[w] = dur 

859 for k in [k for k in go_past_open if w > k]: 

860 regression_path[k] = go_past_open.pop(k) 

861 seen.add(w) 

862 

863 if prev_word is not None and w != prev_word: 

864 if w < prev_word: 

865 regression_in.add(w) 

866 regression_in_count[w] = regression_in_count.get(w, 0) + 1 

867 # BUG-64: a regression *out* is a first-pass event 

868 # (IA_REGRESSION_OUT) — made from a word read in first 

869 # pass, before the eyes first left it forwards — not a 

870 # regression from it at any later time. 

871 if prev_word in first_pass and prev_word not in left_forward: 

872 regression_out.add(prev_word) 

873 else: 

874 left_forward.add(prev_word) 

875 

876 running_max = max(running_max, w) 

877 prev_word = w 

878 

879 if run_word is not None: 

880 first_run.setdefault(run_word, run_duration) 

881 # Windows still open at the end: the reader never moved past them. 

882 regression_path.update(go_past_open) 

883 

884 trial_rows = word_rows.get(tuple(values)) 

885 first_row: dict[float, int] = {} 

886 if trial_rows is not None: 

887 for pos in trial_rows: 

888 first_row.setdefault(word_num[pos], pos) 

889 for w in tot.index: 

890 landing_position = landing_distance = np.nan 

891 pos = first_row.get(float(w)) 

892 if pos is not None: 

893 target = words.iloc[pos] 

894 text_len = max(len(str(target.get("text", ""))), 1) 

895 width = float(pd.to_numeric(target.get("width"), errors="coerce")) 

896 char_width = float(char_advance.iloc[pos]) 

897 if np.isfinite(char_width) and char_width > 0 and width > 0: 

898 rtl = bool(target.get("right_to_left", False)) 

899 # BUG-27: RTL counts from where the glyphs *end* 

900 # (`x + n × advance`), not from the padded box edge 

901 # `x + width` — on a tiling layout those are one whole 

902 # advance apart, so an RTL landing read a letter late 

903 # and disagreed with `aggregation.landing_positions`, 

904 # which measures across the glyph run on both sides. 

905 offset = ( 

906 float(target.get("x")) 

907 + text_len * char_width 

908 - float(ffx.loc[w]) 

909 if rtl 

910 else float(ffx.loc[w]) - float(target.get("x")) 

911 ) 

912 # Unclipped. The glyphs span [1, n + 1); on a tiling 

913 # corpus the box's last cell, [n + 1, n + 2), is the 

914 # space after the word, which belongs to it (BUG-83), so 

915 # a first fixation there reads letter n + 1 — the AOI's 

916 # own last cell — rather than being folded onto letter n. 

917 landing_position = offset / char_width + 1.0 

918 # BUG-65: the word's centre is its glyphs' centre, 

919 # 1 + n / 2 — not (n + 1) / 2, which read every landing 

920 # half a letter right of centre, and not the box's 

921 # centre, which on a tiling box sits half a space 

922 # further right again. 

923 landing_distance = landing_position - (1.0 + text_len / 2.0) 

924 per_word_rows.append( 

925 dict( 

926 **identity, 

927 word_id=w, 

928 _first_fixation_ms=float(ffd.loc[w]), 

929 _first_pass_gaze_ms=float(first_run.get(w, np.nan)), 

930 _regression_path_ms=float(regression_path.get(w, np.nan)), 

931 _total_fixation_ms=float(tot.loc[w]), 

932 _n_fixations=int(n.loc[w]), 

933 _skip_flag=w not in first_pass, 

934 _regression_in_flag=w in regression_in, 

935 _regression_out_flag=w in regression_out, 

936 _first_fix_x=float(ffx.loc[w]), 

937 _first_fix_y=float(ffy.loc[w]), 

938 _initial_landing_position=landing_position, 

939 _initial_landing_distance=landing_distance, 

940 _number_of_regressions_in=int(regression_in_count.get(w, 0)), 

941 _second_pass_duration=float(second_pass.get(w, 0.0)), 

942 _single_fixation_duration=( 

943 float(ffd.loc[w]) 

944 if int(first_pass_counts.get(w, 0)) == 1 

945 else np.nan 

946 ), 

947 ) 

948 ) 

949 

950 if per_word_rows: 

951 new_df = pd.DataFrame(per_word_rows) 

952 updated = computed.merge( 

953 new_df, on=key_cols, how="left", suffixes=("_default", "") 

954 ) 

955 for col in new_df.columns: 

956 if col in key_cols: 

957 continue 

958 default_col = f"{col}_default" 

959 if default_col in updated.columns: 

960 updated[col] = updated[col].where( 

961 updated[col].notna(), updated[default_col] 

962 ) 

963 updated = updated.drop(columns=default_col) 

964 computed = updated 

965 

966 out = out.merge(computed, on=key_cols, how="left") 

967 

968 # Map computed -> canonical name, keeping any existing value. 

969 rename_map = { 

970 "_first_fixation_ms": "first_fixation_ms", 

971 "_first_pass_gaze_ms": "first_pass_gaze_duration_ms", 

972 "_regression_path_ms": "regression_path_duration_ms", 

973 "_total_fixation_ms": "total_fixation_duration_ms", 

974 "_n_fixations": "n_fixations", 

975 "_skip_flag": "skip_flag", 

976 "_regression_in_flag": "regression_in_flag", 

977 "_regression_out_flag": "regression_out_flag", 

978 "_first_fix_x": "first_fix_x", 

979 "_first_fix_y": "first_fix_y", 

980 "_initial_landing_position": "initial_landing_position", 

981 "_initial_landing_distance": "initial_landing_distance", 

982 "_number_of_regressions_in": "number_of_regressions_in", 

983 "_second_pass_duration": "second_pass_duration_ms", 

984 "_single_fixation_duration": "single_fixation_duration_ms", 

985 } 

986 for src, dst in rename_map.items(): 

987 if src not in out.columns: 

988 continue 

989 if dst in out.columns: 

990 out[dst] = out[dst].where(out[dst].notna(), out[src]) 

991 else: 

992 out[dst] = out[src] 

993 out = out.drop(columns=src) 

994 

995 # BUG-63: a word nobody fixated has no first fixation, first pass or go-past 

996 # time — but an imported IA report may say `0` rather than leave the cell 

997 # empty (the bundled OneStop one does, for every such word), and a 0 wins the 

998 # precedence above. Every mean then counted skipped words as 0-ms fixations. 

999 # Total time stays 0 on purpose: the word was read past and got no time. 

1000 if "n_fixations" in out.columns: 

1001 unfixated = pd.to_numeric(out["n_fixations"], errors="coerce").eq(0) 

1002 for col in ( 

1003 "first_fixation_ms", 

1004 "first_pass_gaze_duration_ms", 

1005 "regression_path_duration_ms", 

1006 "single_fixation_duration_ms", 

1007 ): 

1008 if col in out.columns and unfixated.any(): 

1009 out[col] = pd.to_numeric(out[col], errors="coerce").mask(unfixated) 

1010 # The mirror image for second pass, whose rule is "fewer than two runs 

1011 # ⇒ 0": an imported IA_SECOND_RUN_DWELL_TIME leaves those cells empty, 

1012 # so its mean covered only re-read words while the computed one covered 

1013 # every word — one measure, two meanings, depending on the upload. 

1014 if "second_pass_duration_ms" in out.columns: 

1015 known = pd.to_numeric(out["n_fixations"], errors="coerce").notna() 

1016 second = pd.to_numeric(out["second_pass_duration_ms"], errors="coerce") 

1017 out["second_pass_duration_ms"] = second.mask(known & second.isna(), 0.0) 

1018 

1019 # Canonical aliases used elsewhere in the app 

1020 if ( 

1021 "first_pass_gaze_duration_ms" in out.columns 

1022 and "gaze_duration_ms" not in out.columns 

1023 ): 

1024 out["gaze_duration_ms"] = out["first_pass_gaze_duration_ms"] 

1025 

1026 # Ensure dtypes 

1027 for col in ["n_fixations"]: 

1028 if col in out.columns: 

1029 out[col] = ( 

1030 pd.to_numeric(out[col], errors="coerce").fillna(0).astype("Int64") 

1031 ) 

1032 for col in ["skip_flag", "regression_in_flag", "regression_out_flag"]: 

1033 if col in out.columns: 

1034 out[col] = ( 

1035 out[col].astype(object).where(out[col].notna(), False).astype(bool) 

1036 ) 

1037 

1038 return out