Coverage for scanpath_studio/measures.py: 94%
422 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Compute reading measures from fixations and word bounding boxes.
3Definitions follow standard reading-research conventions (Rayner 1998;
4Inhoff & Radach 1998). All measures are per (participant, trial, word). When a
5column already exists on the words dataframe (e.g. pre-aggregated EyeLink IA
6metrics) it is preserved; we only compute values that are not present.
8Canonical output columns added to words:
9- first_fixation_ms — FFD: duration of the first fixation on this word
10- first_pass_gaze_ms — FPRT / gaze duration
11- regression_path_ms — RPD / go-past time
12- total_fixation_ms — TFD / dwell
13- n_fixations — fixation count
14- skip_flag — True if no first-pass fixation
15- regression_in_flag — True if any later fixation returned here
16- regression_out_flag — True if a fixation here was followed by a regression
17- first_fix_x, first_fix_y — landing position of the first-pass first fixation
19The fixations dataframe is also enriched with:
20- word_id — the data's own when it has one, else bbox containment
21- saccade_amplitude — pixel distance from the previous fixation in the trial.
22 Always pixels (BUG-25): EyeLink's degree-valued
23 NEXT_SAC_AMPLITUDE / PREVIOUS_SAC_AMPLITUDE keep their
24 own `*_deg` names rather than sharing this one.
25- progression — 1 if the next fixation moves to a later word, -1 if earlier, 0 otherwise
26- is_regression — True if this fixation lands on a word earlier than the
27 running maximum word reached in the trial
28"""
30from __future__ import annotations
32import numpy as np
33import pandas as pd
35from .multipart import grouping_columns
37# A recorded ``timestamp_ms`` series is trusted as real reading time only when
38# its span covers at least this fraction of the summed fixation durations.
39# Fixations don't overlap, so a genuine recording spans at least its total
40# dwell; the 0,1,2,… row index ``data.normalize_fixations`` synthesises when the
41# source has no timestamps collapses to a few ms and must NOT be read as
42# milliseconds. Shared by the similarity time-curve and the animation clock.
43REAL_TIMESTAMP_DWELL_FRAC = 0.5
46def _assign_word_ids_single(
47 fix_chunk: pd.DataFrame,
48 word_chunk: pd.DataFrame,
49) -> np.ndarray:
50 """Vectorized fixation→word_id assignment for a single trial's frames.
52 Tests every fixation in ``fix_chunk`` against every word box in
53 ``word_chunk`` (no participant/trial grouping — the caller is responsible
54 for slicing to one trial, or for passing frames whose ids deliberately
55 don't match, as the model scanpaths do). A fixation outside every box gets
56 NaN: there is no snapping to a nearby word.
58 Returns a float array aligned to ``fix_chunk`` rows (NaN = out of text).
59 """
60 wx0, wy0, wx1, wy1 = word_box_bounds(word_chunk)
61 wids = word_chunk["word_id"].to_numpy()
63 fx = pd.to_numeric(fix_chunk["x"], errors="coerce").to_numpy(dtype=float)
64 fy = pd.to_numeric(fix_chunk["y"], errors="coerce").to_numpy(dtype=float)
66 in_box = word_box_contains(
67 fx[:, None], fy[:, None], wx0[None, :], wy0[None, :], wx1[None, :], wy1[None, :]
68 )
69 word_idx = np.where(in_box.any(axis=1), in_box.argmax(axis=1), -1)
70 return np.where(word_idx >= 0, wids[np.clip(word_idx, 0, None)], np.nan)
73# --- Tiling layouts: boxes that carry the following space --------------------
74# Interest areas are defined by the *experiment*, not by the tracker — EyeLink
75# Data Viewer just measures against whatever rectangles the IAS file gives it —
76# and a stimulus generator that tiles a monospaced line hands every box the
77# whole following inter-word space as trailing padding. On the bundled demo
78# every box is exactly ``(n_chars + 1) × advance`` px wide and starts at the
79# word's first glyph (checked against the corpus's own stimulus images: per line
80# the ink starts within 3 px of the first box's left edge and stops ~1 advance
81# short of the last box's right edge), so a fixation on the space after a word
82# belongs to that word, as it does in EyeLink's own reports.
83#
84# **The boxes are used exactly as the experiment defined them** (BUG-83).
85# BUG-11 used to pull every tiling boundary back half a space, to the middle of
86# the whitespace. It reassigned 7.4% of the demo's fixations relative to
87# EyeLink's own interest-area assignment, which is the one the corpus' IA_*
88# measures were computed from, so it was reverted: a reading measure has to be
89# computed against the interest areas the study reported.
90#
91# What the layout's shape is still needed for is *inside* a word. A tiling box
92# is one advance wider than its glyph run, so `word_char_advance` divides by
93# ``len(text) + 1`` there (BUG-27), and `word_glyph_span` says where the letters
94# start, for landing positions — assumed to be the box's left edge. (OneStop's
95# own screens centre each box on its word, half a space either side, so the
96# drawn word label sits at the box centre — BUG-97.) `word_box_space_px` only
97# reports a padding width when the layout really looks like "tiling boxes with
98# one trailing space"; glyph-tight AOIs (PoTeC, MultiplEYE) read 0.0.
99_TILING_GAP_TOL_PX = 1.0
100# Box widths are integers in the exports, so `(n_chars + 1) × advance` is only
101# met to within a pixel of rounding; and a stray token (a stripped-out glyph, a
102# joined punctuation mark) shouldn't disqualify an otherwise regular layout.
103_ADVANCE_RESIDUAL_TOL_PX = 1.5
104_ADVANCE_MIN_AGREEMENT = 0.95
107def _sample_trial_words(words: pd.DataFrame) -> pd.DataFrame:
108 """One trial's words — the layout is a property of the corpus, not a trial.
110 Necessary as well as cheap: every trial is drawn at the same screen
111 positions, so clustering lines across a whole corpus would merge different
112 trials' words into one "line" and read their interleaved x as gaps.
113 """
114 keys = grouping_columns(words)
115 if not keys:
116 return words
117 first = words[keys].iloc[0]
118 mask = pd.Series(True, index=words.index)
119 for key in keys:
120 mask &= words[key] == first[key]
121 return words[mask]
124def word_box_space_px(words: pd.DataFrame) -> float:
125 """Width of the trailing inter-word padding baked into each box, or ``0.0``.
127 Non-zero only for a monospaced, *tiling* layout whose boxes are consistently
128 ``(n_chars + 1)`` advances wide — the shape that carries the whole space as
129 trailing padding. Anything else (glyph-tight AOIs, proportional fonts,
130 missing text) reports 0.0, i.e. "every box is its glyph run".
132 It never moves a box edge (BUG-83): it only tells the within-word accessors
133 — :func:`word_char_advance` and :func:`word_glyph_span` — how many character
134 cells a box holds.
135 """
136 needed = {"x", "width", "text"}
137 if words is None or words.empty or not needed <= set(words.columns):
138 return 0.0
139 sample = _sample_trial_words(words)
140 x = pd.to_numeric(sample["x"], errors="coerce")
141 width = pd.to_numeric(sample["width"], errors="coerce")
142 chars = sample["text"].astype(str).str.len()
143 ok = x.notna() & width.notna() & (chars > 0) & (width > 0)
144 if ok.sum() < 3:
145 return 0.0
146 # One advance per character plus exactly one for the trailing space.
147 advance = float((width[ok] / (chars[ok] + 1)).median())
148 if advance <= 0:
149 return 0.0
150 residual = (width[ok] - (chars[ok] + 1) * advance).abs()
151 if (residual <= _ADVANCE_RESIDUAL_TOL_PX).mean() < _ADVANCE_MIN_AGREEMENT:
152 return 0.0 # not monospaced-with-one-trailing-space
153 # And the boxes must actually tile: a layout with real gaps is glyph-tight,
154 # so its boundaries already sit in the whitespace.
155 lines = cluster_word_lines(sample)
156 for _, line_words in sample[ok].groupby(lines[ok], sort=False):
157 ordered = line_words.sort_values("x")
158 if len(ordered) < 2:
159 continue
160 left = pd.to_numeric(ordered["x"], errors="coerce").to_numpy()
161 right = left + pd.to_numeric(ordered["width"], errors="coerce").to_numpy()
162 gaps = left[1:] - right[:-1]
163 if len(gaps) and np.nanmax(np.abs(gaps)) > _TILING_GAP_TOL_PX:
164 return 0.0
165 return advance
168def word_box_bounds(
169 words: pd.DataFrame,
170) -> tuple[np.ndarray, np.ndarray, np.ndarray, np.ndarray]:
171 """``(x0, y0, x1, y1)`` interest-area edges, exactly as the data defines them.
173 **The** accessor for word-box geometry: anything that tests a point against a
174 box or draws one goes through here — fixation→word assignment, the in-text
175 flag, the drawn outlines, the word heatmaps, the critical-span frame, drift
176 correction and the model scanpaths — so there is one definition of where a
177 word's interest area is. It is ``x .. x + width`` by ``y .. y + height``,
178 the experiment's own rectangles (BUG-83). On a tiling corpus that includes
179 the space after the word, which is what EyeLink's reports assume.
181 Pure: it reads ``x``/``y``/``width``/``height`` and returns arrays.
183 **Not** the frame for a position *inside* a word: a tiling box's right-hand
184 cell is a space, so letters are counted from ``x`` in
185 :func:`word_char_advance` units, and :func:`word_glyph_span` says where the
186 glyphs are.
187 """
188 if words is None or words.empty:
189 # A column-less empty frame is a legitimate "no words" input.
190 empty = np.empty(0, dtype=float)
191 return empty, empty.copy(), empty.copy(), empty.copy()
192 x = pd.to_numeric(words["x"], errors="coerce").to_numpy(dtype=float)
193 y = pd.to_numeric(words["y"], errors="coerce").to_numpy(dtype=float)
194 w = pd.to_numeric(words["width"], errors="coerce").to_numpy(dtype=float)
195 h = pd.to_numeric(words["height"], errors="coerce").to_numpy(dtype=float)
196 return x, y, x + w, y + h
199def word_box_contains(
200 px: np.ndarray,
201 py: np.ndarray,
202 x0: np.ndarray,
203 y0: np.ndarray,
204 x1: np.ndarray,
205 y1: np.ndarray,
206) -> np.ndarray:
207 """Is each point inside each box? **The** containment rule, broadcasting.
209 Half-open — ``x0 <= x < x1`` and ``y0 <= y < y1`` — so a box ``width`` px
210 wide holds exactly ``width`` pixel columns, and a point on the edge two
211 tiling boxes share belongs to the box that *starts* there: the word to the
212 right, the line below. That is how EyeLink assigned every such fixation in
213 the bundled demo (30 of them, all integer coordinates on a shared edge),
214 where a closed test handed each to the earlier box instead (BUG-83).
215 Assignment, the out-of-text flag and the word heatmap all test with this,
216 so a fixation is counted towards exactly one word.
217 """
218 return (px >= x0) & (px < x1) & (py >= y0) & (py < y1)
221def word_char_advance(
222 words: pd.DataFrame,
223 *,
224 layout: pd.DataFrame | None = None,
225 chars: np.ndarray | None = None,
226) -> np.ndarray:
227 """Width of one character in each word, in px — the *within-word* scale.
229 **The** accessor for anything that measures a position *inside* a word in
230 letters (initial landing position, a saccade's launch/landing letter), the
231 way :func:`word_box_bounds` is the accessor for the boundary between words.
233 ``width / len(text)`` is wrong on a tiling corpus (BUG-27, found by VAL-5):
234 it divides a box that is ``len(text) + 1`` advances wide by ``len(text)``,
235 so every letter was reported ~``(n+1)/n`` too wide and every landing
236 position that far into the word. The fix is :func:`word_box_space_px`'s
237 detection applied to the denominator — ``width / (len(text) + 1)`` for a
238 tiling layout, ``width / len(text)`` for a glyph-tight one.
240 The *origin* of a within-word position is the word's ``x``, which on every
241 layout this app recognises is both where the box starts and where the first
242 glyph does. On a tiling box the last cell, ``[x + n × advance, x + width)``,
243 is the space after the word — letter ``n + 1`` of the interest area.
245 Pass ``layout`` — the full trial — when ``words`` is a *subset* (a
246 highlighted span, one word): tiling is a property of the whole line, and a
247 subset's holes read as glyph-tight gaps.
249 ``chars`` lets a caller that already counted the characters hand them in —
250 ``.astype(str).str.len()`` is a Python-level pass over the frame, and
251 :func:`word_glyph_span` needs the same counts for the glyph run.
252 """
253 if words is None or words.empty or not {"width", "text"} <= set(words.columns):
254 # NaN, never `np.empty` — that returns *uninitialized* floats, and a
255 # garbage positive one passes the `isfinite and > 0` guard at the call
256 # site and lands a silently wrong landing position in a cached frame.
257 return np.full(0 if words is None else len(words), np.nan, dtype=float)
258 width = pd.to_numeric(words["width"], errors="coerce").to_numpy(dtype=float)
259 counts = word_char_counts(words) if chars is None else chars
260 padded = word_box_space_px(words if layout is None else layout) > 0
261 return width / (counts + (1.0 if padded else 0.0))
264def word_char_counts(words: pd.DataFrame) -> np.ndarray:
265 """Characters per word, floor 1 — the companion of :func:`word_char_advance`."""
266 return words["text"].astype(str).str.len().clip(lower=1).to_numpy(dtype=float)
269def word_glyph_span(
270 words: pd.DataFrame, *, layout: pd.DataFrame | None = None
271) -> tuple[np.ndarray, np.ndarray]:
272 """``(start, run)`` — where each word's glyphs begin, and how wide they are.
274 Not an interest area: ``agg.landing_positions`` measures a landing (and
275 mirrors an RTL one) across it. On a glyph-tight layout the run is the whole
276 box; on a tiling one it is ``len(text)`` :func:`word_char_advance` units,
277 one advance short of the box, whose last cell is the space after the word.
278 It assumes the glyphs start at ``x``. OneStop's experiment screens instead
279 centre each tiling box on its word (half a space either side), which is
280 why the drawn word label is centred in the box rather than placed on this
281 run (BUG-97).
283 Falls back to the box ``width`` for a frame with no ``text`` column, where
284 there are no letters to count. ``layout`` is as for
285 :func:`word_char_advance`.
286 """
287 if words is None or words.empty:
288 empty = np.empty(0, dtype=float)
289 return empty, empty.copy()
290 start = pd.to_numeric(words["x"], errors="coerce").to_numpy(dtype=float)
291 width = pd.to_numeric(words["width"], errors="coerce").to_numpy(dtype=float)
292 if "text" not in words.columns:
293 return start, width
294 chars = word_char_counts(words)
295 run = word_char_advance(words, layout=layout, chars=chars) * chars
296 return start, np.where(np.isfinite(run) & (run > 0), run, width)
299def assign_fixations_to_words(
300 fixations: pd.DataFrame,
301 words: pd.DataFrame,
302 *,
303 overwrite: bool = False,
304) -> pd.DataFrame:
305 """Give each fixation the word it belongs to.
307 When the fixations carry a ``word_id`` (the user mapped one, e.g. EyeLink's
308 ``CURRENT_FIX_INTEREST_AREA_ID``), it is used exactly as given — a blank
309 stays blank, since the data's own "no word" is an answer, not a gap — and
310 nothing is computed. Only a frame with no word ids at all (or
311 ``overwrite=True``) is assigned from geometry: the word box the fixation
312 falls in, else NaN.
314 Box edges come from :func:`word_box_bounds` — the experiment's own
315 rectangles (BUG-83) — so on a tiling corpus a fixation on the space after a
316 word is credited to that word, exactly as EyeLink's interest-area report
317 credits it.
318 """
319 if fixations.empty or words.empty:
320 return fixations
322 out = fixations.copy()
323 if "word_id" not in out.columns or overwrite:
324 out["word_id"] = np.nan
326 need_idx = out["word_id"].isna()
327 if not need_idx.all():
328 return out
330 # Per (participant, trial), do a fast vectorized box-test against that
331 # trial's words.
332 keys = grouping_columns(out)
333 if keys != grouping_columns(words):
334 raise ValueError(
335 "Screen ID is set in only one table; set it in both or neither."
336 )
337 groups = out[need_idx].groupby(keys, sort=False)
338 word_groups = words.groupby(keys, sort=False)
340 assignments = pd.Series(np.nan, index=out.index[need_idx], dtype=float)
341 for group_key, fix_chunk in groups:
342 lookup_key = group_key if len(keys) > 1 else group_key[0]
343 try:
344 wchunk = word_groups.get_group(lookup_key)
345 except KeyError:
346 continue
347 if wchunk.empty:
348 continue
349 assignments.loc[fix_chunk.index] = _assign_word_ids_single(fix_chunk, wchunk)
351 out.loc[need_idx, "word_id"] = assignments
352 return out
355def materialize_runs(fixations: pd.DataFrame) -> pd.DataFrame:
356 """Add trial run, line run, and per-word visit/pass columns (PRE-16)."""
357 if fixations.empty:
358 return fixations.copy()
359 derived = (
360 "run",
361 "linerun",
362 "word_runid",
363 "word_run",
364 "word_run_fix",
365 "nrun",
366 "reread",
367 )
368 out = fixations.drop(columns=[c for c in derived if c in fixations]).copy()
369 if "word_id" not in out:
370 out["word_id"] = pd.Series(pd.NA, index=out.index, dtype="Float64")
371 excluded = out.get("excluded", pd.Series(False, index=out.index)).fillna(False)
372 active = out.loc[~excluded.astype(bool)].copy()
373 for column in derived:
374 out[column] = pd.NA
375 if active.empty:
376 return out.sort_index()
377 keys = grouping_columns(active)
378 order_cols = keys + (["timestamp_ms"] if "timestamp_ms" in out else [])
379 active = active.sort_values(order_cols, kind="stable")
380 group = active.groupby(keys, sort=False, dropna=False) if keys else [((), active)]
381 pieces = []
382 for _, chunk in group:
383 chunk = chunk.copy()
384 word = pd.to_numeric(
385 chunk.get("word_id", pd.Series(pd.NA, index=chunk.index)),
386 errors="coerce",
387 )
388 line_source = (
389 chunk["line_id"]
390 if "line_id" in chunk
391 else chunk.get("line_idx", pd.Series(pd.NA, index=chunk.index))
392 )
393 line = pd.to_numeric(line_source, errors="coerce")
394 chunk["run"] = (word.diff().fillna(0) < 0).cumsum().astype("Int64") + 1
395 chunk["linerun"] = (
396 (line.ne(line.shift()) & line.notna()).cumsum().astype("Int64")
397 )
398 visit_start = word.ne(word.shift()) | word.isna()
399 chunk["word_runid"] = visit_start.cumsum().astype("Int64")
400 valid = chunk[word.notna()].copy()
401 if not valid.empty:
402 visits = valid[["word_id", "word_runid"]].drop_duplicates()
403 visits["word_run"] = visits.groupby("word_id", sort=False).cumcount() + 1
404 lookup = visits.set_index(["word_id", "word_runid"])["word_run"]
405 visit_keys = pd.MultiIndex.from_frame(chunk[["word_id", "word_runid"]])
406 chunk["word_run"] = lookup.reindex(visit_keys).to_numpy()
407 chunk["word_run_fix"] = (
408 chunk.groupby(["word_id", "word_runid"], dropna=False).cumcount() + 1
409 )
410 else:
411 chunk["word_run"] = pd.NA
412 chunk["word_run_fix"] = pd.NA
413 chunk["nrun"] = chunk.groupby("word_id", dropna=False)["word_runid"].transform(
414 "nunique"
415 )
416 chunk["reread"] = pd.to_numeric(chunk["word_run"], errors="coerce").gt(1)
417 pieces.append(chunk)
418 computed = pd.concat(pieces).sort_index() if pieces else active
419 for column in derived:
420 out.loc[computed.index, column] = computed[column]
421 return out.sort_index()
424def enrich_fixations(fixations: pd.DataFrame, words: pd.DataFrame) -> pd.DataFrame:
425 """Add saccade_amplitude, progression, and is_regression to fixations."""
426 if fixations.empty:
427 return fixations
428 out = fixations.copy()
429 keys = grouping_columns(out)
430 out = out.sort_values(keys + ["timestamp_ms"])
432 g = out.groupby(keys, sort=False)
433 dx = g["x"].diff()
434 dy = g["y"].diff()
435 if "saccade_amplitude" not in out.columns:
436 out["saccade_amplitude"] = np.sqrt(dx * dx + dy * dy)
437 else:
438 out["saccade_amplitude"] = out["saccade_amplitude"].where(
439 out["saccade_amplitude"].notna(), np.sqrt(dx * dx + dy * dy)
440 )
441 incoming = np.degrees(np.arctan2(-dy, dx))
442 out["angle_incoming"] = incoming
443 g = out.groupby(keys, sort=False)
444 out["angle_outgoing"] = g["angle_incoming"].shift(-1)
446 next_word = g["word_id"].shift(-1)
447 out["progression"] = np.sign(next_word - out["word_id"]).fillna(0).astype(int)
449 running_max = g["word_id"].cummax()
450 out["is_regression"] = (out["word_id"] < running_max).fillna(False).astype(bool)
451 return materialize_runs(out)
454def classify_saccades(fixations: pd.DataFrame, words: pd.DataFrame) -> pd.Series:
455 """Classify each *outgoing* saccade by its reading type (VIZ-8).
457 A saccade is the segment from one fixation to the next in reading order; its
458 class is stored on the *departing* fixation, so the last fixation of every
459 trial (which has no outgoing saccade) is ``None``. Classes follow the classic
460 reading schematic:
462 - ``refixation`` — lands back on the same word (``word_id`` unchanged),
463 - ``regression`` — moves to an earlier line, or backward within a line,
464 - ``return_sweep`` — sweeps down to a later line (forward reading),
465 - ``forward`` — advances to the next word on the same line,
466 - ``skip`` — advances past one or more words on the same line,
467 - ``other`` — a real saccade whose endpoints can't be classified (an
468 out-of-text fixation with no assigned word).
470 Line membership comes from :func:`assign_fixation_lines` (word-box geometry),
471 word order from the fixation ``word_id`` assignment. Returns an object Series
472 aligned to ``fixations.index`` (pure — no plotting), so both the render path
473 (``plots._add_saccade_layer``) and any future analysis can reuse it. Works on
474 a single already-sliced trial or a multi-trial frame (grouped by
475 participant/trial).
476 """
477 if fixations.empty:
478 return pd.Series([], dtype=object, index=fixations.index)
480 keys = grouping_columns(fixations)
481 sort_cols = keys + (["timestamp_ms"] if "timestamp_ms" in fixations.columns else [])
482 order = fixations.sort_values(sort_cols).index if sort_cols else fixations.index
484 if "word_id" in fixations.columns:
485 word = pd.to_numeric(fixations["word_id"], errors="coerce").reindex(order)
486 else:
487 word = pd.Series(np.nan, index=order, dtype=float)
488 if words is not None and not words.empty:
489 line = assign_fixation_lines(fixations, words).reindex(order)
490 else:
491 line = pd.Series(np.nan, index=order, dtype=float)
493 if keys:
494 grp = [fixations[k].reindex(order) for k in keys]
495 next_word = word.groupby(grp).shift(-1)
496 next_line = line.groupby(grp).shift(-1)
497 next_pos = pd.Series(np.arange(len(order)), index=order).groupby(grp).shift(-1)
498 else:
499 next_word = word.shift(-1)
500 next_line = line.shift(-1)
501 next_pos = pd.Series(np.arange(len(order)), index=order).shift(-1)
503 word_arr = word.to_numpy(dtype=float)
504 next_word_arr = next_word.to_numpy(dtype=float)
505 dw = next_word_arr - word_arr
506 dl = (next_line - line).to_numpy(dtype=float)
507 has_next = next_pos.notna().to_numpy()
508 # A saccade is classifiable only when BOTH endpoints have an assigned word;
509 # otherwise the word/line deltas are meaningless (an off-text fixation still
510 # gets a *nearest* line, which would spuriously read as a line change).
511 both_words = ~np.isnan(word_arr) & ~np.isnan(next_word_arr)
512 same_line = np.isnan(dl) | (dl == 0)
514 # Priority order: same word first, then line moves, then within-line word moves.
515 conds = [
516 dw == 0, # refixation
517 dl < 0, # regression — up to an earlier line
518 dl > 0, # return sweep — down to a later line
519 same_line & (dw < 0), # regression — backward within the line
520 same_line & (dw == 1), # forward
521 same_line & (dw >= 2), # skip
522 ]
523 choices = [
524 "refixation",
525 "regression",
526 "return_sweep",
527 "regression",
528 "forward",
529 "skip",
530 ]
531 classed = np.select(conds, choices, default="other").astype(object)
532 classed[~both_words] = "other" # a real saccade, but an endpoint is off-text
533 classed[~has_next] = None # last fixation of a trial: no outgoing saccade
534 # dtype=object is load-bearing: pandas 3.0 infers a dedicated `str` dtype for
535 # an all-string+None array and coerces the None sentinel to NaN, breaking the
536 # "None for the last fixation" contract. Pin object so None survives.
537 return pd.Series(classed, index=order, dtype=object).reindex(fixations.index)
540def rebased_fixation_onsets(ordered_fixations: pd.DataFrame) -> np.ndarray:
541 """Fixation onset times (ms), rebased so the first fixation is t=0.
543 ``ordered_fixations`` must already be in reading order (sorted by
544 ``timestamp_ms``); the returned array is aligned to its rows. Uses the
545 recorded ``timestamp_ms`` when they look like real times — their span is at
546 least ``REAL_TIMESTAMP_DWELL_FRAC`` of the summed durations — otherwise lays
547 fixations back-to-back by their durations, so a synthesised 0,1,2,… index
548 doesn't crush the time axis. Fixations normalization numbered because the
549 table had no onset (``data.TIMESTAMP_SYNTHESIZED``) always take the
550 durations — the same estimate the reading summaries use. Shared by the similarity time-curve
551 (:func:`scanpath_studio.similarity._rebased_onsets`) and the animation clock
552 (:func:`scanpath_studio.plots._scanpath_anim_specs`).
553 """
554 if ordered_fixations.empty:
555 return np.array([], dtype=float)
556 if "duration_ms" in ordered_fixations.columns:
557 dur = (
558 pd.to_numeric(ordered_fixations["duration_ms"], errors="coerce")
559 .fillna(0)
560 .to_numpy(dtype=float)
561 )
562 else:
563 dur = np.zeros(len(ordered_fixations))
564 contiguous = np.concatenate(([0.0], np.cumsum(dur)[:-1])) if len(dur) else dur
565 from .data import timestamps_synthesized
567 if timestamps_synthesized(ordered_fixations):
568 return contiguous
569 if "timestamp_ms" in ordered_fixations.columns:
570 ts = pd.to_numeric(ordered_fixations["timestamp_ms"], errors="coerce").to_numpy(
571 dtype=float
572 )
573 total_dwell = float(dur.sum()) if len(dur) else 0.0
574 if (
575 len(ts)
576 and not np.isnan(ts).any()
577 and (ts[-1] - ts[0]) >= REAL_TIMESTAMP_DWELL_FRAC * total_dwell
578 ):
579 return ts - ts[0]
580 return contiguous
583# ---------------------------------------------------------------------------
584# Geometry helpers: line clustering, in-text test, fixation -> line.
585#
586# These power the "highlight out-of-text fixations" and "color fixations by
587# line" plot options. They are deliberately pure (no Streamlit, no plotting)
588# so they can be unit-tested against a known synthetic layout.
589# ---------------------------------------------------------------------------
592def _in_any_box(fix_chunk: pd.DataFrame, word_chunk: pd.DataFrame) -> np.ndarray:
593 """Boolean array (aligned to fix_chunk order): is each fixation inside any box?"""
594 x0, y0, x1, y1 = word_box_bounds(word_chunk)
595 fx = pd.to_numeric(fix_chunk["x"], errors="coerce").to_numpy(dtype=float)
596 fy = pd.to_numeric(fix_chunk["y"], errors="coerce").to_numpy(dtype=float)
597 inside = word_box_contains(
598 fx[:, None], fy[:, None], x0[None, :], y0[None, :], x1[None, :], y1[None, :]
599 )
600 return inside.any(axis=1)
603def cluster_word_lines(words: pd.DataFrame, tol_frac: float = 0.5) -> pd.Series:
604 """Assign each word a 0-based line id by clustering on vertical position.
606 OneStop IA exports rarely carry a real per-word line number (the
607 normalized ``line_idx`` is often a constant), so we infer visual lines from
608 word-box geometry: words are sorted by vertical center and a new line
609 starts whenever the center jumps by more than ``tol_frac`` of the median
610 word height. Lines are numbered top-to-bottom starting at 0.
612 Returns an int Series aligned to ``words.index`` (empty when ``words`` is
613 empty). Mirrors the line clustering used by ``plots.build_critical_span_overlay``.
614 """
615 if words.empty:
616 return pd.Series([], dtype="int64", index=words.index)
617 heights = pd.to_numeric(words["height"], errors="coerce")
618 typical_h = float(heights.median()) if heights.notna().any() else 1.0
619 typical_h = typical_h if typical_h > 0 else 1.0
620 y_center = pd.to_numeric(words["y"], errors="coerce") + heights.fillna(0) / 2.0
621 order = y_center.sort_values(kind="stable")
622 line_of_sorted = (
623 (order.diff().fillna(0) > typical_h * tol_frac).cumsum().astype(int)
624 )
625 return line_of_sorted.reindex(words.index)
628def _groupwise(fixations: pd.DataFrame, words: pd.DataFrame, fn) -> pd.Series:
629 """Apply ``fn(fix_chunk, word_chunk)`` per (participant, trial), aligning
630 the per-chunk result back onto a Series indexed like ``fixations``.
632 Falls back to a single group when the id columns are absent (e.g. the
633 figure builders pass already-sliced single-trial frames)."""
634 keys = grouping_columns(fixations)
635 # object dtype so a chunk fn may return bools (in-text) or floats (line ids)
636 # without triggering a dtype-incompatibility cast; callers coerce the result.
637 out = pd.Series(np.nan, index=fixations.index, dtype=object)
638 has_groups = bool(keys) and keys == grouping_columns(words)
639 if not has_groups:
640 out.loc[:] = fn(fixations, words)
641 return out
642 word_groups = words.groupby(keys, sort=False)
643 for key, fix_chunk in fixations.groupby(keys, sort=False):
644 try:
645 word_chunk = word_groups.get_group(key)
646 except KeyError:
647 continue
648 if word_chunk.empty:
649 continue
650 out.loc[fix_chunk.index] = fn(fix_chunk, word_chunk)
651 return out
654def fixation_in_text_mask(fixations: pd.DataFrame, words: pd.DataFrame) -> pd.Series:
655 """Boolean Series: True where a fixation falls inside any word box.
657 "Out-of-text" fixations are simply ``~fixation_in_text_mask(...)``. Works
658 on multi-trial frames (grouped by participant/trial) or on a single
659 already-sliced trial. Fixations with non-finite coordinates count as
660 out-of-text (mask = False)."""
661 if fixations.empty:
662 return pd.Series([], dtype=bool, index=fixations.index)
663 if words.empty:
664 return pd.Series(False, index=fixations.index)
665 res = _groupwise(fixations, words, _in_any_box)
666 return res.where(res.notna(), other=False).astype(bool)
669def assign_fixation_lines(fixations: pd.DataFrame, words: pd.DataFrame) -> pd.Series:
670 """Assign each fixation the 0-based line id of the nearest text line.
672 Lines are derived from word geometry via :func:`cluster_word_lines`; each
673 fixation is mapped to the line whose mean vertical center is closest to the
674 fixation's y. Returns a float Series (NaN where unmappable) aligned to
675 ``fixations.index`` so it can be used as a categorical color field."""
676 if fixations.empty:
677 return pd.Series([], dtype="float64", index=fixations.index)
678 if words.empty:
679 return pd.Series(np.nan, index=fixations.index, dtype="float64")
681 def _nearest_line(fix_chunk: pd.DataFrame, word_chunk: pd.DataFrame) -> np.ndarray:
682 lines = cluster_word_lines(word_chunk)
683 y_center = (
684 pd.to_numeric(word_chunk["y"], errors="coerce")
685 + pd.to_numeric(word_chunk["height"], errors="coerce").fillna(0) / 2.0
686 )
687 centers = y_center.groupby(lines).mean()
688 line_ids = centers.index.to_numpy(dtype=float)
689 line_cy = centers.to_numpy(dtype=float)
690 fy = pd.to_numeric(fix_chunk["y"], errors="coerce").to_numpy(dtype=float)
691 dist = np.abs(fy[:, None] - line_cy[None, :])
692 nearest = np.where(np.isnan(fy), -1, dist.argmin(axis=1))
693 result = np.where(nearest >= 0, line_ids[np.clip(nearest, 0, None)], np.nan)
694 return result
696 return pd.to_numeric(_groupwise(fixations, words, _nearest_line), errors="coerce")
699def compute_per_word_measures(
700 fixations: pd.DataFrame, words: pd.DataFrame
701) -> pd.DataFrame:
702 """Compute canonical reading measures per word.
704 Returns a copy of `words` with computed columns added. Existing values on
705 `words` (e.g. EyeLink IA metrics) take precedence over computed ones.
706 """
707 if words.empty:
708 return words.copy()
710 enriched = (
711 enrich_fixations(assign_fixations_to_words(fixations, words), words)
712 if not fixations.empty
713 else fixations
714 )
716 out = words.copy()
717 key_cols = grouping_columns(words, include_word=True)
718 # VAL-5: the letter scale for the landing measures, resolved once on the
719 # whole frame (tiling is a property of the layout, so it must not be
720 # re-detected per trial word, which is a subset with holes in it).
721 char_advance = pd.Series(word_char_advance(words), index=words.index)
723 # Initialize defaults
724 computed = (
725 words[key_cols]
726 .drop_duplicates()
727 .assign(
728 _first_fixation_ms=np.nan,
729 _first_pass_gaze_ms=np.nan,
730 _regression_path_ms=np.nan,
731 _total_fixation_ms=0.0,
732 _n_fixations=0,
733 _skip_flag=True,
734 _regression_in_flag=False,
735 _regression_out_flag=False,
736 _first_fix_x=np.nan,
737 _first_fix_y=np.nan,
738 _initial_landing_position=np.nan,
739 _initial_landing_distance=np.nan,
740 _number_of_regressions_in=0,
741 _second_pass_duration=0.0,
742 _single_fixation_duration=np.nan,
743 )
744 )
746 # PERF-12: each trial's word rows, found once. The landing measures below
747 # used to locate a fixated word by masking the *whole* words frame — one
748 # full-frame comparison per fixated word, per identity column — which made
749 # this function quadratic in corpus size (1.7 s on the demo, 22 s at 8×).
750 word_num = pd.to_numeric(words["word_id"], errors="coerce").to_numpy(dtype=float)
751 identity_cols = grouping_columns(enriched) if not enriched.empty else []
752 word_rows: dict[tuple, np.ndarray] = {}
753 if identity_cols and set(identity_cols) <= set(words.columns):
754 grouped = words.groupby(identity_cols, sort=False).indices
755 word_rows = {
756 (k if isinstance(k, tuple) else (k,)): v for k, v in grouped.items()
757 }
759 analysis_fixations = enriched
760 if "excluded" in enriched.columns:
761 analysis_fixations = enriched[~enriched["excluded"].fillna(False).astype(bool)]
762 if not analysis_fixations.empty and "word_id" in analysis_fixations.columns:
763 per_word_rows = []
764 # Group fixations by trial to walk them in temporal order.
765 trial_keys = grouping_columns(analysis_fixations)
766 for group_key, trial_chunk in analysis_fixations.groupby(
767 trial_keys, sort=False
768 ):
769 values = group_key if isinstance(group_key, tuple) else (group_key,)
770 identity = dict(zip(trial_keys, values))
771 # BUG-66: the walk below sees the off-text fixations too — one lands
772 # between two runs, so it must end the first of them, exactly as
773 # `materialize_runs` numbers `word_run` (and so second-pass). Walking
774 # only the in-text rows glued the two runs into one first pass, and a
775 # word's first + second pass then added up to more than its total.
776 trial_chunk = trial_chunk.sort_values("timestamp_ms")
777 fix_chunk = trial_chunk.dropna(subset=["word_id"])
778 if fix_chunk.empty:
779 continue
780 # Total / n / first-fixation are per-word aggregations
781 grp = fix_chunk.groupby("word_id")
782 tot = grp["duration_ms"].sum()
783 n = grp.size()
784 ffd = grp["duration_ms"].first()
785 ffx = grp["x"].first()
786 ffy = grp["y"].first()
787 second_pass = (
788 fix_chunk[
789 pd.to_numeric(fix_chunk.get("word_run"), errors="coerce").eq(2)
790 ]
791 .groupby("word_id")["duration_ms"]
792 .sum()
793 )
794 first_pass_counts = (
795 fix_chunk[
796 pd.to_numeric(fix_chunk.get("word_run"), errors="coerce").eq(1)
797 ]
798 .groupby("word_id")
799 .size()
800 )
802 # One walk of the trial in time order. The definitions are EyeLink's
803 # IA_* ones, so a computed measure means the same as an imported one
804 # (imported values take precedence, and the two must not disagree
805 # about what a column is): validated against the bundled OneStop IA
806 # report, which EyeLink Data Viewer produced from these fixations.
807 first_run: dict[float, float] = {} # IA_FIRST_RUN_DWELL_TIME
808 first_pass: set = set() # not IA_SKIP
809 regression_path: dict[float, float] = {}
810 regression_in: set = set()
811 regression_out: set = set()
812 regression_in_count: dict[float, int] = {}
814 running_max = -np.inf
815 run_word: float | None = None
816 run_duration = 0.0
817 seen: set = set()
818 go_past_open: dict[float, float] = {}
819 left_forward: set = set()
821 prev_word: float | None = None
822 for row in trial_chunk.itertuples():
823 dur = float(row.duration_ms)
824 if pd.isna(row.word_id):
825 # Outside every box: ends the current run (BUG-66), but
826 # opens, extends and closes no go-past window.
827 if run_word is not None:
828 first_run.setdefault(run_word, run_duration)
829 run_word = None
830 continue
831 w = float(row.word_id)
833 # First run: the word's first unbroken stretch of fixations,
834 # whenever it begins (as IA_FIRST_RUN_DWELL_TIME).
835 if run_word == w:
836 run_duration += dur
837 else:
838 if run_word is not None:
839 first_run.setdefault(run_word, run_duration)
840 run_word = w if w not in first_run else None
841 run_duration = dur
843 # BUG-62: first pass means entered *before any later word was
844 # fixated*. A word first reached by a regression was skipped,
845 # which is what IA_SKIP says and what `skip_flag` must say.
846 if w not in seen and w > running_max:
847 first_pass.add(w)
849 # BUG-61: go-past (regression-path) time runs from the word's
850 # first fixation until the first fixation on a later word, and
851 # EVERY fixation in between counts — a first visit to a skipped
852 # earlier word during the regression included. Adding only
853 # revisits dropped those, so RPD always came out short.
854 for k in go_past_open:
855 if w <= k:
856 go_past_open[k] += dur
857 if w not in seen:
858 go_past_open[w] = dur
859 for k in [k for k in go_past_open if w > k]:
860 regression_path[k] = go_past_open.pop(k)
861 seen.add(w)
863 if prev_word is not None and w != prev_word:
864 if w < prev_word:
865 regression_in.add(w)
866 regression_in_count[w] = regression_in_count.get(w, 0) + 1
867 # BUG-64: a regression *out* is a first-pass event
868 # (IA_REGRESSION_OUT) — made from a word read in first
869 # pass, before the eyes first left it forwards — not a
870 # regression from it at any later time.
871 if prev_word in first_pass and prev_word not in left_forward:
872 regression_out.add(prev_word)
873 else:
874 left_forward.add(prev_word)
876 running_max = max(running_max, w)
877 prev_word = w
879 if run_word is not None:
880 first_run.setdefault(run_word, run_duration)
881 # Windows still open at the end: the reader never moved past them.
882 regression_path.update(go_past_open)
884 trial_rows = word_rows.get(tuple(values))
885 first_row: dict[float, int] = {}
886 if trial_rows is not None:
887 for pos in trial_rows:
888 first_row.setdefault(word_num[pos], pos)
889 for w in tot.index:
890 landing_position = landing_distance = np.nan
891 pos = first_row.get(float(w))
892 if pos is not None:
893 target = words.iloc[pos]
894 text_len = max(len(str(target.get("text", ""))), 1)
895 width = float(pd.to_numeric(target.get("width"), errors="coerce"))
896 char_width = float(char_advance.iloc[pos])
897 if np.isfinite(char_width) and char_width > 0 and width > 0:
898 rtl = bool(target.get("right_to_left", False))
899 # BUG-27: RTL counts from where the glyphs *end*
900 # (`x + n × advance`), not from the padded box edge
901 # `x + width` — on a tiling layout those are one whole
902 # advance apart, so an RTL landing read a letter late
903 # and disagreed with `aggregation.landing_positions`,
904 # which measures across the glyph run on both sides.
905 offset = (
906 float(target.get("x"))
907 + text_len * char_width
908 - float(ffx.loc[w])
909 if rtl
910 else float(ffx.loc[w]) - float(target.get("x"))
911 )
912 # Unclipped. The glyphs span [1, n + 1); on a tiling
913 # corpus the box's last cell, [n + 1, n + 2), is the
914 # space after the word, which belongs to it (BUG-83), so
915 # a first fixation there reads letter n + 1 — the AOI's
916 # own last cell — rather than being folded onto letter n.
917 landing_position = offset / char_width + 1.0
918 # BUG-65: the word's centre is its glyphs' centre,
919 # 1 + n / 2 — not (n + 1) / 2, which read every landing
920 # half a letter right of centre, and not the box's
921 # centre, which on a tiling box sits half a space
922 # further right again.
923 landing_distance = landing_position - (1.0 + text_len / 2.0)
924 per_word_rows.append(
925 dict(
926 **identity,
927 word_id=w,
928 _first_fixation_ms=float(ffd.loc[w]),
929 _first_pass_gaze_ms=float(first_run.get(w, np.nan)),
930 _regression_path_ms=float(regression_path.get(w, np.nan)),
931 _total_fixation_ms=float(tot.loc[w]),
932 _n_fixations=int(n.loc[w]),
933 _skip_flag=w not in first_pass,
934 _regression_in_flag=w in regression_in,
935 _regression_out_flag=w in regression_out,
936 _first_fix_x=float(ffx.loc[w]),
937 _first_fix_y=float(ffy.loc[w]),
938 _initial_landing_position=landing_position,
939 _initial_landing_distance=landing_distance,
940 _number_of_regressions_in=int(regression_in_count.get(w, 0)),
941 _second_pass_duration=float(second_pass.get(w, 0.0)),
942 _single_fixation_duration=(
943 float(ffd.loc[w])
944 if int(first_pass_counts.get(w, 0)) == 1
945 else np.nan
946 ),
947 )
948 )
950 if per_word_rows:
951 new_df = pd.DataFrame(per_word_rows)
952 updated = computed.merge(
953 new_df, on=key_cols, how="left", suffixes=("_default", "")
954 )
955 for col in new_df.columns:
956 if col in key_cols:
957 continue
958 default_col = f"{col}_default"
959 if default_col in updated.columns:
960 updated[col] = updated[col].where(
961 updated[col].notna(), updated[default_col]
962 )
963 updated = updated.drop(columns=default_col)
964 computed = updated
966 out = out.merge(computed, on=key_cols, how="left")
968 # Map computed -> canonical name, keeping any existing value.
969 rename_map = {
970 "_first_fixation_ms": "first_fixation_ms",
971 "_first_pass_gaze_ms": "first_pass_gaze_duration_ms",
972 "_regression_path_ms": "regression_path_duration_ms",
973 "_total_fixation_ms": "total_fixation_duration_ms",
974 "_n_fixations": "n_fixations",
975 "_skip_flag": "skip_flag",
976 "_regression_in_flag": "regression_in_flag",
977 "_regression_out_flag": "regression_out_flag",
978 "_first_fix_x": "first_fix_x",
979 "_first_fix_y": "first_fix_y",
980 "_initial_landing_position": "initial_landing_position",
981 "_initial_landing_distance": "initial_landing_distance",
982 "_number_of_regressions_in": "number_of_regressions_in",
983 "_second_pass_duration": "second_pass_duration_ms",
984 "_single_fixation_duration": "single_fixation_duration_ms",
985 }
986 for src, dst in rename_map.items():
987 if src not in out.columns:
988 continue
989 if dst in out.columns:
990 out[dst] = out[dst].where(out[dst].notna(), out[src])
991 else:
992 out[dst] = out[src]
993 out = out.drop(columns=src)
995 # BUG-63: a word nobody fixated has no first fixation, first pass or go-past
996 # time — but an imported IA report may say `0` rather than leave the cell
997 # empty (the bundled OneStop one does, for every such word), and a 0 wins the
998 # precedence above. Every mean then counted skipped words as 0-ms fixations.
999 # Total time stays 0 on purpose: the word was read past and got no time.
1000 if "n_fixations" in out.columns:
1001 unfixated = pd.to_numeric(out["n_fixations"], errors="coerce").eq(0)
1002 for col in (
1003 "first_fixation_ms",
1004 "first_pass_gaze_duration_ms",
1005 "regression_path_duration_ms",
1006 "single_fixation_duration_ms",
1007 ):
1008 if col in out.columns and unfixated.any():
1009 out[col] = pd.to_numeric(out[col], errors="coerce").mask(unfixated)
1010 # The mirror image for second pass, whose rule is "fewer than two runs
1011 # ⇒ 0": an imported IA_SECOND_RUN_DWELL_TIME leaves those cells empty,
1012 # so its mean covered only re-read words while the computed one covered
1013 # every word — one measure, two meanings, depending on the upload.
1014 if "second_pass_duration_ms" in out.columns:
1015 known = pd.to_numeric(out["n_fixations"], errors="coerce").notna()
1016 second = pd.to_numeric(out["second_pass_duration_ms"], errors="coerce")
1017 out["second_pass_duration_ms"] = second.mask(known & second.isna(), 0.0)
1019 # Canonical aliases used elsewhere in the app
1020 if (
1021 "first_pass_gaze_duration_ms" in out.columns
1022 and "gaze_duration_ms" not in out.columns
1023 ):
1024 out["gaze_duration_ms"] = out["first_pass_gaze_duration_ms"]
1026 # Ensure dtypes
1027 for col in ["n_fixations"]:
1028 if col in out.columns:
1029 out[col] = (
1030 pd.to_numeric(out[col], errors="coerce").fillna(0).astype("Int64")
1031 )
1032 for col in ["skip_flag", "regression_in_flag", "regression_out_flag"]:
1033 if col in out.columns:
1034 out[col] = (
1035 out[col].astype(object).where(out[col].notna(), False).astype(bool)
1036 )
1038 return out