Coverage for scanpath_studio/aggregation.py: 96%
652 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Pure aggregation helpers for the Corpus Analysis subtabs.
3Headless (no Streamlit), so they're unit-testable. They turn the filtered
4words / fixations frames into the small summary tables the plot builders draw:
5the per-trial-index trend and per-text read counts first. The heavy work is
6plain pandas groupby; the tabs cache the results with ``@st.cache_data``.
8The lower block (from :data:`MEASURES` onward) backs the question-oriented
9analysis sections — *per text* (one text, many readers), *per reader* (one
10reader, many trials), *per group* (a cohort), and *group comparison* (two
11cohorts). Every section reads a single :class:`Measure` (the shared measure
12picker), an aggregation (mean/median/sum) and a spread (SD/IQR/SEM/bootstrap
13CI), and may z-score within reader — see :data:`MEASURES`,
14:func:`aggregate_value`, :func:`spread_bounds`, :func:`add_normalized_column`.
15"""
17from __future__ import annotations
19from collections.abc import Mapping, Sequence
20from dataclasses import dataclass
22import numpy as np
23import pandas as pd
25from .measures import classify_saccades
26from .multipart import SCREEN_ID, SCREEN_INDEX, grouping_columns, has_screen_identity
29def metric_by_trial_index(
30 frame: pd.DataFrame, metric: str, *, agg: str = "mean"
31) -> pd.DataFrame:
32 """Average of ``metric`` per trial index across all trials.
34 ``frame`` must carry a ``trial_index`` column (see
35 :func:`data.derive_trial_index`) and the ``metric`` column. Each trial
36 contributes its own per-trial aggregate (``agg``), then those are averaged
37 within each trial index. Returns ``DataFrame[trial_index, value, sem,
38 n_trials]`` sorted by trial index.
39 """
40 cols = {"participant_id", "trial_id", "trial_index", metric}
41 if frame.empty or not cols <= set(frame.columns):
42 return pd.DataFrame(columns=["trial_index", "value", "sem", "n_trials"])
43 df = frame[["participant_id", "trial_id", "trial_index"]].copy()
44 df["_m"] = pd.to_numeric(frame[metric], errors="coerce")
45 df = df.dropna(subset=["trial_index", "_m"])
46 if df.empty:
47 return pd.DataFrame(columns=["trial_index", "value", "sem", "n_trials"])
48 per_trial = (
49 df.groupby(["participant_id", "trial_id", "trial_index"])["_m"]
50 .agg(agg)
51 .reset_index()
52 )
53 out = (
54 per_trial.groupby("trial_index")["_m"]
55 .agg(["mean", "sem", "count"])
56 .reset_index()
57 )
58 out.columns = ["trial_index", "value", "sem", "n_trials"]
59 out["sem"] = out["sem"].fillna(0.0)
60 return out.sort_values("trial_index").reset_index(drop=True)
63def text_read_counts(words: pd.DataFrame, text_col: str) -> pd.DataFrame:
64 """Per-text participant counts: ``DataFrame[text, n_participants]`` sorted
65 by count desc — used to populate the per-text heatmap picker and annotate
66 sample sizes."""
67 if words.empty or text_col not in words.columns or "participant_id" not in words:
68 return pd.DataFrame(columns=["text", "n_participants"])
69 counts = words.groupby(text_col)["participant_id"].nunique().reset_index()
70 counts.columns = ["text", "n_participants"]
71 return counts.sort_values("n_participants", ascending=False).reset_index(drop=True)
74# =============================================================================
75# Question-oriented analysis sections (AN-1 … AN-28)
76# =============================================================================
77#
78# Everything below is pure pandas/numpy feeding ``plots.make_*`` builders, one
79# tidy DataFrame per view. The shared cross-cutting controls — measure picker
80# (AN-23), aggregation/spread (AN-24), within-reader normalization (AN-25), and
81# the min-observations guard (AN-26) — are expressed here so the figures and the
82# CSV downloads (AN-27) all read the same numbers.
85@dataclass(frozen=True)
86class Measure:
87 """One selectable eye-movement measure (the shared measure picker, AN-23).
89 ``frame`` is ``"words"`` (a per-word reading measure, keyed by ``word_id``)
90 or ``"fixations"`` (a per-fixation measure). ``per_word`` marks the
91 word-level measures that support the per-word *profile* views (AN-1/2/3/…);
92 ``is_rate`` marks boolean flags aggregated as a 0–1 rate (skip / regression).
93 """
95 key: str
96 label: str
97 frame: str # "words" | "fixations"
98 column: str
99 unit: str # axis-label unit, e.g. "ms", "px", "" (rates)
100 per_word: bool
101 is_rate: bool = False
103 @property
104 def axis_label(self) -> str:
105 return f"{self.label} ({self.unit})" if self.unit else self.label
108# Registry — insertion order is the picker order. TFD is the default.
109MEASURES: dict[str, Measure] = {
110 m.key: m
111 for m in (
112 Measure(
113 "tfd",
114 "Total fixation duration — TFD",
115 "words",
116 "total_fixation_duration_ms",
117 "ms",
118 True,
119 ),
120 Measure(
121 "ffd",
122 "First fixation duration — FFD",
123 "words",
124 "first_fixation_ms",
125 "ms",
126 True,
127 ),
128 Measure(
129 "fprt",
130 "First-pass reading time — FPRT",
131 "words",
132 "first_pass_gaze_duration_ms",
133 "ms",
134 True,
135 ),
136 Measure(
137 "rpd",
138 "Regression-path — RPD",
139 "words",
140 "regression_path_duration_ms",
141 "ms",
142 True,
143 ),
144 Measure("nfix", "Fixations per word", "words", "n_fixations", "", True),
145 Measure("skip", "Skip rate", "words", "skip_flag", "", True, is_rate=True),
146 Measure(
147 "reg_in",
148 "Regression-in rate",
149 "words",
150 "regression_in_flag",
151 "",
152 True,
153 is_rate=True,
154 ),
155 Measure(
156 "reg_out",
157 "Regression-out rate",
158 "words",
159 "regression_out_flag",
160 "",
161 True,
162 is_rate=True,
163 ),
164 Measure(
165 "landing_position",
166 "Initial landing position",
167 "words",
168 "initial_landing_position",
169 "letters",
170 True,
171 ),
172 Measure(
173 "landing_distance",
174 "Centered landing distance",
175 "words",
176 "initial_landing_distance",
177 "letters",
178 True,
179 ),
180 Measure(
181 "reg_in_count",
182 "Regressions into word",
183 "words",
184 "number_of_regressions_in",
185 "",
186 True,
187 ),
188 Measure(
189 "second_pass",
190 "Second-pass duration",
191 "words",
192 "second_pass_duration_ms",
193 "ms",
194 True,
195 ),
196 Measure(
197 "single_fix",
198 "Single-fixation duration",
199 "words",
200 "single_fixation_duration_ms",
201 "ms",
202 True,
203 ),
204 Measure(
205 "fix_dur", "Fixation duration", "fixations", "duration_ms", "ms", False
206 ),
207 # BUG-25: "px" is honest now — `saccade_amplitude` is always the pixel
208 # distance between consecutive fixations. EyeLink's degree-valued
209 # amplitudes normalize to `next_/prev_saccade_amplitude_deg` and never
210 # reach this measure, so the axis label can't disagree with the data.
211 Measure(
212 "sacc_amp",
213 "Saccade amplitude",
214 "fixations",
215 "saccade_amplitude",
216 "px",
217 False,
218 ),
219 )
220}
222# Bundled per-word linguistic features (AN-5). label → (column, is_categorical).
223LINGUISTIC_FEATURES: dict[str, tuple[str, bool]] = {
224 "GPT-2 surprisal": ("gpt2_surprisal", False),
225 "Word frequency (wordfreq)": ("wordfreq_frequency", False),
226 "Word length": ("word_length", False),
227 "Part of speech": ("universal_pos", True),
228}
231def available_measures(
232 words: pd.DataFrame, fixations: pd.DataFrame, *, per_word_only: bool = False
233) -> list[Measure]:
234 """The measures whose backing column is actually present in the data."""
235 out: list[Measure] = []
236 for m in MEASURES.values():
237 if per_word_only and not m.per_word:
238 continue
239 frame = words if m.frame == "words" else fixations
240 if frame is not None and m.column in getattr(frame, "columns", []):
241 out.append(m)
242 return out
245def available_features(words: pd.DataFrame) -> dict[str, tuple[str, bool]]:
246 """Linguistic features present in this words frame (AN-5)."""
247 if words is None or words.empty:
248 return {}
249 return {
250 label: spec
251 for label, spec in LINGUISTIC_FEATURES.items()
252 if spec[0] in words.columns
253 }
256# --- Aggregation + spread primitives (AN-24) ---------------------------------
258_AGG_FUNCS = {"mean": np.nanmean, "median": np.nanmedian, "sum": np.nansum}
261def aggregate_value(values: np.ndarray, agg: str = "mean") -> float:
262 """Collapse ``values`` to a single number under ``agg`` (mean/median/sum)."""
263 arr = np.asarray(values, dtype="float64")
264 arr = arr[~np.isnan(arr)]
265 if arr.size == 0:
266 return float("nan")
267 return float(_AGG_FUNCS.get(agg, np.nanmean)(arr))
270def bootstrap_ci(
271 values: np.ndarray,
272 *,
273 agg: str = "mean",
274 n_boot: int = 1000,
275 ci: float = 95.0,
276 seed: int = 0,
277) -> tuple[float, float]:
278 """Percentile bootstrap CI of the ``agg`` statistic (AN-24 spread option)."""
279 arr = np.asarray(values, dtype="float64")
280 arr = arr[~np.isnan(arr)]
281 if arr.size < 2:
282 v = aggregate_value(arr, agg)
283 return (v, v)
284 rng = np.random.default_rng(seed)
285 idx = rng.integers(0, arr.size, size=(n_boot, arr.size))
286 stats = _AGG_FUNCS.get(agg, np.nanmean)(arr[idx], axis=1)
287 lo = float(np.percentile(stats, (100.0 - ci) / 2.0))
288 hi = float(np.percentile(stats, 100.0 - (100.0 - ci) / 2.0))
289 return (lo, hi)
292def spread_bounds(values: np.ndarray, center: float, spread: str, *, agg: str = "mean"):
293 """``(lo, hi)`` band for ``values`` around ``center`` under ``spread``.
295 ``SD`` → ±1 std · ``SEM`` → ±std/√n · ``IQR`` → 25th/75th percentiles ·
296 ``Bootstrap CI`` → 95 % percentile bootstrap of the aggregate.
297 """
298 arr = np.asarray(values, dtype="float64")
299 arr = arr[~np.isnan(arr)]
300 if arr.size == 0 or np.isnan(center):
301 return (center, center)
302 if spread == "IQR":
303 return (float(np.percentile(arr, 25)), float(np.percentile(arr, 75)))
304 if spread == "Bootstrap CI":
305 return bootstrap_ci(arr, agg=agg)
306 if agg == "sum":
307 # SD/SEM describe the spread of individual observations, which is
308 # meaningless around a *total*; bootstrap the aggregate so the band
309 # actually brackets the plotted sum.
310 return bootstrap_ci(arr, agg=agg)
311 sd = float(np.std(arr, ddof=1)) if arr.size > 1 else 0.0
312 half = sd / np.sqrt(arr.size) if spread == "SEM" else sd
313 return (center - half, center + half)
316def add_normalized_column(
317 frame: pd.DataFrame,
318 col: str,
319 *,
320 by: str = "participant_id",
321 out_col: str | None = None,
322) -> pd.DataFrame:
323 """Return ``frame`` with ``col`` z-scored within each ``by`` group (AN-25).
325 Slow and fast readers then compare on *shape*, not absolute level. Groups of
326 one (or zero variance) map to 0. ``out_col`` defaults to overwriting ``col``.
327 """
328 out = frame.copy()
329 out_col = out_col or col
330 vals = pd.to_numeric(out[col], errors="coerce")
331 if by in out.columns:
332 grp = vals.groupby(out[by])
333 mean = grp.transform("mean")
334 std = grp.transform("std")
335 else:
336 mean = vals.mean()
337 std = vals.std()
338 z = (
339 (vals - mean) / std.replace(0, np.nan)
340 if hasattr(std, "replace")
341 else ((vals - mean) / (std or np.nan))
342 )
343 # `fillna(0.0)` alone conflates two different undefined-z cases: a
344 # zero-variance or singleton group (where 0 *is* the group mean, so 0 is
345 # right) and a genuinely missing observation, which would re-enter the
346 # distribution as an exactly-average data point. Keep the first, restore NaN
347 # for the second — so normalizing doesn't change how many observations there
348 # are (ENG-1).
349 out[out_col] = z.fillna(0.0).where(vals.notna())
350 return out
353#: Sentinel for "every screen" — only :func:`text_screen_options` passes it, to
354#: enumerate the screens *before* one has been picked. It is deliberately not
355#: part of the public helpers' contract: pooling across screens is the defect
356#: BUG-26 fixed, so there is no user-reachable way back to it.
357_ALL_SCREENS = object()
360def text_screen_options(frame: pd.DataFrame, text_col: str, text_id) -> list[str]:
361 """The screens one text is spread over, in reading order (BUG-26).
363 Empty for a single-screen dataset — which is every corpus without DATA-21
364 screen identity, so callers can treat "no options" as "there is nothing to
365 pick". Ordered by ``screen_index`` when present (MultiplEYE ranks it by first
366 fixation onset, i.e. the order the reader actually went through the pages),
367 else by first appearance.
368 """
369 if frame is None or frame.empty or not has_screen_identity(frame):
370 return []
371 sub = _text_subset(frame, text_col, text_id, screen_id=_ALL_SCREENS)
372 if sub.empty:
373 return []
374 if SCREEN_INDEX in sub.columns:
375 order = (
376 sub[[SCREEN_ID, SCREEN_INDEX]]
377 .assign(**{SCREEN_INDEX: pd.to_numeric(sub[SCREEN_INDEX], errors="coerce")})
378 .groupby(SCREEN_ID, sort=False)[SCREEN_INDEX]
379 .min()
380 .sort_values(kind="stable")
381 )
382 return [str(value) for value in order.index]
383 return [str(value) for value in sub[SCREEN_ID].astype(str).unique()]
386def _text_subset(
387 frame: pd.DataFrame, text_col: str, text_id, screen_id=None
388) -> pd.DataFrame:
389 """One text's rows, scoped to **one screen** when the frame has screens.
391 BUG-26: ``word_id`` is unique only *within* a screen (which is why
392 :func:`multipart.grouping_columns` appends ``screen_id``), so a per-text
393 aggregation keyed on ``(text_id, word_id)`` pools page 1's word 0 with page
394 2's — and, on MultiplEYE since DATA-24, with each comprehension-question
395 screen's first word as well.
397 ``screen_id=None`` therefore does **not** mean "all screens": on a frame with
398 screen identity it scopes to the *first* screen in reading order, so a caller
399 that hasn't been updated gets a coherent single-screen answer rather than a
400 silently pooled one. Frames without screen identity ignore the argument
401 entirely, which is every dataset that is not multipart.
402 """
403 if frame is None or frame.empty:
404 return pd.DataFrame()
405 sub = frame
406 if text_col and text_col in sub.columns:
407 sub = sub[sub[text_col].astype(str) == str(text_id)]
408 if screen_id is _ALL_SCREENS or not has_screen_identity(sub) or sub.empty:
409 return sub
410 screens = sub[SCREEN_ID].astype(str)
411 if screen_id is None:
412 wanted = _first_screen(sub, screens)
413 if wanted is None:
414 return sub
415 else:
416 wanted = str(screen_id)
417 return sub[screens == wanted]
420def _first_screen(sub: pd.DataFrame, screens: pd.Series) -> str | None:
421 """The lowest-``screen_index`` screen id in ``sub``, else the first seen."""
422 if SCREEN_INDEX in sub.columns:
423 index = pd.to_numeric(sub[SCREEN_INDEX], errors="coerce")
424 if index.notna().any():
425 return str(screens[index.idxmin()])
426 return str(screens.iloc[0]) if len(screens) else None
429def _measure_series(frame: pd.DataFrame, measure: Measure) -> pd.Series:
430 """Numeric series for ``measure``; booleans/rates coerced to 0/1 floats."""
431 s = frame[measure.column]
432 if measure.is_rate:
433 if pd.api.types.is_bool_dtype(s):
434 return s.astype("float64")
435 return pd.to_numeric(s, errors="coerce").astype("float64")
436 return pd.to_numeric(s, errors="coerce")
439# --- Per text: one text, many readers (AN-1 … AN-6) --------------------------
442def per_reader_word_measure(
443 words: pd.DataFrame,
444 text_col: str,
445 text_id,
446 measure: Measure,
447 *,
448 agg: str = "mean",
449 normalize: bool = False,
450 screen_id=None,
451) -> pd.DataFrame:
452 """Tidy ``[participant_id, word_id, value, word_text]`` for one text (AN-1/2).
454 One row per (reader, word). ``agg`` collapses any repeated readings; values
455 are optionally z-scored within reader first (AN-25). ``screen_id`` picks the
456 screen on a multipart text — see :func:`_text_subset` for why there is no
457 "all screens".
458 """
459 cols = {"participant_id", "word_id", measure.column}
460 sub = _text_subset(words, text_col, text_id, screen_id)
461 if sub.empty or not cols <= set(sub.columns):
462 return pd.DataFrame(columns=["participant_id", "word_id", "value", "word_text"])
463 df = sub[["participant_id", "word_id"]].copy()
464 df["value"] = _measure_series(sub, measure).to_numpy()
465 if "text" in sub.columns:
466 df["word_text"] = sub["text"].to_numpy()
467 df = df.dropna(subset=["word_id", "value"])
468 if df.empty:
469 return pd.DataFrame(columns=["participant_id", "word_id", "value", "word_text"])
470 if normalize and not measure.is_rate:
471 df = add_normalized_column(df, "value")
472 keys = ["participant_id", "word_id"]
473 agg_map = {"value": agg}
474 if "word_text" in df.columns:
475 agg_map["word_text"] = "first"
476 out = df.groupby(keys, as_index=False).agg(agg_map)
477 return out.sort_values(["participant_id", "word_id"]).reset_index(drop=True)
480def cohort_word_profile(
481 words: pd.DataFrame,
482 text_col: str,
483 text_id,
484 measure: Measure,
485 *,
486 agg: str = "mean",
487 spread: str = "SD",
488 normalize: bool = False,
489 min_readers: int = 1,
490 screen_id=None,
491) -> pd.DataFrame:
492 """Per-word cohort centre + spread band across readers (AN-3 / AN-15).
494 Returns ``[word_id, value, lo, hi, n, enough, word_text]`` where ``value`` is
495 the ``agg`` across readers of each reader's per-word value, ``lo``/``hi`` the
496 ``spread`` band, ``n`` the contributing reader count and ``enough`` the
497 min-readers guard (AN-26).
498 """
499 per = per_reader_word_measure(
500 words,
501 text_col,
502 text_id,
503 measure,
504 agg=agg,
505 normalize=normalize,
506 screen_id=screen_id,
507 )
508 cols = ["word_id", "value", "lo", "hi", "n", "enough", "word_text"]
509 if per.empty:
510 return pd.DataFrame(columns=cols)
511 rows = []
512 texts = (
513 per.groupby("word_id")["word_text"].first()
514 if "word_text" in per.columns
515 else None
516 )
517 for wid, grp in per.groupby("word_id"):
518 vals = grp["value"].to_numpy()
519 center = aggregate_value(vals, agg)
520 lo, hi = spread_bounds(vals, center, spread, agg=agg)
521 rows.append(
522 {
523 "word_id": wid,
524 "value": center,
525 "lo": lo,
526 "hi": hi,
527 "n": int(np.sum(~np.isnan(vals))),
528 "enough": int(np.sum(~np.isnan(vals))) >= min_readers,
529 "word_text": texts.get(wid, "") if texts is not None else "",
530 }
531 )
532 return (
533 pd.DataFrame(rows, columns=cols).sort_values("word_id").reset_index(drop=True)
534 )
537def word_box_aggregate(
538 words: pd.DataFrame,
539 text_col: str,
540 text_id,
541 measure: Measure,
542 *,
543 agg: str = "mean",
544 screen_id=None,
545) -> pd.DataFrame:
546 """One-row-per-word frame: word-box geometry + a ``value`` column = the
547 ``measure`` aggregated across readers (AN-4 stimulus tint).
549 Carries the canonical geometry (x/y/width/height/text/line_idx) + a synthetic
550 participant/trial so it feeds straight into ``plots.make_scanpath_figure``'s
551 words-only path. One screen only (BUG-26): the boxes are measured against
552 their own screen's origin, so two screens' geometry would draw on top of each
553 other however the values were keyed.
554 """
555 sub = _text_subset(words, text_col, text_id, screen_id)
556 if sub.empty or "word_id" not in sub.columns or measure.column not in sub.columns:
557 return pd.DataFrame()
558 geom_cols = [
559 c for c in ("x", "y", "width", "height", "text", "line_idx") if c in sub.columns
560 ]
561 if not geom_cols:
562 return pd.DataFrame()
563 work = sub[["word_id"] + geom_cols].copy()
564 work["_m"] = _measure_series(sub, measure).to_numpy()
565 grouped = work.groupby("word_id")
566 out = grouped[geom_cols].first()
567 out["value"] = grouped["_m"].agg(lambda s: aggregate_value(s.to_numpy(), agg))
568 out = out.reset_index()
569 out["participant_id"] = "aggregate"
570 out["trial_id"] = str(text_id)
571 return out
574def word_measure_vs_feature(
575 words: pd.DataFrame,
576 text_col: str,
577 text_id,
578 measure: Measure,
579 feature_col: str,
580 *,
581 agg: str = "mean",
582 normalize: bool = False,
583 screen_id=None,
584) -> pd.DataFrame:
585 """Per-word ``[word_id, value, feature, word_text]`` for a measure vs a
586 bundled linguistic feature (AN-5). ``value`` is the cross-reader aggregate;
587 ``feature`` is the per-word feature (constant across readers)."""
588 per = per_reader_word_measure(
589 words,
590 text_col,
591 text_id,
592 measure,
593 agg=agg,
594 normalize=normalize,
595 screen_id=screen_id,
596 )
597 sub = _text_subset(words, text_col, text_id, screen_id)
598 if per.empty or feature_col not in sub.columns:
599 return pd.DataFrame(columns=["word_id", "value", "feature", "word_text"])
600 center_agg = {"value": ("value", lambda s: aggregate_value(s.to_numpy(), agg))}
601 if "word_text" in per.columns:
602 # Gate on column presence — never synthesize a count under this name.
603 center_agg["word_text"] = ("word_text", "first")
604 center = per.groupby("word_id").agg(**center_agg).reset_index()
605 feat = sub.groupby("word_id")[feature_col].first().rename("feature").reset_index()
606 out = center.merge(feat, on="word_id", how="left")
607 return out.dropna(subset=["value"]).reset_index(drop=True)
610#: ``(rate, readers behind it, min-readers verdict)`` for each series
611#: :func:`word_rate_profile` returns.
612RATE_SERIES = (
613 ("skip_rate", "n_skip", "enough_skip"),
614 ("regression_in_rate", "n_regression_in", "enough_regression_in"),
615)
616RATE_SERIES_COLUMNS = (
617 "skip_rate",
618 "regression_in_rate",
619 "n_skip",
620 "n_regression_in",
621 "enough_skip",
622 "enough_regression_in",
623)
626def word_rate_profile(
627 words: pd.DataFrame,
628 text_col: str,
629 text_id,
630 *,
631 min_readers: int = 1,
632 screen_id=None,
633) -> pd.DataFrame:
634 """Per-word skip / regression-in rates across readers (AN-6).
636 Returns ``[word_id, skip_rate, regression_in_rate, n_skip, n_regression_in,
637 enough_skip, enough_regression_in, word_text]``. Each rate has its own
638 denominator: ``n_*`` counts the readers who *reported* that flag for the
639 word — a missing flag is no observation, not a "no" — and ``enough_*``
640 applies ``min_readers`` to that count alone, since a dataset can record
641 skips for every word and regressions only for the ones read in first pass.
642 """
643 sub = _text_subset(words, text_col, text_id, screen_id)
644 cols = ["word_id", *RATE_SERIES_COLUMNS, "word_text"]
645 if sub.empty or "word_id" not in sub.columns:
646 return pd.DataFrame(columns=cols)
647 work = sub[["word_id"]].copy()
648 for src, dst in (
649 ("skip_flag", "skip_rate"),
650 ("regression_in_flag", "regression_in_rate"),
651 ):
652 if src in sub.columns:
653 s = sub[src]
654 work[dst] = (
655 s.astype("float64")
656 if pd.api.types.is_bool_dtype(s)
657 else pd.to_numeric(s, errors="coerce")
658 ).to_numpy()
659 else:
660 work[dst] = np.nan
661 if "text" in sub.columns:
662 work["word_text"] = sub["text"].to_numpy()
663 # Collapse to one row per (word, reader) FIRST, the way per_reader_word_measure
664 # does. Without it `n` counts rows, so a reader who read the text twice counts
665 # as two readers and clears a min_readers guard they shouldn't — and the rates
666 # below are row-weighted, over-counting whoever re-read (ENG-1).
667 if "participant_id" in sub.columns:
668 work["participant_id"] = sub["participant_id"].to_numpy()
669 text_by_word = (
670 work.groupby("word_id")["word_text"].first()
671 if "word_text" in work.columns
672 else None
673 )
674 work = work.groupby(["word_id", "participant_id"], as_index=False).mean(
675 numeric_only=True
676 )
677 if text_by_word is not None:
678 work["word_text"] = work["word_id"].map(text_by_word)
679 grouped = work.groupby("word_id")
680 # `count` is the readers with a value: the per-reader mean is NaN only for
681 # a reader who reported nothing for that word.
682 out = grouped.agg(
683 skip_rate=("skip_rate", "mean"),
684 regression_in_rate=("regression_in_rate", "mean"),
685 n_skip=("skip_rate", "count"),
686 n_regression_in=("regression_in_rate", "count"),
687 ).reset_index()
688 if "word_text" in work.columns:
689 out = out.merge(
690 grouped["word_text"].first().reset_index(), on="word_id", how="left"
691 )
692 else:
693 out["word_text"] = ""
694 for _rate, n, enough in RATE_SERIES:
695 out[enough] = out[n] >= max(int(min_readers), 1)
696 return out[cols].sort_values("word_id").reset_index(drop=True)
699# --- Per reader: one reader, many trials (AN-7 … AN-13) ----------------------
702def measure_values(
703 frame: pd.DataFrame, measure: Measure, *, normalize: bool = False
704) -> np.ndarray:
705 """Flat array of a measure's values (for distribution plots)."""
706 if frame is None or frame.empty or measure.column not in frame.columns:
707 return np.array([], dtype="float64")
708 work = frame.copy()
709 work["_m"] = _measure_series(work, measure).to_numpy()
710 if normalize and not measure.is_rate:
711 work = add_normalized_column(work, "_m")
712 return work["_m"].dropna().to_numpy()
715def reader_means(frame: pd.DataFrame, measure: Measure) -> np.ndarray | None:
716 """One value per reader — the mean of their observations (AN-21, BUG-82).
718 The unit the Groups summary compares. Every word or fixation of one reader
719 is correlated with that reader's others, so pooling the observations
720 (``measure_values``) would let one reader's 1 307 words outweigh another
721 reader's 200. ``None`` when the frame names no readers, so the caller can
722 say it is falling back to observations.
723 """
724 if frame is None or frame.empty or measure.column not in frame.columns:
725 return np.array([], dtype="float64")
726 if "participant_id" not in frame.columns:
727 return None
728 values = pd.Series(_measure_series(frame, measure).to_numpy(), index=frame.index)
729 means = values.groupby(frame["participant_id"].astype(str)).mean()
730 return means.dropna().to_numpy(dtype="float64")
733def reader_vs_cohort_values(
734 frame: pd.DataFrame,
735 participant_id,
736 measure: Measure,
737 *,
738 normalize: bool = False,
739) -> dict[str, np.ndarray]:
740 """``{"This participant": …, "Cohort": …}`` value arrays for a measure (AN-7)."""
741 if frame is None or frame.empty or "participant_id" not in frame.columns:
742 return {}
743 work = frame.copy()
744 work["_m"] = _measure_series(work, measure).to_numpy()
745 if normalize and not measure.is_rate:
746 work = add_normalized_column(work, "_m")
747 is_target = work["participant_id"].astype(str) == str(participant_id)
748 out: dict[str, np.ndarray] = {}
749 me = work.loc[is_target, "_m"].dropna().to_numpy()
750 others = work.loc[~is_target, "_m"].dropna().to_numpy()
751 if me.size:
752 out["This participant"] = me
753 if others.size:
754 out["Cohort"] = others
755 return out
758#: How a summary's reading time was measured (`reading_time_source`).
759READING_TIME_RECORDED = "recorded"
760READING_TIME_ESTIMATED = "estimate (summed fixation durations; no timestamps)"
763def _trial_reading_time_ms(fixations: pd.DataFrame) -> pd.DataFrame:
764 """Per-reading (and per-screen) reading time in ms, and how it was measured.
766 Recorded timestamps give the span, last fixation end − first fixation
767 start. Without them — no timestamp column, or one normalization numbered
768 0, 1, 2, … because the table had no onset (``data.TIMESTAMP_SYNTHESIZED``)
769 — the fixations are laid end to end by their durations, the clock the
770 replay uses (``measures.rebased_fixation_onsets``): an estimate that leaves
771 out every gap between them, so ``reading_time_source`` says so."""
772 from .data import TIMESTAMP_SYNTHESIZED
774 columns = ["reading_time_ms", "reading_time_source"]
775 if fixations.empty or not {"participant_id", "trial_id"} <= set(fixations.columns):
776 return pd.DataFrame(columns=["participant_id", "trial_id", *columns])
777 keys = grouping_columns(fixations)
778 df = fixations[keys].copy()
779 dur = pd.to_numeric(fixations.get("duration_ms"), errors="coerce")
780 df["_d"] = dur
781 if TIMESTAMP_SYNTHESIZED in fixations.columns:
782 df["_synth"] = fixations[TIMESTAMP_SYNTHESIZED].fillna(False).astype(bool)
783 else:
784 df["_synth"] = "timestamp_ms" not in fixations.columns
785 if "timestamp_ms" in fixations.columns:
786 ts = pd.to_numeric(fixations["timestamp_ms"], errors="coerce")
787 df["_start"] = ts
788 df["_end"] = ts + dur.fillna(0)
789 else:
790 df["_start"] = df["_end"] = np.nan
791 grp = df.groupby(keys)
792 estimated = grp["_synth"].any()
793 span = grp["_end"].max() - grp["_start"].min()
794 summed = grp["_d"].sum()
795 out = pd.DataFrame(
796 {
797 "reading_time_ms": summed.where(estimated, span),
798 "reading_time_source": np.where(
799 estimated, READING_TIME_ESTIMATED, READING_TIME_RECORDED
800 ),
801 }
802 )
803 return out.reset_index()
806def _first_present(frame: pd.DataFrame, columns: Sequence[str]):
807 """First non-null value from the first available column."""
808 if frame is None or frame.empty:
809 return None
810 for column in columns:
811 if column not in frame.columns:
812 continue
813 values = frame[column].dropna()
814 if not values.empty:
815 return values.iloc[0]
816 return None
819def _as_accuracy(value) -> float | None:
820 """Coerce common correctness encodings to 0/1 without guessing blanks."""
821 if value is None or pd.isna(value):
822 return None
823 if isinstance(value, (bool, np.bool_)):
824 return float(value)
825 if isinstance(value, (int, float, np.integer, np.floating)):
826 return float(bool(value))
827 text = str(value).strip().lower()
828 if text in {"true", "t", "yes", "y", "correct", "1"}:
829 return 1.0
830 if text in {"false", "f", "no", "n", "incorrect", "0"}:
831 return 0.0
832 return None
835def _run_summary(fixations: pd.DataFrame, n_words: int) -> dict[str, float]:
836 """Run/refixation and first-pass/rereading stats for one ordered trial.
838 A run is a contiguous visit to one word. The first run on each word belongs
839 to first pass; later visits are rereading. This works on normalized streams
840 without requiring a corpus-specific ``pass_index`` convention.
841 """
842 if fixations.empty or "word_id" not in fixations.columns:
843 return {}
844 order = [
845 c for c in ("timestamp_ms", "order_in_trial", "fixation_id") if c in fixations
846 ]
847 fx = fixations.sort_values(order, kind="stable") if order else fixations.copy()
848 word = fx["word_id"]
849 valid = word.notna()
850 if not bool(valid.any()):
851 return {}
852 run_start = word.ne(word.shift()) | ~valid | ~valid.shift(fill_value=False)
853 run_id = run_start.cumsum()
854 visits = pd.DataFrame(
855 {
856 "word_id": word.to_numpy(),
857 "run_id": run_id.to_numpy(),
858 "duration_ms": pd.to_numeric(
859 fx.get("duration_ms"), errors="coerce"
860 ).to_numpy(),
861 },
862 index=fx.index,
863 ).loc[valid]
864 runs = (
865 visits.groupby(["word_id", "run_id"], sort=False)
866 .agg(duration_ms=("duration_ms", "sum"), n_fixations=("duration_ms", "size"))
867 .reset_index()
868 )
869 runs["visit_index"] = runs.groupby("word_id", sort=False).cumcount()
870 first = runs["visit_index"] == 0
871 denominator = n_words or int(runs["word_id"].nunique())
872 return {
873 "nrun": len(runs),
874 "first_pass_ms": float(runs.loc[first, "duration_ms"].sum()),
875 "rereading_ms": float(runs.loc[~first, "duration_ms"].sum()),
876 "refixation_rate": float(
877 (runs.loc[first, "n_fixations"] > 1).sum() / denominator
878 )
879 if denominator
880 else np.nan,
881 }
884def trial_summary_table(words: pd.DataFrame, fixations: pd.DataFrame) -> pd.DataFrame:
885 """One export-ready summary row per trial screen (or legacy trial) (AN-30).
887 Includes the standard reading totals plus run-derived refixation and
888 first-pass/rereading measures. Missing source concepts stay absent/NaN
889 rather than being silently invented (for example blink count).
890 """
891 source = fixations if fixations is not None and not fixations.empty else words
892 key_columns = grouping_columns(source)
893 keys: list[tuple] = []
894 for frame in (fixations, words):
895 if frame is None or frame.empty or not set(key_columns) <= set(frame.columns):
896 continue
897 pairs = frame[key_columns].drop_duplicates()
898 keys.extend(tuple(str(value) for value in row) for row in pairs.to_numpy())
899 keys = list(dict.fromkeys(keys))
900 rows: list[dict] = []
901 for identity in keys:
902 selected = dict(zip(key_columns, identity))
904 # `selected` bound as a default: the closure is called inside this
905 # same iteration, so late binding is harmless today, but binding it
906 # makes that a property of the code rather than of the call order.
907 def _slice(frame: pd.DataFrame, selected=selected) -> pd.DataFrame:
908 if (
909 frame is None
910 or frame.empty
911 or not set(key_columns) <= set(frame.columns)
912 ):
913 return pd.DataFrame()
914 mask = pd.Series(True, index=frame.index)
915 for column, value in selected.items():
916 mask &= frame[column].astype(str) == value
917 return frame[mask]
919 fx, wd = _slice(fixations), _slice(words)
920 row: dict = dict(selected)
921 text_id = _first_present(
922 wd if not wd.empty else fx, ("text_id", "unique_text_id")
923 )
924 if text_id is not None:
925 row["text_id"] = text_id
926 n_words = (
927 int(wd["word_id"].nunique())
928 if not wd.empty and "word_id" in wd.columns
929 else 0
930 )
931 row["n_words"] = n_words
932 if not fx.empty and "excluded" in fx.columns:
933 fx = fx.loc[~fx["excluded"].fillna(False).astype(bool)]
934 if not fx.empty:
935 duration = pd.to_numeric(
936 fx.get("duration_ms", pd.Series(np.nan, index=fx.index)),
937 errors="coerce",
938 )
939 row["n_fixations"] = len(fx)
940 if duration.notna().any():
941 row["mean_fixation_ms"] = float(duration.mean())
942 row["total_fixation_ms"] = float(duration.sum())
943 amplitude = pd.to_numeric(
944 fx.get("saccade_amplitude", pd.Series(np.nan, index=fx.index)),
945 errors="coerce",
946 )
947 if amplitude.notna().any():
948 row["mean_saccade_px"] = float(amplitude.mean())
949 regression = (
950 pd.to_numeric(
951 fx.get("is_regression", pd.Series(False, index=fx.index)),
952 errors="coerce",
953 )
954 .fillna(0)
955 .astype(bool)
956 )
957 # BUG-68: "forward" is not simply "not a regression" — the sweep
958 # back to the start of the next line advances in reading order
959 # but is a line-length leftward jump, and counting it put the
960 # demo's mean forward saccade at 270 px instead of 172.
961 # `saccade_amplitude` is the saccade *into* each fixation, while
962 # `classify_saccades` labels the one *out of* it, hence the shift.
963 progressive = ~regression
964 # A sweep is a line change, so it needs the boxes' geometry; a
965 # words table without it keeps the plain progressive split.
966 if not wd.empty and {"y", "height"} <= set(wd.columns):
967 order = (
968 fx.sort_values("timestamp_ms").index
969 if "timestamp_ms" in fx.columns
970 else fx.index
971 )
972 outgoing = classify_saccades(fx, wd).reindex(order)
973 incoming = pd.Series(
974 outgoing.shift(1).to_numpy(), index=order
975 ).reindex(fx.index)
976 progressive &= incoming.ne("return_sweep")
977 forward = amplitude[progressive]
978 if forward.notna().any():
979 row["mean_forward_saccade_px"] = float(forward.mean())
980 if "is_regression" in fx.columns:
981 row["regression_rate"] = float(
982 pd.to_numeric(fx["is_regression"], errors="coerce").mean()
983 )
984 reading = _trial_reading_time_ms(fx)
985 if not reading.empty:
986 row["reading_time_ms"] = float(reading.iloc[0]["reading_time_ms"])
987 row["reading_time_source"] = reading.iloc[0]["reading_time_source"]
988 row.update(_run_summary(fx, n_words))
989 if "blink_count" in fx.columns:
990 blink = pd.to_numeric(fx["blink_count"], errors="coerce").dropna()
991 if not blink.empty:
992 row["blink_count"] = float(blink.max())
993 elif "is_blink" in fx.columns:
994 row["blink_count"] = int(
995 pd.to_numeric(fx["is_blink"], errors="coerce")
996 .fillna(0)
997 .astype(bool)
998 .sum()
999 )
1000 if not wd.empty:
1001 if "skip_flag" in wd.columns:
1002 row["skip_rate"] = float(
1003 pd.to_numeric(wd["skip_flag"], errors="coerce").mean()
1004 )
1005 if "regression_in_flag" in wd.columns:
1006 row["regression_in_rate"] = float(
1007 pd.to_numeric(wd["regression_in_flag"], errors="coerce").mean()
1008 )
1009 if "regression_in_rate" not in row and "regression_rate" in row:
1010 row["regression_in_rate"] = row["regression_rate"]
1011 reading_ms = row.get("reading_time_ms")
1012 if n_words and reading_ms and reading_ms > 0:
1013 row["wpm"] = float(n_words / (reading_ms / 60000.0))
1014 correct = _as_accuracy(
1015 _first_present(
1016 wd if not wd.empty else fx, ("is_correct", "question_correct")
1017 )
1018 )
1019 if correct is not None:
1020 row["question_correct"] = correct
1021 rows.append(row)
1022 return pd.DataFrame(rows)
1025def reader_summary(
1026 words: pd.DataFrame, fixations: pd.DataFrame, participant_id
1027) -> dict[str, float]:
1028 """Compact reading-profile stats for one reader (AN-8).
1030 WPM, mean fixation duration, fixation count, regression rate, skip rate,
1031 mean saccade amplitude, trial count — computed from whatever columns exist.
1032 """
1033 return _summary_row(words, fixations, participant_id)
1036def _summary_row(words, fixations, pid) -> dict[str, float]:
1037 out: dict[str, float] = {"participant_id": str(pid)}
1038 fx = (
1039 fixations[fixations["participant_id"].astype(str) == str(pid)]
1040 if not fixations.empty and "participant_id" in fixations.columns
1041 else pd.DataFrame()
1042 )
1043 wd = (
1044 words[words["participant_id"].astype(str) == str(pid)]
1045 if not words.empty and "participant_id" in words.columns
1046 else pd.DataFrame()
1047 )
1048 if not fx.empty:
1049 out["n_trials"] = (
1050 int(fx.groupby(["participant_id", "trial_id"]).ngroups)
1051 if {"participant_id", "trial_id"} <= set(fx.columns)
1052 else 0
1053 )
1054 out["n_fixations"] = len(fx)
1055 if "duration_ms" in fx.columns:
1056 out["mean_fixation_ms"] = float(
1057 pd.to_numeric(fx["duration_ms"], errors="coerce").mean()
1058 )
1059 if "saccade_amplitude" in fx.columns:
1060 out["mean_saccade_px"] = float(
1061 pd.to_numeric(fx["saccade_amplitude"], errors="coerce").mean()
1062 )
1063 if "is_regression" in fx.columns:
1064 out["regression_rate"] = float(
1065 pd.to_numeric(fx["is_regression"], errors="coerce").mean()
1066 )
1067 rt = _trial_reading_time_ms(fx)
1068 total_ms = float(pd.to_numeric(rt["reading_time_ms"], errors="coerce").sum())
1069 # BUG-67: a word is identified by its screen too — MultiplEYE restarts
1070 # `word_id` on every page, so (trial, word) folded a trial's pages
1071 # onto each other and under-counted the words read per minute.
1072 n_words = (
1073 int(wd.groupby(grouping_columns(wd, include_word=True)).ngroups)
1074 if not wd.empty and {"trial_id", "word_id"} <= set(wd.columns)
1075 else 0
1076 )
1077 if total_ms > 0 and n_words:
1078 out["wpm"] = float(n_words / (total_ms / 60000.0))
1079 out["reading_time_source"] = (
1080 READING_TIME_ESTIMATED
1081 if (rt["reading_time_source"] == READING_TIME_ESTIMATED).any()
1082 else READING_TIME_RECORDED
1083 )
1084 trials = trial_summary_table(wd, fx)
1085 if not trials.empty:
1086 for column in (
1087 "total_fixation_ms",
1088 "blink_count",
1089 "nrun",
1090 "first_pass_ms",
1091 "rereading_ms",
1092 "n_question_correct",
1093 ):
1094 source = (
1095 "question_correct" if column == "n_question_correct" else column
1096 )
1097 if source in trials.columns:
1098 values = pd.to_numeric(trials[source], errors="coerce").dropna()
1099 if not values.empty:
1100 out[column] = float(values.sum())
1101 for column in (
1102 "mean_forward_saccade_px",
1103 "refixation_rate",
1104 "regression_in_rate",
1105 "comprehension_accuracy",
1106 ):
1107 source = (
1108 "question_correct" if column == "comprehension_accuracy" else column
1109 )
1110 if source in trials.columns:
1111 values = pd.to_numeric(trials[source], errors="coerce").dropna()
1112 if not values.empty:
1113 out[column] = float(values.mean())
1114 if not wd.empty and "skip_flag" in wd.columns:
1115 out["skip_rate"] = float(pd.to_numeric(wd["skip_flag"], errors="coerce").mean())
1116 return out
1119def cohort_summary_table(
1120 words: pd.DataFrame,
1121 fixations: pd.DataFrame,
1122 *,
1123 participants: Sequence | None = None,
1124) -> pd.DataFrame:
1125 """One summary row per reader (AN-16 / AN-8 cohort percentiles)."""
1126 pids: list = []
1127 for frame in (fixations, words):
1128 if frame is not None and not frame.empty and "participant_id" in frame.columns:
1129 pids = list(pd.unique(frame["participant_id"].astype(str)))
1130 break
1131 if participants is not None:
1132 keep = {str(p) for p in participants}
1133 pids = [p for p in pids if p in keep]
1134 if not pids:
1135 return pd.DataFrame()
1136 rows = [_summary_row(words, fixations, p) for p in pids]
1137 return pd.DataFrame(rows)
1140def reader_summary_table(
1141 words: pd.DataFrame,
1142 fixations: pd.DataFrame,
1143 *,
1144 participants: Sequence | None = None,
1145) -> pd.DataFrame:
1146 """First-class per-reader summary table (AN-30).
1148 Kept as an explicit public name rather than forcing callers to know the
1149 older ``cohort_summary_table`` terminology.
1150 """
1151 return cohort_summary_table(words, fixations, participants=participants)
1154def metric_over_time(
1155 fixations: pd.DataFrame,
1156 measure: Measure,
1157 *,
1158 participant_id=None,
1159 by: str = "order_in_trial",
1160) -> pd.DataFrame:
1161 """Mean of a per-fixation measure vs within-trial index (AN-9).
1163 Returns ``[x, value, sem, n]``. ``by`` is ``order_in_trial`` (default) or
1164 ``timestamp_ms``. Filtered to ``participant_id`` when given.
1165 """
1166 if (
1167 fixations.empty
1168 or measure.column not in fixations.columns
1169 or by not in fixations
1170 ):
1171 return pd.DataFrame(columns=["x", "value", "sem", "n"])
1172 fx = fixations
1173 if participant_id is not None and "participant_id" in fx.columns:
1174 fx = fx[fx["participant_id"].astype(str) == str(participant_id)]
1175 df = pd.DataFrame(
1176 {
1177 "x": pd.to_numeric(fx[by], errors="coerce"),
1178 "_m": _measure_series(fx, measure).to_numpy(),
1179 }
1180 ).dropna()
1181 if df.empty:
1182 return pd.DataFrame(columns=["x", "value", "sem", "n"])
1183 out = df.groupby("x")["_m"].agg(["mean", "sem", "count"]).reset_index()
1184 out.columns = ["x", "value", "sem", "n"]
1185 out["sem"] = out["sem"].fillna(0.0)
1186 return out.sort_values("x").reset_index(drop=True)
1189def saccade_vs_duration(
1190 fixations: pd.DataFrame, *, participant_id=None
1191) -> pd.DataFrame:
1192 """``[duration_ms, saccade_amplitude]`` rows for the oculomotor scatter (AN-10)."""
1193 need = {"duration_ms", "saccade_amplitude"}
1194 if fixations.empty or not need <= set(fixations.columns):
1195 return pd.DataFrame(columns=["duration_ms", "saccade_amplitude"])
1196 fx = fixations
1197 if participant_id is not None and "participant_id" in fx.columns:
1198 fx = fx[fx["participant_id"].astype(str) == str(participant_id)]
1199 out = pd.DataFrame(
1200 {
1201 "duration_ms": pd.to_numeric(fx["duration_ms"], errors="coerce"),
1202 "saccade_amplitude": pd.to_numeric(
1203 fx["saccade_amplitude"], errors="coerce"
1204 ),
1205 }
1206 ).dropna()
1207 return out.reset_index(drop=True)
1210def progressive_regressive_counts(
1211 fixations: pd.DataFrame, *, participant_id=None
1212) -> pd.DataFrame:
1213 """Per-trial progressive / regressive saccade counts + share (AN-11).
1215 Returns ``[trial_id, progressive, regressive, regression_share]``.
1216 """
1217 cols = ["trial_id", "progressive", "regressive", "regression_share"]
1218 if (
1219 fixations.empty
1220 or "is_regression" not in fixations.columns
1221 or "trial_id" not in fixations.columns
1222 ):
1223 return pd.DataFrame(columns=cols)
1224 fx = fixations
1225 if participant_id is not None and "participant_id" in fx.columns:
1226 fx = fx[fx["participant_id"].astype(str) == str(participant_id)]
1227 if fx.empty:
1228 return pd.DataFrame(columns=cols)
1229 reg = pd.to_numeric(fx["is_regression"], errors="coerce").fillna(0).astype(bool)
1230 df = pd.DataFrame({"trial_id": fx["trial_id"].to_numpy(), "_reg": reg.to_numpy()})
1231 out = df.groupby("trial_id")["_reg"].agg(["sum", "size"]).reset_index()
1232 out.columns = ["trial_id", "regressive", "_total"]
1233 out["progressive"] = out["_total"] - out["regressive"]
1234 out["regression_share"] = out["regressive"] / out["_total"].replace(0, np.nan)
1235 return out[cols].sort_values("trial_id").reset_index(drop=True)
1238def ensure_fixation_enrichment(
1239 fixations: pd.DataFrame, words: pd.DataFrame
1240) -> pd.DataFrame:
1241 """Add ``is_regression`` / ``saccade_amplitude`` to ``fixations`` if missing.
1243 The pre-aggregated OneStop path keeps the EyeLink IA measures on *words* and
1244 never enriches the fixation stream, so derived per-fixation columns (AN-11)
1245 aren't there. When ``word_id`` is present we run the same native enrichment
1246 (``measures.enrich_fixations``) the computed path uses; otherwise the frame
1247 is returned unchanged. Cheap and idempotent.
1248 """
1249 if fixations is None or fixations.empty:
1250 return fixations
1251 if "is_regression" in fixations.columns or "word_id" not in fixations.columns:
1252 return fixations
1253 try:
1254 from .measures import enrich_fixations
1256 return enrich_fixations(fixations, words)
1257 except Exception: # pragma: no cover - defensive
1258 return fixations
1261def landing_positions(
1262 words: pd.DataFrame,
1263 fixations: pd.DataFrame | None = None,
1264 *,
1265 participant_id=None,
1266 as_fraction: bool = True,
1267) -> np.ndarray:
1268 """Within-word landing positions of the first fixation on each word (AN-12).
1270 Prefers the pre-computed ``first_fix_x`` on ``words``. When that's absent —
1271 the pre-aggregated OneStop path — it derives the landing from ``fixations``:
1272 the earliest fixation on each ``(participant, trial, word)``, joined to the
1273 word box. Returns the landing as a *fraction of the word's interest area* by
1274 default, else the px distance from where the word starts.
1276 The fraction is ``(x_fix − x) / width`` — 0 at the box's leading edge, 1 at
1277 its trailing edge — over the experiment's own box (BUG-83). It is the same
1278 quantity as ``initial_landing_position``, in different units: that letter
1279 position minus one, over the box's ``width / advance`` character cells. On a
1280 glyph-tight corpus the box *is* the glyph run, so 0 is the first letter's
1281 edge and 1 the last's. On a tiling corpus the box's last cell is the space
1282 after the word, so the glyphs fill ``[0, n / (n + 1))`` and a first fixation
1283 on that space — which belongs to this word, as in EyeLink's report — reads
1284 between ``n / (n + 1)`` and 1, where it used to be clipped onto 1.0 (15% of
1285 the demo's landings piled up there). Right-to-left words are mirrored the
1286 way the letter position is: counted from where the glyphs end.
1288 Not clipped: a first fixation the word got although it lies outside the box
1289 horizontally (an imported ``word_id``)
1290 reads below 0 or above 1, rather than piling onto an edge it did not land on.
1291 """
1292 from .measures import word_glyph_span
1294 if words is not None and not words.empty and "first_fix_x" in words.columns:
1295 wd = words
1296 if participant_id is not None and "participant_id" in wd.columns:
1297 wd = wd[wd["participant_id"].astype(str) == str(participant_id)]
1298 ffx = pd.to_numeric(wd.get("first_fix_x"), errors="coerce")
1299 starts, runs = word_glyph_span(wd, layout=words)
1300 left = pd.Series(starts, index=wd.index)
1301 run = pd.Series(runs, index=wd.index)
1302 width = pd.to_numeric(wd["width"], errors="coerce")
1303 dist = ffx - left
1304 if "right_to_left" in wd.columns:
1305 rtl = wd["right_to_left"].fillna(False).astype(bool)
1306 dist = dist.where(~rtl, run - dist)
1307 mask = ffx.notna() & left.notna() & width.notna() & (width > 0)
1308 if "skip_flag" in wd.columns:
1309 mask &= ~pd.to_numeric(wd["skip_flag"], errors="coerce").fillna(0).astype(
1310 bool
1311 )
1312 dist = dist[mask]
1313 width = width[mask]
1314 elif (
1315 fixations is not None
1316 and not fixations.empty
1317 and {"word_id", "x"} <= set(fixations.columns)
1318 and words is not None
1319 and {"word_id", "x", "width"} <= set(words.columns)
1320 ):
1321 fx = fixations
1322 if participant_id is not None and "participant_id" in fx.columns:
1323 fx = fx[fx["participant_id"].astype(str) == str(participant_id)]
1324 if fx.empty:
1325 return np.array([], dtype="float64")
1326 order_col = (
1327 "order_in_trial" if "order_in_trial" in fx.columns else "timestamp_ms"
1328 )
1329 keys = grouping_columns(fx, include_word=True)
1330 first = (
1331 fx.sort_values(order_col)
1332 .dropna(subset=["word_id"])
1333 .drop_duplicates(keys, keep="first")[keys + ["x"]]
1334 .rename(columns={"x": "_fx"})
1335 )
1336 box_keys = [k for k in keys if k in words.columns]
1337 extra = ["right_to_left"] if "right_to_left" in words else []
1338 # Dedup to one row per word *first*, then measure. `words` here is the
1339 # whole filtered corpus — one row per word per reader — so the glyph run
1340 # was being computed for every repetition of every box. `layout=` is what
1341 # makes that safe: tiling detection still sees the full frame, which it
1342 # has to, since a deduped subset's holes read as glyph-tight gaps.
1343 # `text` rides along because the glyph run is counted from it — without
1344 # it an RTL landing would be mirrored across the padded box instead.
1345 text_col = ["text"] if "text" in words.columns else []
1346 box = words[box_keys + ["x", "width", *text_col, *extra]].drop_duplicates(
1347 box_keys
1348 )
1349 starts, runs = word_glyph_span(box, layout=words)
1350 box = box.assign(_left=starts, _run=runs)
1351 merged = first.merge(box, on=box_keys, how="inner")
1352 left = pd.to_numeric(merged["_left"], errors="coerce")
1353 run = pd.to_numeric(merged["_run"], errors="coerce")
1354 width = pd.to_numeric(merged["width"], errors="coerce")
1355 dist = pd.to_numeric(merged["_fx"], errors="coerce") - left
1356 if "right_to_left" in merged:
1357 rtl = merged["right_to_left"].fillna(False).astype(bool)
1358 dist = dist.where(~rtl, run - dist)
1359 ok = dist.notna() & width.notna() & (width > 0)
1360 dist, width = dist[ok], width[ok]
1361 else:
1362 return np.array([], dtype="float64")
1363 if as_fraction:
1364 return (dist / width).dropna().to_numpy()
1365 return dist.dropna().to_numpy()
1368def per_participant_trend(
1369 frame: pd.DataFrame, metric: str, *, agg: str = "mean"
1370) -> pd.DataFrame:
1371 """Per-participant per-trial-index trend (faint lines behind AN-17).
1373 Returns ``[participant_id, trial_index, value]``.
1374 """
1375 cols = {"participant_id", "trial_id", "trial_index", metric}
1376 if frame.empty or not cols <= set(frame.columns):
1377 return pd.DataFrame(columns=["participant_id", "trial_index", "value"])
1378 df = frame[["participant_id", "trial_id", "trial_index"]].copy()
1379 df["_m"] = pd.to_numeric(frame[metric], errors="coerce")
1380 df = df.dropna(subset=["trial_index", "_m"])
1381 per_trial = (
1382 df.groupby(["participant_id", "trial_id", "trial_index"])["_m"]
1383 .agg(agg)
1384 .reset_index()
1385 )
1386 out = (
1387 per_trial.groupby(["participant_id", "trial_index"])["_m"].mean().reset_index()
1388 )
1389 out.columns = ["participant_id", "trial_index", "value"]
1390 return out.sort_values(["participant_id", "trial_index"]).reset_index(drop=True)
1393# --- Groups + group comparison (AN-14 … AN-22) -------------------------------
1396def group_mask(frame: pd.DataFrame, spec: Mapping[str, Sequence]) -> pd.Series:
1397 """Boolean row mask for a group ``spec`` = ``{column: [allowed values]}``.
1399 Columns are AND-ed; within a column the row matches any listed value
1400 (compared as strings). Unifies both group-definition modes: a *field split*
1401 is two specs over the same column with disjoint values, while *independent
1402 filter sets* are specs with several columns. An empty spec selects all rows.
1404 A key may also be a **tuple of columns**, whose values are tuples: a
1405 composite key the row must match as a whole (AN-31 — a trial-metadata
1406 cohort is a set of ``(participant_id, trial_id)`` readings, which two
1407 independent column constraints cannot express: reader p1's trial t1 and
1408 reader p2's trial t2 are not p1's t2). Like a single column, a composite key
1409 a frame does not fully carry constrains nothing on that frame.
1410 """
1411 if frame is None or frame.empty:
1412 return pd.Series([], dtype=bool)
1413 mask = pd.Series(True, index=frame.index)
1414 for col, vals in (spec or {}).items():
1415 if not vals:
1416 continue
1417 if isinstance(col, tuple):
1418 if not set(col) <= set(frame.columns):
1419 continue
1420 allowed = {tuple(str(part) for part in v) for v in vals}
1421 keys = pd.MultiIndex.from_arrays([frame[c].astype(str) for c in col])
1422 mask &= keys.isin(allowed)
1423 elif col in frame.columns:
1424 allowed = {str(v) for v in vals}
1425 mask &= frame[col].astype(str).isin(allowed)
1426 return mask
1429def apply_group(frame: pd.DataFrame, spec: Mapping[str, Sequence]) -> pd.DataFrame:
1430 """``frame`` rows selected by ``spec`` (see :func:`group_mask`)."""
1431 if frame is None or frame.empty:
1432 return frame
1433 return frame[group_mask(frame, spec)]
1436def _shared_screen(words: pd.DataFrame, text_col: str, text_id, screen_id):
1437 """The one screen both cohorts of a word comparison describe (BUG-114).
1439 Resolved on the frame *before* it is split: left to each cohort,
1440 :func:`_text_subset` picks each one's own first screen, so a cohort that
1441 never saw page 1 compared its page 2 against the other's page 1, word id
1442 for word id. A cohort without the chosen screen then contributes nothing.
1443 """
1444 if screen_id is not None:
1445 return screen_id
1446 options = text_screen_options(words, text_col, text_id)
1447 return options[0] if options else None
1450def distinct_group_labels(label_a: str, label_b: str) -> tuple[str, str]:
1451 """The two cohorts' labels, told apart when they read the same (BUG-111).
1453 The charts key their series by label, so two cohorts both named "Control"
1454 became one: B's values replaced A's without a word. Equal labels get a
1455 visible ``(A)`` / ``(B)``.
1456 """
1457 if str(label_a).strip() == str(label_b).strip():
1458 return f"{label_a} (A)", f"{label_b} (B)"
1459 return label_a, label_b
1462def two_group_values(
1463 frame: pd.DataFrame,
1464 measure: Measure,
1465 spec_a: Mapping,
1466 spec_b: Mapping,
1467 *,
1468 label_a: str = "Group A",
1469 label_b: str = "Group B",
1470 normalize: bool = False,
1471) -> dict[str, np.ndarray]:
1472 """``{label_a: values, label_b: values}`` for overlaid distributions (AN-18)."""
1473 label_a, label_b = distinct_group_labels(label_a, label_b)
1474 out: dict[str, np.ndarray] = {}
1475 a = measure_values(apply_group(frame, spec_a), measure, normalize=normalize)
1476 b = measure_values(apply_group(frame, spec_b), measure, normalize=normalize)
1477 if a.size:
1478 out[label_a] = a
1479 if b.size:
1480 out[label_b] = b
1481 return out
1484def group_word_difference(
1485 words: pd.DataFrame,
1486 text_col: str,
1487 text_id,
1488 measure: Measure,
1489 spec_a: Mapping,
1490 spec_b: Mapping,
1491 *,
1492 agg: str = "mean",
1493 min_readers: int = 1,
1494 screen_id=None,
1495) -> pd.DataFrame:
1496 """Per-word A−B difference profile for one text (AN-19).
1498 On multipart data both cohorts describe ``screen_id`` — the text's first
1499 screen when it is ``None`` (BUG-114). Returns
1500 ``[word_id, a, b, diff, n_a, n_b, enough, word_text]``.
1501 """
1502 cols = ["word_id", "a", "b", "diff", "n_a", "n_b", "enough", "word_text"]
1503 screen_id = _shared_screen(words, text_col, text_id, screen_id)
1504 pa = cohort_word_profile(
1505 apply_group(words, spec_a),
1506 text_col,
1507 text_id,
1508 measure,
1509 agg=agg,
1510 min_readers=min_readers,
1511 screen_id=screen_id,
1512 )
1513 pb = cohort_word_profile(
1514 apply_group(words, spec_b),
1515 text_col,
1516 text_id,
1517 measure,
1518 agg=agg,
1519 min_readers=min_readers,
1520 screen_id=screen_id,
1521 )
1522 if pa.empty and pb.empty:
1523 return pd.DataFrame(columns=cols)
1524 a = pa[["word_id", "value", "n", "word_text"]].rename(
1525 columns={"value": "a", "n": "n_a"}
1526 )
1527 b = pb[["word_id", "value", "n"]].rename(columns={"value": "b", "n": "n_b"})
1528 out = a.merge(b, on="word_id", how="outer")
1529 out["diff"] = out["a"] - out["b"]
1530 # `a` always carries word_text (cohort_word_profile always returns it), so the
1531 # outer merge leaves NaN word_text on B-only words — backfill, don't no-op.
1532 out["word_text"] = out["word_text"].fillna("")
1533 out["n_a"] = out["n_a"].fillna(0).astype(int)
1534 out["n_b"] = out["n_b"].fillna(0).astype(int)
1535 out["enough"] = (out["n_a"] >= min_readers) & (out["n_b"] >= min_readers)
1536 return out[cols].sort_values("word_id").reset_index(drop=True)
1539def two_group_word_profiles(
1540 words: pd.DataFrame,
1541 text_col: str,
1542 text_id,
1543 measure: Measure,
1544 spec_a: Mapping,
1545 spec_b: Mapping,
1546 *,
1547 agg: str = "mean",
1548 label_a: str = "Group A",
1549 label_b: str = "Group B",
1550 screen_id=None,
1551) -> pd.DataFrame:
1552 """Long ``[group, word_id, value]`` for the stacked two-group heatmap (AN-22).
1554 Both cohorts describe one screen, as in :func:`group_word_difference`."""
1555 label_a, label_b = distinct_group_labels(label_a, label_b)
1556 screen_id = _shared_screen(words, text_col, text_id, screen_id)
1557 frames = []
1558 for spec, label in ((spec_a, label_a), (spec_b, label_b)):
1559 prof = cohort_word_profile(
1560 apply_group(words, spec),
1561 text_col,
1562 text_id,
1563 measure,
1564 agg=agg,
1565 screen_id=screen_id,
1566 )
1567 if not prof.empty:
1568 frames.append(prof[["word_id", "value"]].assign(group=label))
1569 if not frames:
1570 return pd.DataFrame(columns=["group", "word_id", "value"])
1571 return pd.concat(frames, ignore_index=True)[["group", "word_id", "value"]]
1574def paired_group_summary(
1575 frame: pd.DataFrame,
1576 measures: Sequence[Measure],
1577 spec_a: Mapping,
1578 spec_b: Mapping,
1579 *,
1580 agg: str = "mean",
1581 label_a: str = "Group A",
1582 label_b: str = "Group B",
1583 words: pd.DataFrame | None = None,
1584 fixations: pd.DataFrame | None = None,
1585 normalize: bool = False,
1586) -> pd.DataFrame:
1587 """Per-measure group summaries for the paired bars (AN-20).
1589 Returns ``[measure, group, value, n_observations]``: ``value`` is ``agg``
1590 over every word or fixation of the group pooled across participants, and
1591 ``n_observations`` counts those pooled values. No error bars this release
1592 (#374): a spread over pooled words from the same few participants reads as
1593 a precision the data does not have. ``measures`` may mix word- and
1594 fixation-level measures; pass ``words``/``fixations`` so each reads its
1595 backing frame (``frame`` is the fallback).
1596 """
1597 label_a, label_b = distinct_group_labels(label_a, label_b)
1598 rows = []
1599 for m in measures:
1600 src = (
1601 (words if m.frame == "words" else fixations)
1602 if (words is not None or fixations is not None)
1603 else frame
1604 )
1605 if src is None:
1606 continue
1607 for spec, label in ((spec_a, label_a), (spec_b, label_b)):
1608 vals = measure_values(apply_group(src, spec), m, normalize=normalize)
1609 rows.append(
1610 {
1611 "measure": m.label,
1612 "group": label,
1613 "value": aggregate_value(vals, agg),
1614 "n_observations": int(vals.size),
1615 }
1616 )
1617 return pd.DataFrame(rows, columns=["measure", "group", "value", "n_observations"])
1620def group_mean_difference(
1621 values_a: np.ndarray,
1622 values_b: np.ndarray,
1623) -> dict[str, object]:
1624 """Descriptive two-group summary: the means, their difference, and Cohen's *d*.
1626 Returns ``mean_a``, ``mean_b``, ``n_a``, ``n_b``, ``mean_diff`` and
1627 ``cohen_d`` (pooled-SD standardized difference). Purely descriptive — no
1628 significance test. ``cohen_d`` assumes two separate sets of values; the
1629 caller decides whether to show it (the Groups view does not when the same
1630 readers are in both groups).
1631 """
1632 a = np.asarray(values_a, dtype="float64")
1633 a = a[~np.isnan(a)]
1634 b = np.asarray(values_b, dtype="float64")
1635 b = b[~np.isnan(b)]
1636 out: dict[str, object] = {
1637 "mean_a": float(np.mean(a)) if a.size else float("nan"),
1638 "mean_b": float(np.mean(b)) if b.size else float("nan"),
1639 "n_a": int(a.size),
1640 "n_b": int(b.size),
1641 "cohen_d": float("nan"),
1642 }
1643 out["mean_diff"] = out["mean_a"] - out["mean_b"]
1644 if a.size < 2 or b.size < 2:
1645 return out
1646 # Pooled-SD Cohen's d.
1647 va, vb = np.var(a, ddof=1), np.var(b, ddof=1)
1648 pooled = np.sqrt(((a.size - 1) * va + (b.size - 1) * vb) / (a.size + b.size - 2))
1649 # NaN (not 0.0) when pooled SD is 0 but the means differ — 0.0 would falsely
1650 # read as "no effect" next to a non-zero mean difference. Matches the n<2 /
1651 # empty-input sentinels above.
1652 out["cohen_d"] = (
1653 float((np.mean(a) - np.mean(b)) / pooled) if pooled > 0 else float("nan")
1654 )
1655 return out