Coverage for scanpath_studio/preprocessing.py: 94%
398 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Optional reading-data preprocessing and derived analysis tables.
3The stage is deliberately pure and disabled by default (PRE-1). It soft-marks
4excluded rows, preserves originals for any changed value, and exposes the same
5functions to the UI, CLI, public API, and bulk export.
6"""
8from __future__ import annotations
10import itertools
11import math
12from collections.abc import Mapping, Sequence
14import numpy as np
15import pandas as pd
17from .measures import materialize_runs
18from .multipart import SCREEN_ID, grouping_columns
20DEFAULT_PREPROCESSING = {
21 "enabled": False,
22 "short_policy": "Off",
23 "short_threshold_ms": 80.0,
24 "merge_distance_chars": 1.0,
25 "discard_blink_adjacent": True,
26}
29def detect_right_to_left(text: str) -> bool:
30 """Majority-script direction detection for Hebrew/Arabic trial text."""
31 rtl = ltr = 0
32 for char in str(text or ""):
33 code = ord(char)
34 if 0x0590 <= code <= 0x08FF or 0xFB1D <= code <= 0xFEFC:
35 rtl += 1
36 elif char.isalpha():
37 ltr += 1
38 return rtl > ltr
41def add_text_direction(words: pd.DataFrame) -> pd.DataFrame:
42 """Attach ``right_to_left`` unless the dataset already supplies it (PRE-6)."""
43 if words.empty or "right_to_left" in words or "text" not in words:
44 return words
45 out = words.copy()
46 keys = grouping_columns(out)
47 if keys:
48 # Detect ONCE PER TRIAL and let `transform` broadcast the scalar, rather
49 # than materialising the whole trial's joined text on every row and then
50 # scanning it per row. The old form was quadratic in trial length — for a
51 # 500-word trial it scanned ~500x more characters than it needed to, and
52 # on an ordinary upload it was the single slowest step of the load
53 # (measured 18.3s -> 0.30s on 144k words), all of it *after* the wizard
54 # had already said "Dataset added" — so it read as the app hanging.
55 # `dropna` (BUG-53): a frame that did not come through `normalize_words`
56 # can still carry a missing word, and `" ".join` raises on the float.
57 out["right_to_left"] = out.groupby(keys, sort=False)["text"].transform(
58 lambda values: detect_right_to_left(" ".join(values.dropna().astype(str)))
59 )
60 else:
61 out["right_to_left"] = detect_right_to_left(
62 " ".join(out["text"].dropna().astype(str))
63 )
64 return out
67def _character_width(words: pd.DataFrame, word_id) -> float:
68 """One character's width for this word — BUG-27's shared letter scale.
70 This scales the *merge distance*, which the user sets in **characters** and
71 which travels on the share link and the saved config
72 (`session_keys.GLOBAL_PREPROC_MERGE_DISTANCE_CHARS`). It used to divide by
73 `len(text)`, so on a tiling corpus "merge within 1 character" silently meant
74 1.25 characters for a four-letter word and 1.07 for a fifteen-letter one —
75 the same length-dependent error BUG-27 fixed in the landing measures, on a
76 setting whose whole point is to mean one thing.
77 """
78 from .measures import word_char_advance
80 match = words[pd.to_numeric(words.get("word_id"), errors="coerce") == word_id]
81 if match.empty:
82 return 1.0
83 # `layout=words`: tiling is a property of the whole line, and a one-row
84 # subset reads as glyph-tight.
85 advance = word_char_advance(match, layout=words)
86 value = float(advance[0]) if len(advance) else np.nan
87 return max(value, 1.0) if np.isfinite(value) else 1.0
90def merge_short_fixations(
91 fixations: pd.DataFrame,
92 words: pd.DataFrame,
93 *,
94 threshold_ms: float = 80.0,
95 distance_chars: float = 1.0,
96 discard_unmerged: bool | str = False,
97) -> tuple[pd.DataFrame, int, int]:
98 """Merge short-and-close fixations into a temporal neighbour (PRE-14)."""
99 if fixations.empty:
100 return fixations.copy(), 0, 0
101 out = fixations.sort_values(
102 grouping_columns(fixations)
103 + (["timestamp_ms"] if "timestamp_ms" in fixations else []),
104 kind="stable",
105 ).copy()
106 out["original_duration_ms"] = pd.to_numeric(out["duration_ms"], errors="coerce")
107 excluded = (
108 out.get("excluded", pd.Series(False, index=out.index)).astype(bool).copy()
109 )
110 reasons = (
111 out.get("excluded_reason", pd.Series("", index=out.index))
112 .fillna("")
113 .astype(str)
114 .copy()
115 )
116 merged = dropped = 0
117 keys = grouping_columns(out)
118 groups = out.groupby(keys, sort=False).groups.values() if keys else [out.index]
119 for indices in groups:
120 indices = list(indices)
121 for position, index in enumerate(indices):
122 duration = float(
123 pd.to_numeric(out.at[index, "duration_ms"], errors="coerce")
124 )
125 if (
126 not np.isfinite(duration)
127 or duration >= threshold_ms
128 or excluded.at[index]
129 ):
130 continue
131 candidates = []
132 for neighbor_pos in (position - 1, position + 1):
133 if 0 <= neighbor_pos < len(indices):
134 neighbor = indices[neighbor_pos]
135 if not excluded.at[neighbor]:
136 distance = abs(
137 float(out.at[index, "x"]) - float(out.at[neighbor, "x"])
138 )
139 char_width = _character_width(
140 words,
141 pd.to_numeric(out.at[index, "word_id"], errors="coerce"),
142 )
143 if distance <= distance_chars * char_width:
144 candidates.append((distance, neighbor))
145 if candidates:
146 _, target = min(candidates)
147 out.at[target, "duration_ms"] = (
148 float(out.at[target, "duration_ms"]) + duration
149 )
150 excluded.at[index] = True
151 reasons.at[index] = "merged_short"
152 merged += 1
153 elif discard_unmerged is True or (
154 discard_unmerged == "terminal" and position == len(indices) - 1
155 ):
156 excluded.at[index] = True
157 reasons.at[index] = "short_unmerged"
158 dropped += 1
159 out["excluded"] = excluded
160 out["excluded_reason"] = reasons
161 return out.sort_index(), merged, dropped
164def _blink_flags(fixations: pd.DataFrame) -> pd.Series:
165 for column in ("is_blink", "blink", "blink_flag"):
166 if column in fixations:
167 return fixations[column].fillna(False).astype(bool)
168 return pd.Series(False, index=fixations.index)
171def preprocess_fixations(
172 fixations: pd.DataFrame,
173 words: pd.DataFrame,
174 *,
175 settings: Mapping | None = None,
176) -> tuple[pd.DataFrame, pd.DataFrame]:
177 """Run the optional PRE-1 stage and return ``(fixations, QA report)``.
179 When disabled, the original dataframe is returned byte-for-byte and the
180 report is empty. Enabled rows are never hard-dropped; ``excluded`` and
181 ``excluded_reason`` retain the audit trail.
182 """
183 cfg = {**DEFAULT_PREPROCESSING, **dict(settings or {})}
184 if not cfg["enabled"]:
185 return fixations, pd.DataFrame()
186 out = fixations.copy()
187 out["excluded"] = out.get("excluded", False)
188 out["excluded_reason"] = out.get("excluded_reason", "")
189 blink = _blink_flags(out)
190 out["is_blink"] = blink
191 keys = grouping_columns(out)
192 if keys:
193 grouped = out.groupby(keys, sort=False)
194 out["blink_before"] = grouped["is_blink"].shift(-1).fillna(False).astype(bool)
195 out["blink_after"] = grouped["is_blink"].shift(1).fillna(False).astype(bool)
196 else:
197 out["blink_before"] = blink.shift(-1).fillna(False)
198 out["blink_after"] = blink.shift(1).fillna(False)
199 if cfg["discard_blink_adjacent"]:
200 mask = out["blink_before"] | out["blink_after"] | out["is_blink"]
201 out.loc[mask, "excluded"] = True
202 out.loc[mask, "excluded_reason"] = "blink_adjacent"
204 policy = str(cfg["short_policy"])
205 merged = dropped = 0
206 if policy in {"Merge", "Merge then discard"}:
207 prior_excluded = out["excluded"].copy()
208 out, merged, dropped = merge_short_fixations(
209 out,
210 words,
211 threshold_ms=float(cfg["short_threshold_ms"]),
212 distance_chars=float(cfg["merge_distance_chars"]),
213 discard_unmerged=True if policy == "Merge then discard" else "terminal",
214 )
215 out["excluded"] = out["excluded"] | prior_excluded
216 elif policy == "Discard":
217 mask = pd.to_numeric(out["duration_ms"], errors="coerce") < float(
218 cfg["short_threshold_ms"]
219 )
220 out.loc[mask, "excluded"] = True
221 out.loc[mask, "excluded_reason"] = "short"
222 dropped = int(mask.sum())
223 out = materialize_runs(out)
224 report = cleaning_report(out, short_policy=policy)
225 if not report.empty:
226 report["n_merged"] = merged
227 report["n_short_discarded"] = dropped
228 return out, report
231def cleaning_report(
232 fixations: pd.DataFrame, *, short_policy: str = "Off", high_word_fixations: int = 12
233) -> pd.DataFrame:
234 """One exportable QA/provenance row per trial (PRE-15)."""
235 if fixations.empty:
236 return pd.DataFrame()
237 keys = grouping_columns(fixations)
238 rows = []
239 for identity, chunk in fixations.groupby(keys, sort=False, dropna=False):
240 identity = identity if isinstance(identity, tuple) else (identity,)
241 excluded = chunk.get("excluded", pd.Series(False, index=chunk.index)).astype(
242 bool
243 )
244 reasons = chunk.get("excluded_reason", pd.Series("", index=chunk.index)).astype(
245 str
246 )
247 retained = chunk.loc[~excluded]
248 word_counts = (
249 retained.dropna(subset=["word_id"]).groupby("word_id").size()
250 if "word_id" in chunk
251 else pd.Series(dtype=int)
252 )
253 n = len(chunk)
254 row = dict(zip(keys, identity))
255 row.update(
256 n_fixations_before=n,
257 n_fixations_after=int((~excluded).sum()),
258 n_excluded=int(excluded.sum()),
259 excluded_pct=float(excluded.mean()) if n else 0.0,
260 n_blink_adjacent=int(reasons.eq("blink_adjacent").sum()),
261 blink_adjacent_pct=float(reasons.eq("blink_adjacent").mean()) if n else 0.0,
262 n_short_discarded=int(reasons.isin(["short", "short_unmerged"]).sum()),
263 short_discarded_pct=float(reasons.isin(["short", "short_unmerged"]).mean())
264 if n
265 else 0.0,
266 n_merged=int(reasons.eq("merged_short").sum()),
267 merged_pct=float(reasons.eq("merged_short").mean()) if n else 0.0,
268 short_policy=short_policy,
269 max_fixations_on_word=int(word_counts.max())
270 if not word_counts.empty
271 else 0,
272 suspicious_word_load=bool(
273 not word_counts.empty and int(word_counts.max()) >= high_word_fixations
274 ),
275 )
276 rows.append(row)
277 return pd.DataFrame(rows)
280def infer_sentence_ids(words: pd.DataFrame) -> pd.DataFrame:
281 """Preserve dataset sentence ids or infer them from terminal punctuation."""
282 if words.empty or "sentence_id" in words:
283 return words.copy()
284 out = words.copy()
285 keys = grouping_columns(out)
286 pieces = []
287 for _, chunk in out.groupby(keys, sort=False, dropna=False):
288 chunk = chunk.sort_values([c for c in ("line_idx", "word_id") if c in chunk])
289 endings = chunk["text"].astype(str).str.contains(r"[.!?][\"'”’)]*$", regex=True)
290 chunk["sentence_id"] = endings.shift(fill_value=False).cumsum() + 1
291 pieces.append(chunk)
292 return pd.concat(pieces).sort_index()
295def sentence_measures(words: pd.DataFrame, fixations: pd.DataFrame) -> pd.DataFrame:
296 """First-class per-sentence reading table derived from run structure (PRE-11)."""
297 words = infer_sentence_ids(words)
298 if words.empty:
299 return pd.DataFrame()
300 identity = grouping_columns(words)
301 word_keys = [*identity, "word_id"]
302 join = words[[*word_keys, "sentence_id"]].drop_duplicates()
303 fix = materialize_runs(fixations).merge(join, on=word_keys, how="left")
304 rows = []
305 included = (
306 ~fix["excluded"].fillna(False).astype(bool)
307 if "excluded" in fix
308 else pd.Series(True, index=fix.index)
309 )
310 analysis_fix = fix.loc[included]
311 sentence_keys = [*identity, "sentence_id"]
312 for group_key, wchunk in words.groupby(sentence_keys, sort=False):
313 values = group_key if isinstance(group_key, tuple) else (group_key,)
314 selected = dict(zip(sentence_keys, values))
315 sentence_id = selected["sentence_id"]
316 fx_mask = pd.Series(True, index=analysis_fix.index)
317 for column, value in selected.items():
318 fx_mask &= analysis_fix[column] == value
319 fx = analysis_fix[fx_mask]
320 first_run = pd.to_numeric(
321 fx.get("word_run", pd.Series(1, index=fx.index)), errors="coerce"
322 ).fillna(1)
323 first = fx[first_run.eq(1)]
324 reread = fx.drop(first.index)
325 duration = pd.to_numeric(
326 fx.get("duration_ms", pd.Series(np.nan, index=fx.index)), errors="coerce"
327 ).sum()
328 n_words = int(wchunk["word_id"].nunique())
329 trial_mask = pd.Series(True, index=analysis_fix.index)
330 for column in identity:
331 trial_mask &= analysis_fix[column] == selected[column]
332 trial_sequence = analysis_fix[trial_mask].sort_values("timestamp_ms")
333 sentence_sequence = pd.to_numeric(
334 trial_sequence.get("sentence_id"), errors="coerce"
335 )
336 transitions = sentence_sequence.diff()
337 reg_in = bool(((sentence_sequence == sentence_id) & transitions.lt(0)).any())
338 reg_out = bool(
339 ((sentence_sequence.shift() == sentence_id) & transitions.lt(0)).any()
340 )
341 first_position = np.flatnonzero(sentence_sequence.to_numpy() == sentence_id)
342 gopast = np.nan
343 gopast_sel = np.nan
344 firstpass = trial_sequence.iloc[0:0]
345 first_forward = trial_sequence.iloc[0:0]
346 first_reread = trial_sequence.iloc[0:0]
347 lookback = trial_sequence.iloc[0:0]
348 lookfrom = trial_sequence.iloc[0:0]
349 if len(first_position):
350 start = int(first_position[0])
351 later = np.flatnonzero(
352 sentence_sequence.iloc[start:].to_numpy() > sentence_id
353 )
354 stop = start + int(later[0]) if len(later) else len(trial_sequence)
355 window = trial_sequence.iloc[start:stop]
356 window_sentence = pd.to_numeric(window["sentence_id"], errors="coerce")
357 firstpass = window[window_sentence.eq(sentence_id)]
358 lookback = window[window_sentence.lt(sentence_id)]
359 if lookback.empty:
360 first_forward = firstpass
361 else:
362 first_lookback_time = lookback.index[0]
363 before = window.index.get_loc(first_lookback_time)
364 forward_window = window.iloc[:before]
365 reread_window = window.iloc[before:]
366 first_forward = forward_window[
367 pd.to_numeric(forward_window["sentence_id"], errors="coerce").eq(
368 sentence_id
369 )
370 ]
371 first_reread = reread_window[
372 pd.to_numeric(reread_window["sentence_id"], errors="coerce").eq(
373 sentence_id
374 )
375 ]
376 lookfrom = trial_sequence.iloc[stop:][
377 pd.to_numeric(
378 trial_sequence.iloc[stop:]["sentence_id"], errors="coerce"
379 ).eq(sentence_id)
380 ]
381 gopast = float(pd.to_numeric(window["duration_ms"], errors="coerce").sum())
382 selective = window[
383 pd.to_numeric(window["sentence_id"], errors="coerce") >= sentence_id
384 ]
385 gopast_sel = float(
386 pd.to_numeric(selective["duration_ms"], errors="coerce").sum()
387 )
388 rows.append(
389 {
390 **{column: selected[column] for column in identity},
391 "text_id": wchunk.iloc[0].get(
392 "text_id", wchunk.iloc[0].get("unique_text_id", pd.NA)
393 ),
394 "sentence_id": sentence_id,
395 "n_words": n_words,
396 "skip": fx.empty,
397 "nrun": int(fx["word_runid"].nunique())
398 if not fx.empty and "word_runid" in fx
399 else 0,
400 "reread": not reread.empty,
401 "reg_in": reg_in,
402 "reg_out": reg_out,
403 "total_n_fixations": len(fx),
404 "total_dur": float(duration),
405 "rate_wpm": n_words / (duration / 60_000) if duration > 0 else np.nan,
406 "firstpass_n_fixations": len(firstpass),
407 "firstpass_dur": float(
408 pd.to_numeric(
409 firstpass.get(
410 "duration_ms", pd.Series(np.nan, index=firstpass.index)
411 ),
412 errors="coerce",
413 ).sum()
414 ),
415 "gopast": gopast,
416 "gopast_sel": gopast_sel,
417 "firstpass_forward_n_fixations": len(first_forward),
418 "firstpass_forward_dur": float(
419 pd.to_numeric(
420 first_forward.get(
421 "duration_ms", pd.Series(np.nan, index=first_forward.index)
422 ),
423 errors="coerce",
424 ).sum()
425 ),
426 "firstpass_reread_n_fixations": len(first_reread),
427 "firstpass_reread_dur": float(
428 pd.to_numeric(
429 first_reread.get(
430 "duration_ms", pd.Series(np.nan, index=first_reread.index)
431 ),
432 errors="coerce",
433 ).sum()
434 ),
435 "lookback_n_fixations": len(lookback),
436 "lookback_dur": float(
437 pd.to_numeric(
438 lookback.get(
439 "duration_ms", pd.Series(np.nan, index=lookback.index)
440 ),
441 errors="coerce",
442 ).sum()
443 ),
444 "lookfrom_n_fixations": len(lookfrom),
445 "lookfrom_dur": float(
446 pd.to_numeric(
447 lookfrom.get(
448 "duration_ms", pd.Series(np.nan, index=lookfrom.index)
449 ),
450 errors="coerce",
451 ).sum()
452 ),
453 }
454 )
455 return pd.DataFrame(rows)
458def character_grid(words: pd.DataFrame) -> pd.DataFrame:
459 """Expand word AOIs to comparable letter-position geometry (PRE-19)."""
460 rows = []
461 words = infer_sentence_ids(words)
462 if words.empty:
463 return pd.DataFrame()
464 if "line_idx" not in words:
465 from .measures import cluster_word_lines
467 words = words.copy()
468 words["line_idx"] = cluster_word_lines(words)
469 keys = grouping_columns(words)
470 groups = words.groupby(keys, sort=False, dropna=False) if keys else [((), words)]
471 for _, trial in groups:
472 trial = trial.sort_values(["line_idx", "word_id"], kind="stable")
473 global_position = 0
474 line_positions: dict[object, int] = {}
475 words_seen: dict[object, int] = {}
476 # One advance per word, resolved once for the trial — tiling detection is
477 # a property of the layout, so it must not run per word.
478 from .measures import word_char_advance
480 advances = dict(zip(trial.index, word_char_advance(trial, layout=words)))
481 for word in trial.itertuples():
482 text = str(getattr(word, "text", ""))
483 if not text:
484 continue
485 line_id = getattr(word, "line_idx", 0)
486 if words_seen.get(line_id, 0):
487 line_positions[line_id] += 1 # inter-word space
488 global_position += 1
489 words_seen[line_id] = words_seen.get(line_id, 0) + 1
490 line_positions.setdefault(line_id, 0)
491 # BUG-27: one advance, not `width / len(text)` — on a tiling corpus
492 # the latter stretches the glyph row across the trailing padding, so
493 # every character box after the first sat progressively too far right.
494 char_width = advances.get(word.Index, float(word.width))
495 rtl = bool(getattr(word, "right_to_left", False))
496 for physical_offset in range(len(text)):
497 logical_offset = (
498 len(text) - physical_offset - 1 if rtl else physical_offset
499 )
500 char = text[logical_offset]
501 x = float(word.x) + physical_offset * char_width
502 letword = logical_offset + 1
503 letline = line_positions[line_id] + letword
504 letternum = global_position + letword
505 rows.append(
506 {
507 "participant_id": getattr(word, "participant_id", pd.NA),
508 "trial_id": getattr(word, "trial_id", pd.NA),
509 **(
510 {SCREEN_ID: getattr(word, SCREEN_ID)}
511 if hasattr(word, SCREEN_ID)
512 else {}
513 ),
514 "word_id": word.word_id,
515 "sentence_id": getattr(word, "sentence_id", pd.NA),
516 "line_idx": line_id,
517 "right_to_left": rtl,
518 "character": char,
519 "letternum": letternum,
520 "letline": letline,
521 "letword": letword,
522 "x": x,
523 "y": float(word.y),
524 "center_x": x + char_width / 2.0,
525 "center_y": float(word.y) + float(word.height) / 2.0,
526 "width": char_width,
527 "height": float(word.height),
528 }
529 )
530 line_positions[line_id] += len(text)
531 global_position += len(text)
532 return pd.DataFrame(rows)
535def duration_mass_table(
536 words: pd.DataFrame,
537 fixations: pd.DataFrame,
538 *,
539 sigma_chars: float = 1.0,
540) -> pd.DataFrame:
541 """Distribute fixation duration over nearby character centres (PRE-8).
543 Each fixation contributes a Gaussian kernel whose weights sum to its
544 duration. The returned character table is therefore a reproducible support
545 surface and preserves total included dwell time.
546 """
547 chars = character_grid(words)
548 if chars.empty:
549 return chars.assign(duration_mass_ms=pd.Series(dtype=float))
550 out = chars.copy()
551 out["duration_mass_ms"] = 0.0
552 analysis = fixations
553 if "excluded" in analysis:
554 analysis = analysis.loc[~analysis["excluded"].fillna(False).astype(bool)]
555 keys = grouping_columns(analysis)
556 if keys != grouping_columns(out):
557 raise ValueError("Character and fixation tables use different part identities.")
558 groups = (
559 analysis.groupby(keys, sort=False, dropna=False) if keys else [((), analysis)]
560 )
561 for identity, chunk in groups:
562 identity = identity if isinstance(identity, tuple) else (identity,)
563 support = out
564 for key, value in zip(keys, identity):
565 support = support[support[key] == value]
566 if support.empty:
567 continue
568 widths = pd.to_numeric(support["width"], errors="coerce").dropna()
569 sigma_px = max(float(widths.median()) * float(sigma_chars), 1e-6)
570 sx = support["center_x"].to_numpy(dtype=float)
571 sy = support["center_y"].to_numpy(dtype=float)
572 for fixation in chunk.itertuples():
573 x = pd.to_numeric(getattr(fixation, "x", np.nan), errors="coerce")
574 y = pd.to_numeric(getattr(fixation, "y", np.nan), errors="coerce")
575 duration = pd.to_numeric(
576 getattr(fixation, "duration_ms", np.nan), errors="coerce"
577 )
578 if pd.isna(x) or pd.isna(y) or pd.isna(duration) or duration <= 0:
579 continue
580 kernel = np.exp(
581 -((sx - float(x)) ** 2 + (sy - float(y)) ** 2) / (2 * sigma_px**2)
582 )
583 total = float(kernel.sum())
584 if total <= 0:
585 kernel[int(np.argmin((sx - float(x)) ** 2 + (sy - float(y)) ** 2))] = (
586 1.0
587 )
588 total = 1.0
589 out.loc[support.index, "duration_mass_ms"] += (
590 kernel / total * float(duration)
591 )
592 return out
595def _word_letter_geometry(
596 words: pd.DataFrame | None, keys: Sequence[str]
597) -> dict[tuple, tuple[float, float, float, bool]]:
598 """``(identity…, word_id) -> (x, glyph run, character advance, right_to_left)``.
600 Built once per :func:`saccade_table` call so the per-saccade letter-position
601 lookup is a dict hit rather than a scan of the whole words frame (PERF-3).
602 The first row wins for a duplicated key, which is what the old
603 ``target.iloc[0]`` did.
605 BUG-27: the advance comes from :func:`measures.word_char_advance` rather than
606 the local ``width / len(text)`` this used to divide by — the same accessor
607 the per-word landing measures use, so a saccade's launch/landing letter and
608 the word's landing position are on one scale (they disagreed by the
609 inter-word padding on a tiling corpus). The second slot is the **glyph run**
610 (``n × advance``) and not the box ``width`` for the same reason: it is only
611 read to mirror an RTL offset, and the padded width would put an RTL landing
612 one whole advance late.
613 """
614 if words is None or words.empty or "word_id" not in words:
615 return {}
616 from .measures import word_char_advance, word_char_counts
618 ids = pd.to_numeric(words["word_id"], errors="coerce")
619 present = [key for key in keys if key in words]
620 has_text = "text" in words
621 counts = word_char_counts(words) if has_text else None
622 advances = word_char_advance(words, chars=counts) if has_text else None
623 rtl = words.get("right_to_left", None)
624 geometry: dict[tuple, tuple[float, float, float, bool]] = {}
625 for position, word_id in enumerate(ids.to_numpy()):
626 if pd.isna(word_id):
627 continue
628 key = tuple(words[column].iat[position] for column in present) + (
629 float(word_id),
630 )
631 if key in geometry:
632 continue
633 width = float(words["width"].iat[position])
634 advance = float(advances[position]) if advances is not None else width
635 run = advance * float(counts[position]) if counts is not None else width
636 geometry[key] = (
637 float(words["x"].iat[position]),
638 run,
639 advance,
640 bool(rtl.iat[position]) if rtl is not None else False,
641 )
642 return geometry
645def _letter_position_in_word(
646 event: dict,
647 geometry: dict[tuple, tuple[float, float, int, bool]],
648 identity: tuple,
649) -> float:
650 """Which character of its word a fixation landed on, 1-based."""
651 word_id = event.get("word_id")
652 if not geometry or pd.isna(word_id):
653 return np.nan
654 found = geometry.get(identity + (float(word_id),))
655 if found is None: # a words frame without the identity columns
656 found = geometry.get((float(word_id),))
657 if found is None:
658 return np.nan
659 word_x, glyph_run, advance, right_to_left = found
660 if not np.isfinite(advance) or advance <= 0:
661 return np.nan
662 offset = float(event["x"]) - word_x
663 if right_to_left:
664 offset = glyph_run - offset
665 return offset / advance + 1
668def saccade_table(
669 fixations: pd.DataFrame,
670 *,
671 pixels_per_degree: float | None = None,
672 raw_gaze: pd.DataFrame | None = None,
673 words: pd.DataFrame | None = None,
674) -> pd.DataFrame:
675 """Materialize one row per saccade with geometry and visual-angle units."""
676 if fixations.empty:
677 return pd.DataFrame()
678 out = []
679 analysis = fixations
680 if "excluded" in analysis:
681 analysis = analysis.loc[~analysis["excluded"].fillna(False).astype(bool)]
682 keys = grouping_columns(analysis)
683 # PERF-3: the letter-position of a saccade's launch and landing used to be
684 # resolved by scanning the WHOLE words frame per fixation — a `to_numeric`
685 # over every word id plus a boolean mask, twice per saccade. On the bundled
686 # demo that was ~19 s of a rerun on its own, and it grew with corpus × trial
687 # count. Index the geometry once instead: (identity…, word_id) → the four
688 # numbers the formula needs.
689 word_geometry = _word_letter_geometry(words, keys)
690 for identity, chunk in analysis.groupby(keys, sort=False, dropna=False):
691 identity = identity if isinstance(identity, tuple) else (identity,)
692 chunk = chunk.sort_values("timestamp_ms")
693 records = chunk.to_dict("records")
694 for start, end in itertools.pairwise(records):
695 dx, dy = float(end["x"] - start["x"]), float(end["y"] - start["y"])
696 distance = math.hypot(dx, dy)
697 start_duration = pd.to_numeric(start.get("duration_ms", 0), errors="coerce")
698 start_end = float(start["timestamp_ms"]) + (
699 float(start_duration) if pd.notna(start_duration) else 0.0
700 )
701 duration = max(float(end["timestamp_ms"]) - start_end, 0.0)
702 peak_velocity = np.nan
703 if raw_gaze is not None and not raw_gaze.empty and pixels_per_degree:
704 gaze = raw_gaze
705 for key, value in zip(keys, identity):
706 if key in gaze:
707 gaze = gaze[gaze[key] == value]
708 if "timestamp_ms" in gaze:
709 gaze = gaze[
710 pd.to_numeric(gaze["timestamp_ms"], errors="coerce").between(
711 start_end, float(end["timestamp_ms"])
712 )
713 ].sort_values("timestamp_ms")
714 if len(gaze) >= 2:
715 dt = pd.to_numeric(gaze["timestamp_ms"], errors="coerce").diff()
716 gd = np.hypot(
717 pd.to_numeric(gaze["x"], errors="coerce").diff(),
718 pd.to_numeric(gaze["y"], errors="coerce").diff(),
719 )
720 speed = gd / dt.replace(0, np.nan) / pixels_per_degree * 1000
721 peak_velocity = float(speed.max())
723 def _letter_position(event, _identity=identity):
724 return _letter_position_in_word(event, word_geometry, _identity)
726 row = dict(zip(keys, identity))
727 row.update(
728 xs=start["x"],
729 ys=start["y"],
730 xe=end["x"],
731 ye=end["y"],
732 dX=dx,
733 dY=dy,
734 distance_px=distance,
735 duration_ms=duration,
736 angle=math.degrees(math.atan2(-dy, dx)),
737 amplitude_deg=distance / pixels_per_degree
738 if pixels_per_degree and pixels_per_degree > 0
739 else np.nan,
740 peak_velocity_deg_s=peak_velocity,
741 blink_before=bool(start.get("blink_before", False)),
742 blink_after=bool(end.get("blink_after", False)),
743 launch_word_id=start.get("word_id"),
744 landing_word_id=end.get("word_id"),
745 launch_line=start.get("line_id", start.get("line_idx")),
746 landing_line=end.get("line_id", end.get("line_idx")),
747 launch_letter=_letter_position(start),
748 landing_letter=_letter_position(end),
749 )
750 out.append(row)
751 return pd.DataFrame(out)
754def measure_sensitivity(
755 words: pd.DataFrame,
756 fixations: pd.DataFrame,
757 methods: tuple[str, ...] = ("attach", "slice", "consensus"),
758) -> tuple[pd.DataFrame, pd.DataFrame]:
759 """Word-measure spread under several line assignments (PRE-18)."""
760 from . import alignment
761 from .measures import compute_per_word_measures
763 trial_keys = grouping_columns(fixations)
764 if trial_keys != grouping_columns(words):
765 raise ValueError("Words and fixations use different part identities.")
766 combined_parts = []
767 report_parts = []
768 groups = (
769 fixations.groupby(trial_keys, sort=False, dropna=False)
770 if trial_keys
771 else [((), fixations)]
772 )
773 for identity, trial_fixations in groups:
774 identity = identity if isinstance(identity, tuple) else (identity,)
775 trial_words = words
776 for key, value in zip(trial_keys, identity):
777 trial_words = trial_words[trial_words[key] == value]
778 if trial_words.empty:
779 continue
780 trial_combined, trial_report = alignment.correction_sensitivity(
781 trial_fixations, trial_words, methods
782 )
783 for key, value in zip(trial_keys, identity):
784 trial_report[key] = value
785 combined_parts.append(trial_combined)
786 report_parts.append(trial_report)
787 combined = (
788 pd.concat(combined_parts).sort_index() if combined_parts else fixations.copy()
789 )
790 correction_report = (
791 pd.concat(report_parts, ignore_index=True)
792 if report_parts
793 else pd.DataFrame(columns=[*trial_keys, "algorithm"])
794 )
795 keys = grouping_columns(words, include_word=True)
796 merged = words[keys].drop_duplicates().copy()
797 headline = (
798 "first_fixation_ms",
799 "first_pass_gaze_duration_ms",
800 "regression_path_duration_ms",
801 "total_fixation_duration_ms",
802 )
803 for method in methods:
804 corrected = fixations.copy()
805 corrected["y"] = combined[f"y_{method}"]
806 measured = compute_per_word_measures(corrected, words)
807 keep = [column for column in headline if column in measured]
808 measured = measured[keys + keep].rename(
809 columns={column: f"{column}_{method}" for column in keep}
810 )
811 merged = merged.merge(measured, on=keys, how="left")
812 for metric in headline:
813 columns = [
814 f"{metric}_{method}" for method in methods if f"{metric}_{method}" in merged
815 ]
816 if columns:
817 merged[f"{metric}_spread"] = merged[columns].max(axis=1) - merged[
818 columns
819 ].min(axis=1)
820 return merged, correction_report