Coverage for scanpath_studio/api.py: 93%
1106 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Headless programmatic API for scanpath-studio.
3The Streamlit app and this module share one pipeline (``data`` → ``measures``
4→ ``plots``), so a figure produced here goes through the exact same builders as
5the app and is pixel-identical *given the same settings*. The headless defaults
6(``CANONICAL_FIGURE_DEFAULTS``) are the app's default *Scanpath* design —
7fixations, saccades and the text — so a bare call draws the figure the app
8opens on.
9Typical use::
11 import scanpath_studio as sps
13 words, fixations = sps.load_scanpath_data("ia.csv", "fixations.csv")
14 print(sps.list_trials(words, fixations))
15 fig = sps.plot_scanpath(words, fixations, participant="p1", trial="t1")
16 sps.save_figure(fig, "scanpath.html") # or .png/.svg/.pdf (needs Chrome)
18Every keyword accepted by :func:`plots.make_scanpath_figure` /
19:func:`plots.make_scanpath_animation` can be overridden through
20``plot_scanpath`` / ``animate_scanpath`` (e.g. ``show_heatmap=True``);
21:func:`figure_options` lists them with their effective defaults. ``docs/agents.md``
22is the task-oriented guide to this module for scripted / agent use.
23"""
25from __future__ import annotations
27import difflib
28import logging
29from collections.abc import Iterable, Sequence
30from copy import deepcopy
31from dataclasses import replace
32from pathlib import Path
34import pandas as pd
35import plotly.graph_objects as go
36import plotly.io as pio
38# Outside a Streamlit runtime the @st.cache_data decorators in `data` fall
39# back to bare-mode caching and log a "No runtime found" warning per cached
40# function — harmless but noisy for library/CLI users, so quiet those loggers.
41# Order matters twice over: streamlit must be imported first (its get_logger()
42# sets each module logger's level at import, clobbering anything set earlier),
43# and `.data` must be imported after (its decorators fire the warnings at
44# import time). Inside the app a runtime exists and these warnings never fire.
45import streamlit as _st # noqa: F401 (imported for its logging side effect)
47for _name in (
48 "streamlit.runtime.caching.cache_data_api",
49 "streamlit.runtime.scriptrunner_utils.script_run_context",
50):
51 logging.getLogger(_name).setLevel(logging.ERROR)
53from . import column_names as _cn # noqa: E402
54from . import data as _data # noqa: E402
55from . import export as _export # noqa: E402
56from .build_info import BuildInfo # noqa: E402
57from .column_names import ColumnNames # noqa: E402
58from .constants import ( # noqa: E402
59 DEFAULT_BACKGROUND_COLOR,
60 DEFAULT_ORDER_FONT_COLOR,
61 EXPERIMENTAL_ENV_VAR,
62 FONT_FAMILY,
63 PLOTLY_CONFIG,
64 SACCADE_CLASS_ORDER,
65 UNIFORM_COLOR_FIELD,
66 computed_measures_enabled,
67 drift_correction_enabled,
68 palette_settings,
69)
70from .experimental_setup import Provenance, SetupSnapshot # noqa: E402
71from .export import annotate_figure # noqa: E402
72from .multipart import ( # noqa: E402
73 SCREEN_ID,
74 apply_trial_parts_manifest,
75 extract_part,
76 part_catalog,
77 screen_canvas_size,
78)
79from .plots import ( # noqa: E402
80 ANIMATION_FIGURE_OPTIONS,
81 COMPARISON_FIGURE_OPTIONS,
82 FIGURE_OPTION_CHOICES,
83 STATIC_FIGURE_OPTIONS,
84 FigureSettings,
85 _resolve_trial_display_name,
86 add_illustration_label,
87 make_comparison_figure,
88 make_difference_profile_figure,
89 make_distribution_figure,
90 make_scanpath_animation,
91 make_scanpath_figure,
92 make_word_profile_figure,
93 normalize_option_value,
94 normalize_option_values,
95 normalize_palette,
96 replay_page,
97 split_scanpath_layers,
98)
99from .updates import UpdateCheck # noqa: E402
102def build_authored_scanpath(
103 text: str, events: pd.DataFrame | None = None, **layout_options
104) -> tuple[pd.DataFrame, pd.DataFrame]:
105 """Build normalized word/fixation frames from hand-authored reading events.
107 When ``events`` is omitted, one centered fixation per laid-out word is used.
108 ``layout_options`` are forwarded to `authoring.layout_text`.
109 """
110 from .authoring import authored_fixations, default_events, layout_text
112 words = layout_text(text, **layout_options)
113 if events is None:
114 events = default_events(words)
115 return words, authored_fixations(words, events)
118def load_authored_scanpath(
119 source: str | Path,
120) -> tuple[pd.DataFrame, pd.DataFrame]:
121 """Load an authoring file — the JSON the app's **Download authoring file**
122 saves — or its text, as normalized word/fixation frames."""
123 from .authoring import parse_authoring_document
125 raw = str(source)
126 if raw.lstrip().startswith("{"):
127 payload = raw
128 else:
129 payload = Path(source).read_text(encoding="utf-8")
130 document = parse_authoring_document(payload)
131 return build_authored_scanpath(
132 document.text,
133 document.events,
134 **document.layout,
135 )
138TableLike = pd.DataFrame | str | Path
139TablesLike = TableLike | list["TableLike"]
141# The headless rendering is the app's default *Scanpath* design (#374, F21):
142# fixations, saccades and the text, with word boxes, the heatmap and fixation
143# numbers off (controls._VIZ_WIDGET_DEFAULTS), so a bare `plot_scanpath` /
144# `render` draws the figure the app opens on. Turn a layer on with its keyword
145# (`show_heatmap=True`). `heatmap_metric="counts"` is translated to the
146# figure-level `None` in _figure_kwargs, like tabs._build_figure_settings.
147#
148# Every option tracks the app's own default (controls._VIZ_WIDGET_DEFAULTS →
149# controls._collect_viz_settings → tabs._build_figure_settings), so the same
150# call renders the same picture headless as on screen. `figure_options()`
151# prints the merged result.
152_FIGURE_CONTEXT_FIELDS = frozenset(
153 {"canvas_width", "canvas_height", "base_font_size", "font_family"}
154)
155_STATIC_FIGURE_PARAMS = frozenset(STATIC_FIGURE_OPTIONS) - _FIGURE_CONTEXT_FIELDS
156#: CMP-9. `layout`, `compare_stimulus`, `trial_labels` and `canvas_b` are named
157#: parameters of `compare_scanpaths`, so they are not also loose keywords.
158_COMPARISON_FIGURE_PARAMS = frozenset(COMPARISON_FIGURE_OPTIONS) - (
159 _FIGURE_CONTEXT_FIELDS | {"layout", "compare_stimulus", "trial_labels", "canvas_b"}
160)
161_ANIMATION_FIGURE_PARAMS = (
162 frozenset(ANIMATION_FIGURE_OPTIONS)
163 - _FIGURE_CONTEXT_FIELDS
164 - {"playback_speed", "autoplay"}
165) | {"fixations_b", "words_b"}
167_CANONICAL_OPTION_NAMES = {
168 "show_words",
169 "illustration_text",
170 "word_box_color",
171 "word_box_line_opacity",
172 "word_box_fill_color",
173 "word_box_fill_opacity",
174 "show_word_labels",
175 "show_fixations",
176 "show_order",
177 "show_saccades",
178 "show_saccade_arrows",
179 "show_heatmap",
180 "heatmap_style",
181 "heatmap_norm",
182 "x_field",
183 "y_field",
184 "color_by",
185 "heatmap_metric",
186 "marker_size_range",
187 "marker_size_scale",
188 "marker_duration_range",
189 "duration_size_legend",
190 "legend_layout",
191 "order_font_size",
192 "order_font_color",
193 "show_fixation_colorbar",
194 "fixation_colorbar_orientation",
195 "fixation_colorbar_tickangle",
196 "fixation_colorbar_tickfont_size",
197 "show_heatmap_colorbar",
198 "heatmap_colorbar_orientation",
199 "heatmap_colorbar_tickangle",
200 "heatmap_colorbar_tickfont_size",
201 "fixation_color_range",
202 "heatmap_range",
203 "fixation_colorscale",
204 "heatmap_colorscale",
205 "critical_span_style",
206 "highlight_column",
207 "saccade_color",
208 "saccade_style",
209 "saccade_width",
210 "saccade_color_mode",
211 "saccade_class_colors",
212 "saccade_type_legend",
213 "saccade_classes",
214 "saccade_render_mode",
215 "fixation_snap_to_word",
216 "fixation_color",
217 "fixation_symbol",
218 "fixation_opacity",
219 "background_color",
220 "color_by_line",
221 "fit_to_monitor",
222 "show_coordinate_grid",
223 "coordinate_grid_spacing",
224 "line_spacing",
225 "scale_text_to_boxes",
226 "background_image",
227 "background_image_size",
228 "background_image_origin",
229 "background_image_opacity",
230 "word_hover_fields",
231 "fixation_hover_fields",
232}
234CANONICAL_FIGURE_DEFAULTS: dict = FigureSettings.defaults(
235 _CANONICAL_OPTION_NAMES
236) | dict(
237 show_words=False,
238 show_order=False,
239 show_heatmap=False,
240 heatmap_metric="duration_ms",
241 order_font_color=DEFAULT_ORDER_FONT_COLOR,
242 saccade_classes=list(SACCADE_CLASS_ORDER),
243 fixation_opacity=0.7,
244 background_color=DEFAULT_BACKGROUND_COLOR,
245 fit_to_monitor=True,
246 # #374 F26: the A/B legend is on, as in the app.
247 show_legend=True,
248 word_hover_fields=["text", "word_id", "line_idx", "total_fixation_duration_ms"],
249 fixation_hover_fields=["order_in_trial", "duration_ms", "word_id"],
250)
253def _as_dataframe(
254 table: TablesLike, label: str, *, plan_for=None, kind: str | None = None
255) -> pd.DataFrame:
256 if isinstance(table, pd.DataFrame):
257 # DATA-66: a frame this API returned under the dataset's own names goes
258 # back to the internal names it was normalized under, so loading it
259 # again is the round-trip it was before (a converted width sits beside
260 # a renamed left edge, which no detection would pair up).
261 return _cn.to_canonical_frame(table)
262 items = _data.expand_table_inputs(table)
263 for item in items:
264 if not isinstance(item, pd.DataFrame) and not Path(item).is_file():
265 raise FileNotFoundError(
266 f"{label} table not found: {item} (looked in {Path.cwd()})"
267 )
268 # #374 F3: a zip holding both EyeLink reports gives each table its own.
269 return _data.read_tables(items, plan_for=plan_for, kind=kind)
272def _metadata_id_plan(id_column, infer, *extra):
273 """``plan_for`` for a metadata table: read its id column(s) as text.
275 The same protection the data tables get — read as numbers, readers ``1``
276 and ``01`` become one reader before the metadata ever sees them. The id
277 column is the caller's, else the one ``infer`` would pick from the header.
278 """
280 def plan(header) -> _data.ReadPlan:
281 names = [str(name) for name in header]
282 # `infer_*` treats a row-less frame as "no table" — give it one row.
283 resolved = id_column or infer(
284 pd.DataFrame([[None] * len(names)], columns=names)
285 )
286 columns = [*_data.trial_mapping_columns(resolved or []), *extra]
287 return _data.ReadPlan(
288 identity=tuple(c for c in dict.fromkeys(columns) if c and c in names)
289 )
291 return plan
294# ---------------------------------------------------------------------------
295# Schema diagnostics
296#
297# `data.validate_*_schema` says *what* is missing ("missing Trial ID"). A caller
298# scripting against an unfamiliar table also needs *why*: which column names
299# auto-detection looked for, which columns the table actually has, and the exact
300# override to pass. These tables mirror `data.propose_*_schema` field for field —
301# add a field there, add it here.
302# ---------------------------------------------------------------------------
304_SCHEMA_SPECS: dict = {
305 "words": {
306 "title": "Words",
307 "noun": "words",
308 "param": "word_schema",
309 # (schema key, human label, candidate column names) — the fields whose
310 # absence makes `validate_word_schema` fail.
311 "required": (
312 ("trial", "Trial ID", _data.TRIAL_CANDIDATES),
313 ("word_id", "Word ID", _data.WORD_ID_CANDIDATES),
314 ),
315 # …plus one "either group A or group B" requirement.
316 "group_label": "Word box",
317 "groups": (("x", "y", "width", "height"), ("left", "right", "top", "bottom")),
318 "group_candidates": {
319 "x": _data.WORD_X_CANDIDATES,
320 "y": _data.WORD_Y_CANDIDATES,
321 "width": _data.WORD_WIDTH_CANDIDATES,
322 "height": _data.WORD_HEIGHT_CANDIDATES,
323 "left": _data.WORD_LEFT_CANDIDATES,
324 "right": _data.WORD_RIGHT_CANDIDATES,
325 "top": _data.WORD_TOP_CANDIDATES,
326 "bottom": _data.WORD_BOTTOM_CANDIDATES,
327 },
328 "propose": _data.propose_word_schema,
329 },
330 "fixations": {
331 "title": "Fixations",
332 "noun": "fixations",
333 "param": "fix_schema",
334 "required": (
335 ("trial", "Trial ID", _data.TRIAL_CANDIDATES),
336 ("duration", "Duration", _data.FIX_DURATION_CANDIDATES),
337 ),
338 "group_label": "Fixation location",
339 "groups": (("x", "y"), ("word_id",)),
340 "group_candidates": {
341 "x": _data.FIX_X_CANDIDATES,
342 "y": _data.FIX_Y_CANDIDATES,
343 "word_id": _data.FIX_WORD_ID_CANDIDATES,
344 },
345 "propose": _data.propose_fix_schema,
346 },
347 "raw_gaze": {
348 "title": "Raw gaze",
349 "noun": "raw gaze",
350 "param": "raw_gaze_schema",
351 "required": (
352 ("trial", "Trial ID", _data.TRIAL_CANDIDATES),
353 ("x", "X", _data.RAW_GAZE_X_CANDIDATES),
354 ("y", "Y", _data.RAW_GAZE_Y_CANDIDATES),
355 ),
356 "group_label": None,
357 "groups": (),
358 "group_candidates": {},
359 "propose": _data.propose_raw_gaze_schema,
360 },
361}
364def _column_preview(frame: pd.DataFrame, limit: int = 40) -> str:
365 """Comma-separated column names, truncated so a 100-column IA report stays
366 readable in a traceback."""
367 cols = [str(c) for c in frame.columns]
368 shown = ", ".join(cols[:limit])
369 if len(cols) > limit:
370 shown += f", … (+{len(cols) - limit} more)"
371 return shown
374class SchemaError(ValueError):
375 """A table whose columns don't resolve onto the canonical fields.
377 Still a ``ValueError`` with the same message, so ``except ValueError``
378 callers are unaffected. The parts are kept apart for the CLI, whose
379 users cannot pass ``word_schema=``: it keeps :attr:`detail` and replaces
380 :attr:`hint` — the API-vocabulary "pass ``word_schema={…}``" line — with its
381 own ``--word-schema`` one, built from :attr:`mapping` (the mapping skeleton,
382 ``None`` when the fix is to correct a mapping rather than write one)."""
384 def __init__(
385 self, lines: list[str], hint: str, *, param: str, mapping: dict | None = None
386 ) -> None:
387 self.detail = "\n".join(lines)
388 self.hint = hint
389 self.param = param
390 self.mapping = mapping
391 super().__init__(f"{self.detail}\n{hint}")
394def _schema_skeleton_mapping(kind: str, schema: dict) -> dict:
395 """What was detected, ``'<column>'`` for the rest. An explicit schema
396 replaces auto-detection wholesale, so every required key has to be in it —
397 not just the ones that failed."""
398 spec = _SCHEMA_SPECS[kind]
399 keys = [key for key, _, _ in spec["required"]]
400 if spec["groups"]:
401 # Suggest whichever coordinate convention is closest to complete.
402 best = min(
403 spec["groups"],
404 key=lambda group: sum(1 for key in group if not schema.get(key)),
405 )
406 keys += [key for key in best if key not in keys]
407 return {key: schema[key] if schema.get(key) else "<column>" for key in keys}
410def _schema_skeleton(kind: str, schema: dict) -> str:
411 """:func:`_schema_skeleton_mapping` as a copy-pasteable Python literal."""
412 items = ", ".join(
413 f"{key!r}: {value!r}"
414 for key, value in _schema_skeleton_mapping(kind, schema).items()
415 )
416 return "{" + items + "}"
419def _schema_columns(schema: dict) -> list[tuple[str, str]]:
420 """``(schema key, column name)`` for every column a mapping names.
422 Multi-column (composite) mappings — the trial / participant / text id may be
423 a list, see :func:`data.trial_id_series` — expand to one pair per column."""
424 pairs: list = []
425 for key, value in schema.items():
426 if value is None:
427 continue
428 if isinstance(value, (list, tuple, set)):
429 pairs.extend((key, str(col)) for col in value)
430 else:
431 pairs.append((key, str(value)))
432 return pairs
435def _check_mapped_columns(kind: str, frame: pd.DataFrame, schema: dict) -> None:
436 """Reject a schema that maps a column the table doesn't have.
438 Only reachable with a caller-supplied schema — auto-detection only ever
439 picks columns that exist. Without this check a mistyped mapping raises a
440 bare ``KeyError: '<column>'`` from inside ``normalize_*``. (It could also be
441 silently ignored until BUG-58: ``normalize_words`` used to prefer a literal
442 ``unique_trial_id`` column over the mapped trial id; the mapping wins now.)"""
443 spec = _SCHEMA_SPECS[kind]
444 present = {str(c) for c in frame.columns}
445 missing = [(key, col) for key, col in _schema_columns(schema) if col not in present]
446 if not missing:
447 return
448 plural = "" if len(missing) == 1 else "s"
449 lines = [
450 f"{spec['title']} schema maps {len(missing)} column name{plural} the "
451 f"{spec['noun']} table doesn't have:"
452 ]
453 for key, column in missing:
454 close = difflib.get_close_matches(column, sorted(present), n=3, cutoff=0.6)
455 hint = f" (closest: {', '.join(repr(c) for c in close)})" if close else ""
456 lines.append(f" - {spec['param']}[{key!r}] = {column!r}: no such column{hint}")
457 lines.append(
458 f"Columns present in the {spec['noun']} table ({len(frame.columns)}): "
459 f"{_column_preview(frame)}"
460 )
461 raise SchemaError(
462 lines,
463 f"api.propose_schema(table, {kind!r}) returns the auto-detected mapping to "
464 "start from.",
465 param=spec["param"],
466 )
469def _known_schema_keys(kind: str, frame: pd.DataFrame) -> set:
470 """Every key a ``kind`` column mapping can set."""
471 spec = _SCHEMA_SPECS[kind]
472 keys = set(spec["propose"](frame.iloc[:0])) | {"block"}
473 keys |= {key for key, _, _ in spec["required"]}
474 keys |= {key for group in spec["groups"] for key in group}
475 return keys
478def _unknown_schema_keys(kind: str, frame: pd.DataFrame, schema: dict) -> list:
479 """The keys of ``schema`` that name no field (#374: ``x_pos`` for ``x``)."""
480 known = _known_schema_keys(kind, frame)
481 return [key for key in schema if key not in known]
484def _schema_error(
485 kind: str, frame: pd.DataFrame, schema: dict, problems: list, explicit: bool = False
486) -> SchemaError:
487 """Build the ``ValueError`` for a table whose canonical fields don't resolve.
489 Names every canonical field that could not be resolved, the candidate column
490 names auto-detection tried for it, the columns the table actually has, and
491 the explicit mapping to pass instead. ``explicit`` marks a schema the caller
492 supplied — nothing was auto-detected, so the message points at the keys
493 missing from *their* mapping rather than at failed detection."""
494 spec = _SCHEMA_SPECS[kind]
495 param = spec["param"]
496 lines = [f"{spec['title']} column mapping problems: {'; '.join(problems)}"]
497 unknown = _unknown_schema_keys(kind, frame, schema) if explicit else []
498 if unknown:
499 import difflib
501 known = sorted(_known_schema_keys(kind, frame))
502 for key in unknown:
503 close = difflib.get_close_matches(key, known, n=1) or [
504 k for k in known if key.startswith(f"{k}_") or key.endswith(f"_{k}")
505 ]
506 hint = f" — did you mean {close[0]!r}?" if close else ""
507 lines.append(f"{param} has a key that is not a field: {key!r}{hint}")
508 lines.append(f"Fields: {', '.join(known)}")
510 if explicit:
511 bullets = [
512 f" - {label} ({param} key {key!r}): not set in the {param} you passed. "
513 f"Auto-detection (used when {param} is omitted) looks for: "
514 f"{', '.join(candidates)}"
515 for key, label, candidates in spec["required"]
516 if not schema.get(key)
517 ]
518 else:
519 bullets = [
520 f" - {label} ({param} key {key!r}): no column matched. "
521 f"Looked for: {', '.join(candidates)}"
522 for key, label, candidates in spec["required"]
523 if not schema.get(key)
524 ]
525 groups_missing = [
526 (group, [key for key in group if not schema.get(key)])
527 for group in spec["groups"]
528 ]
529 # A group requirement only fails when *every* alternative is incomplete.
530 if groups_missing and all(missing for _, missing in groups_missing):
531 alternatives = " or ".join(
532 f"({', '.join(group)})" for group, _ in groups_missing
533 )
534 detail = "; ".join(
535 f"({', '.join(group)}) is missing {', '.join(missing)}"
536 for group, missing in groups_missing
537 )
538 unresolved = dict.fromkeys(
539 key for _, missing in groups_missing for key in missing
540 )
541 looked = " | ".join(
542 f"{key}: {', '.join(spec['group_candidates'][key])}" for key in unresolved
543 )
544 looked_label = "Auto-detection looks for" if explicit else "Looked for"
545 bullets.append(
546 f" - {spec['group_label']} ({param} keys): need either {alternatives} "
547 f"— {detail}.\n {looked_label} → {looked}"
548 )
549 if bullets:
550 lines.append(
551 f"Missing from the {param} you passed:"
552 if explicit
553 else f"Could not find these columns in the {spec['noun']} table:"
554 )
555 lines.extend(bullets)
556 resolved = ", ".join(
557 f"{key}={value!r}"
558 for key, value in schema.items()
559 if value is not None and key not in unknown
560 )
561 lines.append(
562 f"Fields the {param} does set: {resolved or '(none)'}"
563 if explicit
564 else f"Fields that did resolve: {resolved or '(none)'}"
565 )
566 lines.append(
567 f"Columns present in the {spec['noun']} table ({len(frame.columns)}): "
568 f"{_column_preview(frame)}"
569 )
570 if not explicit:
571 lines.append(
572 "Matching ignores case and separators (IA_LEFT == ia_left == 'Ia Left') "
573 "and takes the first candidate that matches; failing that, a vendor "
574 "prefix or suffix on a known name (AOI_LEFT, LEFT_px) is tried next, "
575 "accepted only when exactly one column qualifies."
576 )
577 hint = (
578 f"An explicit {param} replaces auto-detection wholesale, so it needs every "
579 f"required key, e.g. {param}={_schema_skeleton(kind, schema)} — "
580 f"api.propose_schema(df, {kind!r}) returns the auto-detected mapping."
581 if explicit
582 else f"To override auto-detection pass the full mapping, e.g. "
583 f"{param}={_schema_skeleton(kind, schema)} — "
584 f"api.propose_schema(df, {kind!r}) returns what was detected."
585 )
586 return SchemaError(
587 lines, hint, param=param, mapping=_schema_skeleton_mapping(kind, schema)
588 )
591def propose_schema(table: TablesLike, kind: str = "words") -> dict:
592 """Auto-detected column mapping for a **raw** (un-normalized) table.
594 ``kind`` is ``"words"``, ``"fixations"`` or ``"raw_gaze"``. Returns
595 ``{canonical field: source column or None}`` — the same mapping
596 [`load_scanpath_data`][scanpath_studio.api.load_scanpath_data] infers internally,
597 so it's the place to start when detection got a field wrong or couldn't find one:
598 edit the dict and pass it back as ``word_schema=`` / ``fix_schema=``::
600 from scanpath_studio import api
602 schema = api.propose_schema("ia.csv", "words")
603 schema["trial"] = "TRIAL_LABEL"
604 words, fixations = api.load_scanpath_data("ia.csv", "fix.csv",
605 word_schema=schema)
607 ``table`` is a DataFrame, path, glob or list of paths, like the loader's.
608 """
609 if kind not in _SCHEMA_SPECS:
610 raise ValueError(
611 f"Unknown kind {kind!r}; choose one of {', '.join(_SCHEMA_SPECS)}."
612 )
613 frame = _as_dataframe(
614 table,
615 _SCHEMA_SPECS[kind]["noun"],
616 kind=kind if kind in ("words", "fixations") else None,
617 )
618 return _SCHEMA_SPECS[kind]["propose"](frame)
621_NORMALIZED_ID_COLUMNS = ("participant_id", "trial_id")
624#: DATA-66: the column vocabularies a loader can return frames in.
625NAMES_SOURCE = "source"
626NAMES_CANONICAL = "canonical"
629class ScanpathData(tuple):
630 """What [`load_scanpath_data`][scanpath_studio.api.load_scanpath_data]
631 returns: the ``(words, fixations)`` frames — so
632 ``words, fixations = load_scanpath_data(…)`` unpacks it — plus
633 ``column_names``, the dataset's own name for every canonical column, per
634 table (``{"words": ColumnNames, "fixations": ColumnNames}``).
636 With ``names="source"`` (the default) the frames' columns are the dataset's
637 own names and each frame carries its map, so any API function takes it —
638 sliced, filtered or merged — and the names it writes are yours. With
639 ``names="canonical"`` they are the internal names, the same for every
640 dataset."""
642 def __new__(cls, words, fixations, column_names=None):
643 data = super().__new__(cls, (words, fixations))
644 data.column_names = dict(column_names or {})
645 return data
647 def __getnewargs__(self):
648 return (self[0], self[1], self.column_names)
650 @property
651 def words(self) -> pd.DataFrame:
652 return self[0]
654 @property
655 def fixations(self) -> pd.DataFrame:
656 return self[1]
659def _check_names_choice(names: str) -> None:
660 if names not in (NAMES_SOURCE, NAMES_CANONICAL):
661 raise ValueError(
662 f'names must be "{NAMES_SOURCE}" (the dataset\'s own column names) '
663 f'or "{NAMES_CANONICAL}" (the internal ones), got {names!r}.'
664 )
667def _named_in(
668 frame, label: str, *, optional: bool = False
669) -> tuple[pd.DataFrame, ColumnNames | None]:
670 """DATA-66: ``(canonical frame, its map)`` for a frame handed to the API.
672 A frame :func:`load_scanpath_data` returned under the dataset's own names
673 carries its map (`column_names.attach`); it is renamed back here, and the
674 map is returned for the call's options, figure text and output frames. A
675 canonical frame passes as it is, with no map. ``optional`` takes ``None``
676 as the empty table."""
677 found = _cn.frame_names(frame)
678 frame = _cn.to_canonical_frame(frame)
679 frame = (
680 _optional_frame(frame, label) if optional else _require_normalized(frame, label)
681 )
682 return frame, (found[1] if found else None)
685def _named_out(frame, table: str, names: ColumnNames | None):
686 """An output frame in the names its inputs carried: a table that
687 is the dataset's own under its whole map, else (``names`` already
688 restricted by the caller) only its ids. Canonical inputs, canonical out."""
689 if names is None or frame is None or not isinstance(frame, pd.DataFrame):
690 return frame
691 return _cn.attach(frame, table, names)
694def _call_names(
695 given: dict | None = None, **maps: ColumnNames | None
696) -> ColumnNames | None:
697 """One map across the frames of a call (fixations', then words', then raw
698 gaze's), or ``None`` when every frame was canonical. ``given`` is a
699 ``column_names=`` argument (``ScanpathData.column_names``), which names
700 canonical frames and wins over what the frames carry."""
701 if given:
702 return _cn.across_tables(
703 {
704 table: names
705 if isinstance(names, ColumnNames)
706 else ColumnNames.from_payload(names)
707 for table, names in given.items()
708 }
709 )
710 present = {table: names for table, names in maps.items() if names is not None}
711 return _cn.across_tables(present) if present else None
714def _table_names(
715 given: dict | None, table: str, carried: ColumnNames | None
716) -> ColumnNames | None:
717 """``table``'s own map for a call — from ``column_names=`` when given, else
718 what its frame carried. A word option is read in the words table's names
719 first: the merged map gives a shared column (``word_id``) the fixations'."""
720 if given:
721 names = given.get(table)
722 if names is None:
723 return None
724 return (
725 names if isinstance(names, ColumnNames) else ColumnNames.from_payload(names)
726 )
727 return carried
730def _require_normalized(frame, label: str) -> pd.DataFrame:
731 """Guard the plotting entry points against raw / wrongly-typed input.
733 The builders consume the *normalized* frames :func:`load_scanpath_data`
734 returns; handing them a path or a raw table otherwise fails deep inside with
735 a ``KeyError: 'participant_id'``."""
736 if not isinstance(frame, pd.DataFrame):
737 raise TypeError(
738 f"{label} must be the normalized pandas DataFrame returned by "
739 f"load_scanpath_data(), got {type(frame).__name__}. "
740 "Call words, fixations = load_scanpath_data(words=…, fixations=…) "
741 "first — it reads paths/globs and normalizes column names."
742 )
743 # DATA-66: a frame under the dataset's own names is processed canonically.
744 frame = _cn.to_canonical_frame(frame)
745 missing = [col for col in _NORMALIZED_ID_COLUMNS if col not in frame.columns]
746 if missing:
747 raise ValueError(
748 f"{label} frame is not normalized: missing the canonical column(s) "
749 f"{', '.join(missing)}. Its columns are: {_column_preview(frame)}. "
750 "Pass the frames returned by load_scanpath_data(...) (raw tables have "
751 "to go through it first). A frame it returned under the dataset's own "
752 "names loses that map when merged or concatenated with another table "
753 "(pandas drops DataFrame.attrs there): pass the result through "
754 "load_scanpath_data(...) again, or load with names='canonical'."
755 )
756 return frame
759def load_scanpath_data(
760 words: TablesLike | None = None,
761 fixations: TablesLike | None = None,
762 *,
763 word_schema: dict | None = None,
764 fix_schema: dict | None = None,
765 trial_parts_manifest: dict | None = None,
766 image_root: str | Path | None = None,
767 image_pattern: str = "{text_id}.png",
768 keep_columns: Iterable[str] | None = None,
769 names: str = NAMES_SOURCE,
770) -> ScanpathData:
771 """Load and normalize a words table and/or a fixations table.
773 The columns keep the names your files give them:
774 ``CURRENT_FIX_DURATION``, not ``duration_ms``. A column Scanpath Studio
775 built, converted, computed or changed keeps its internal name, and
776 ``data.column_names`` (a [`ScanpathData`][scanpath_studio.api.ScanpathData])
777 records what every column was called. Every API function takes these
778 frames, and every column option (``color_by=``, hover fields …) takes
779 either name. ``names="canonical"`` returns the internal names instead —
780 the same for every dataset, for code that works across them.
782 ``words`` / ``fixations`` may be DataFrames, paths to ``.csv`` / ``.tsv`` /
783 ``.txt`` / ``.tab`` / ``.parquet`` / ``.feather`` / ``.xlsx`` / ``.xls`` files (or
784 a ``.zip`` of them), glob patterns, or lists of paths — multi-file datasets (one
785 file per participant and/or text) are concatenated, with each file's stem kept in
786 a ``source_file`` column. Column schemas are auto-detected (EyeLink, Gazepoint,
787 Tobii, SMI, Pupil Labs, and snake_case names); pass ``word_schema`` /
788 ``fix_schema`` mappings (field → column name; see
789 [`propose_schema`][scanpath_studio.api.propose_schema])
790 to override detection.
792 ``trial_parts_manifest`` accepts a nested parent-trial/parts definition for
793 datasets whose source tables identify screens through arbitrary selector
794 columns; explicit ``screen_id`` / ``screen_index`` columns can instead be
795 mapped directly in each schema. Either table may be omitted for datasets
796 that ship only one report: the
797 missing side comes back as an empty canonical frame and the plots simply
798 skip that layer. Words without a participant column (stimulus-level AOIs)
799 are copied onto every trial in the fixations — each trial matched by
800 its trial id, else the trial id it had before a repeat's ``_r2`` suffix,
801 else its ``text_id`` (trial ids that embed the participant), with a
802 ``data.StimulusJoinWarning`` (a ``UserWarning``) when some trials match
803 none — and fixations without x/y but with a word/AOI ID are placed at
804 word-box centers. Columns named in ``data.INTERNAL_COLUMNS`` are the
805 pipeline's bookkeeping (``data.drop_internal_columns`` removes them).
807 Normalization keeps the mapped fields and the recognized optional ones
808 (eye, EyeLink's interest-area measures, linguistic features …) and drops the
809 rest. ``keep_columns`` names further columns of your own to carry through
810 under their own names — a pupil size, a detection confidence — from
811 whichever table has them, so a figure can color, hover or plot by them
812 (the app's *Extra fields to keep*; ``render --keep-columns`` on the command line).
814 Returns the normalized ``(words, fixations)`` frames the plotting
815 functions expect. Raises ``ValueError`` if a required field can't be found —
816 the message names the canonical field, the column names auto-detection
817 looked for, and the columns the table actually has — and
818 ``data.StimulusJoinError`` (a ``ValueError``) when a stimulus-level words
819 table shares neither a trial id nor a ``text_id`` with any trial (or,
820 multipart, with every screen a trial has fixations on).
821 """
822 _check_names_choice(names)
823 if words is None and fixations is None:
824 raise ValueError("Provide at least one of words= or fixations=.")
825 # DATA-66: a frame this API already named is loaded under its internal
826 # names; its own map then renames the new one back to the user's.
827 prior = {
828 table: found[1]
829 for table, frame in (("words", words), ("fixations", fixations))
830 if (found := _cn.frame_names(frame)) is not None
831 }
833 if words is not None:
834 # BUG-53: a word spelled "None" or "NA" is a word, not a missing cell.
835 words_df = _as_dataframe(
836 words,
837 "words",
838 plan_for=lambda header: _data.verbatim_text_plan(header, word_schema),
839 kind="words",
840 )
841 explicit = word_schema is not None
842 word_schema = word_schema or _data.propose_word_schema(words_df)
843 _check_mapped_columns("words", words_df, word_schema)
844 problems = _data.validate_word_schema(word_schema)
845 if problems:
846 raise _schema_error("words", words_df, word_schema, problems, explicit)
847 words_norm = _data.normalize_words(
848 words_df,
849 word_schema,
850 keep_columns=_with_optional_fields(
851 keep_columns, _data.WORD_OPTIONAL_FIELDS
852 ),
853 )
854 if trial_parts_manifest is not None:
855 words_norm = apply_trial_parts_manifest(
856 words_norm, words_df, trial_parts_manifest, kind="words"
857 )
858 else:
859 words_norm = _data.empty_words_frame()
861 if fixations is not None:
862 fixations_df = _as_dataframe(
863 fixations,
864 "fixations",
865 plan_for=lambda header: _data.identity_text_plan(header, fix_schema),
866 kind="fixations",
867 )
868 explicit = fix_schema is not None
869 fix_schema = fix_schema or _data.propose_fix_schema(fixations_df)
870 _check_mapped_columns("fixations", fixations_df, fix_schema)
871 problems = _data.validate_fix_schema(fix_schema)
872 if problems:
873 raise _schema_error(
874 "fixations", fixations_df, fix_schema, problems, explicit
875 )
876 fixations_norm = _data.normalize_fixations(
877 fixations_df,
878 fix_schema,
879 keep_columns=_with_optional_fields(keep_columns, _data.FIX_OPTIONAL_FIELDS),
880 )
881 if trial_parts_manifest is not None:
882 fixations_norm = apply_trial_parts_manifest(
883 fixations_norm,
884 fixations_df,
885 trial_parts_manifest,
886 kind="fixations",
887 )
888 else:
889 fixations_norm = _data.empty_fixations_frame()
891 words_norm, fixations_norm, _join, rewrites = _data.harmonize_frames_reporting(
892 words_norm, fixations_norm
893 )
894 if image_root is not None:
895 words_norm = _data.resolve_stimulus_image_paths(
896 words_norm, image_root, image_pattern
897 )
898 fixations_norm = _data.resolve_stimulus_image_paths(
899 fixations_norm, image_root, image_pattern
900 )
901 # DATA-66: what each column was called in these files — from the schemas
902 # and raw columns normalization read, with the columns the fixups rewrote
903 # marked converted.
904 maps = {
905 table: ColumnNames.from_payload(payload)
906 for table, payload in _cn.for_tables(
907 {"words": word_schema, "fixations": fix_schema},
908 {
909 "words": words_df if words is not None else None,
910 "fixations": fixations_df if fixations is not None else None,
911 },
912 rewrites=rewrites,
913 ).items()
914 }
915 maps = {
916 table: names_map.through(prior[table]) if table in prior else names_map
917 for table, names_map in maps.items()
918 }
919 if names == NAMES_SOURCE:
920 words_norm = _cn.attach(words_norm, "words", maps.get("words"))
921 fixations_norm = _cn.attach(fixations_norm, "fixations", maps.get("fixations"))
922 return ScanpathData(words_norm, fixations_norm, maps)
925def _with_optional_fields(
926 keep_columns: Iterable[str] | None, registry: list
927) -> set | None:
928 """``keep_columns`` as the normalizers take it: ``None`` (every recognized
929 optional field, nothing else) when none are named, else those names *plus*
930 every optional field — a non-``None`` set would otherwise limit them."""
931 if not keep_columns:
932 return None
933 if isinstance(keep_columns, str):
934 keep_columns = [keep_columns]
935 return {str(c) for c in keep_columns} | {entry[0] for entry in registry}
938def load_participant_metadata(
939 table: TablesLike,
940 *,
941 id_column: str | None = None,
942 participants: pd.DataFrame | list | None = None,
943):
944 """Load a participant-level metadata table.
946 ``table`` is a DataFrame or a path/glob to a CSV/TSV/Parquet/Excel file with
947 **one row per participant**: an id column plus anything known about them
948 (``native_language``, ``age``, a comprehension score). ``id_column``
949 defaults to the first recognized spelling (``participant_id``, ``subject``,
950 ``RECORDING_SESSION_LABEL``, …).
952 Pass ``participants`` — a normalized frame or a list of ids — to have the
953 join validated against the data you actually loaded; the returned object's
954 ``.report`` then names the participants missing from either side.
956 Returns a
957 `ParticipantMetadata`: the cleaned frame,
958 a field registry (name, label, grain, dtype, missingness), and the join
959 report. Nothing is broadcast onto the words/fixations frames — use
960 `scanpath_studio.metadata.project` to attach chosen columns to a
961 per-trial frame, or ``.values_for(pid)`` for one participant.
963 >>> words, fixations = load_sample_data()
964 >>> meta = load_participant_metadata(
965 ... "readers.csv", participants=fixations
966 ... ) # doctest: +SKIP
967 >>> meta.names # doctest: +SKIP
968 ('native_language', 'age')
969 """
970 from scanpath_studio import metadata as _metadata
972 frame = _as_dataframe(
973 table,
974 "participant metadata",
975 plan_for=_metadata_id_plan(id_column, _metadata.infer_participant_id_column),
976 )
977 resolved = id_column or _metadata.infer_participant_id_column(frame)
978 if not resolved or resolved not in frame.columns:
979 raise ValueError(
980 "Could not find the participant-id column in the metadata table. "
981 f"Columns: {_column_preview(frame)}. Pass id_column= explicitly."
982 )
983 if isinstance(participants, pd.DataFrame):
984 participants = _metadata.participant_ids(_cn.to_canonical_frame(participants))
985 return _metadata.build_participant_metadata(
986 frame,
987 resolved,
988 source_name=getattr(table, "name", None) or "participant metadata",
989 participants=participants,
990 )
993def load_trial_metadata(
994 table: TablesLike,
995 *,
996 id_column: str | None = None,
997 participant_column: str | None = None,
998 trials: pd.DataFrame | None = None,
999):
1000 """Load a trial-level metadata table.
1002 The sibling of
1003 [`load_participant_metadata`][scanpath_studio.api.load_participant_metadata], one
1004 grain down: ``table`` has **one row per trial** — a trial-id column plus anything
1005 known about that trial (a list name, a condition, a per-trial comprehension
1006 score).
1008 **The key is yours to state, and it changes what the table means.** Keyed by
1009 trial id alone, a row describes a *text*, and every trial of it
1010 inherits that row; pass ``participant_column`` to key by participant **and**
1011 trial, so a row describes one *trial*. Nothing in a file says which world
1012 a corpus is in, so this is never inferred — unlike ``id_column``, which
1013 defaults to the first recognized spelling (``trial_id``, ``item_id``,
1014 ``TRIAL_INDEX``, …).
1016 Pass ``trials`` — a normalized fixations/words frame, or any frame with
1017 ``participant_id`` + ``trial_id`` — to have the join validated against the
1018 data you actually loaded; the returned ``.report`` then names the trials
1019 missing from either side.
1021 Returns a `TrialMetadata`: the cleaned
1022 frame, a field registry, and the join report. As with the participant
1023 table, nothing is broadcast onto the words/fixations frames.
1025 >>> words, fixations = load_sample_data()
1026 >>> meta = load_trial_metadata(
1027 ... "readings.csv", trials=fixations
1028 ... ) # doctest: +SKIP
1029 >>> meta.names # doctest: +SKIP
1030 ('list_name', 'comprehension_score')
1031 """
1032 from scanpath_studio import metadata as _metadata
1034 frame = _as_dataframe(
1035 table,
1036 "trial metadata",
1037 plan_for=_metadata_id_plan(
1038 id_column, _metadata.infer_trial_id_column, participant_column
1039 ),
1040 )
1041 resolved = id_column or _metadata.infer_trial_id_column(frame)
1042 if not resolved or resolved not in frame.columns:
1043 raise ValueError(
1044 "Could not find the trial-id column in the metadata table. "
1045 f"Columns: {_column_preview(frame)}. Pass id_column= explicitly."
1046 )
1047 if participant_column and participant_column not in frame.columns:
1048 raise ValueError(
1049 f"participant_column={participant_column!r} is not in the metadata "
1050 f"table. Columns: {_column_preview(frame)}."
1051 )
1052 keys = (
1053 _metadata.trial_keys(_cn.to_canonical_frame(trials))
1054 if trials is not None
1055 else None
1056 )
1057 return _metadata.build_trial_metadata(
1058 frame,
1059 resolved,
1060 participant_column,
1061 source_name=getattr(table, "name", None) or "trial metadata",
1062 keys=keys,
1063 )
1066def load_text_metadata(
1067 table: TablesLike,
1068 *,
1069 id_column: str | list[str] | None = None,
1070 texts: pd.DataFrame | list | None = None,
1071):
1072 """Load a text-level metadata table — the third grain.
1074 ``table`` has **one row per text** — a text-id column plus anything known about that
1075 text (genre, difficulty, a stimulus-level comprehension score). Flat grain, like
1076 [`load_participant_metadata`][scanpath_studio.api.load_participant_metadata]: never
1077 keyed by participant, since a text is a stimulus rather than something one participant owns.
1078 ``id_column`` defaults to the first recognized spelling (``text_id``,
1079 ``paragraph_id``, ``stimulus_id``, …) and may be several columns to build a
1080 composite id, the same way the uploaded data's own Text ID mapping does.
1082 Pass ``texts`` — a normalized fixations/words frame, or any iterable of
1083 text ids — to have the join validated against the data you actually
1084 loaded; the returned ``.report`` then names the texts missing from either
1085 side.
1087 Returns a `TextMetadata`: the cleaned
1088 frame, a field registry, and the join report. As with the other two
1089 grains, nothing is broadcast onto the words/fixations frames.
1091 >>> words, fixations = load_sample_data()
1092 >>> meta = load_text_metadata(
1093 ... "texts.csv", texts=words
1094 ... ) # doctest: +SKIP
1095 >>> meta.names # doctest: +SKIP
1096 ('genre', 'difficulty')
1097 """
1098 from scanpath_studio import metadata as _metadata
1100 frame = _as_dataframe(
1101 table,
1102 "text metadata",
1103 plan_for=_metadata_id_plan(id_column, _metadata.infer_text_id_column),
1104 )
1105 resolved = id_column or _metadata.infer_text_id_column(frame)
1106 if not resolved or any(
1107 c not in frame.columns for c in _metadata.trial_mapping_columns(resolved)
1108 ):
1109 raise ValueError(
1110 "Could not find the text-id column in the metadata table. "
1111 f"Columns: {_column_preview(frame)}. Pass id_column= explicitly."
1112 )
1113 if isinstance(texts, pd.DataFrame):
1114 texts = _metadata.text_keys(_cn.to_canonical_frame(texts))
1115 return _metadata.build_text_metadata(
1116 frame,
1117 resolved,
1118 source_name=getattr(table, "name", None) or "text metadata",
1119 keys=texts,
1120 )
1123def load_sample_data(*, names: str = NAMES_SOURCE) -> ScanpathData:
1124 """Return the bundled OneStop demo, normalized and ready to plot: two
1125 participants, twelve trials each, every one of them with fixations. Under
1126 the demo's own column names; ``names="canonical"`` for the internal ones
1127 (see [`load_scanpath_data`][scanpath_studio.api.load_scanpath_data]).
1129 The frames carry the demo's recorded screen (OneStop's 2560×1440), so
1130 `plot_scanpath` draws them on it without a ``canvas_size``, as
1131 ``scanpath-studio render --sample`` does."""
1132 from .code_snippet import SOURCE_DEMO, source_canvas
1134 data = load_scanpath_data(*_data.load_sample_data(), names=names)
1135 screen = source_canvas(SOURCE_DEMO)
1136 for frame in data:
1137 frame.attrs[RECORDED_SCREEN_ATTR] = screen
1138 return data
1141#: The `DataFrame.attrs` key a frame carries its dataset's recorded screen in,
1142#: ``(width, height)`` px — read when no ``canvas_size`` is passed.
1143RECORDED_SCREEN_ATTR = "scanpath_studio.recorded_screen"
1146def _recorded_screen(*frames) -> tuple[int, int] | None:
1147 """The recorded screen one of ``frames`` carries (`load_sample_data`)."""
1148 for frame in frames:
1149 screen = getattr(frame, "attrs", {}).get(RECORDED_SCREEN_ATTR)
1150 if screen:
1151 return int(screen[0]), int(screen[1])
1152 return None
1155def load_raw_gaze(
1156 table: TablesLike,
1157 *,
1158 raw_gaze_schema: dict | None = None,
1159 names: str = NAMES_SOURCE,
1160) -> pd.DataFrame:
1161 """Load and normalize a raw (sample-level) gaze table for ``raw_gaze=``.
1163 The third table [`plot_scanpath`][scanpath_studio.api.plot_scanpath] can draw, under
1164 the fixations: one row per eye-tracker sample, with a participant, a trial, ``x`` /
1165 ``y`` and usually a timestamp. ``table`` is a DataFrame, path, glob or list of
1166 paths, like [`load_scanpath_data`][scanpath_studio.api.load_scanpath_data]'s, and
1167 the columns are auto-detected the same way; pass ``raw_gaze_schema`` (field →
1168 column, see ``api.propose_schema(table, "raw_gaze")``) to override the detection.
1169 ``plot_scanpath`` keeps only the plotted trial's (and screen's) samples, so one
1170 table can serve a whole corpus::
1172 raw_gaze = sps.load_raw_gaze("gaze_samples.csv")
1173 fig = sps.plot_scanpath(words, fixations, "p1", "t3", raw_gaze=raw_gaze)
1175 Under the table's own column names, like ``load_scanpath_data``'s;
1176 ``names="canonical"`` for the internal ones.
1177 """
1178 _check_names_choice(names)
1179 frame = _as_dataframe(
1180 table,
1181 "raw gaze",
1182 plan_for=lambda header: _data.identity_text_plan(
1183 header, raw_gaze_schema, kind="raw_gaze"
1184 ),
1185 )
1186 explicit = raw_gaze_schema is not None
1187 schema = raw_gaze_schema or _data.propose_raw_gaze_schema(frame)
1188 _check_mapped_columns("raw_gaze", frame, schema)
1189 problems = _data.validate_raw_gaze_schema(schema)
1190 if problems:
1191 raise _schema_error("raw_gaze", frame, schema, problems, explicit)
1192 normalized = _data.normalize_raw_gaze(frame, schema)
1193 if names == NAMES_CANONICAL:
1194 return normalized
1195 return _cn.attach(
1196 normalized, "raw_gaze", _cn.from_schema("raw_gaze", schema, frame.columns)
1197 )
1200def load_sample_raw_gaze(*, names: str = NAMES_SOURCE) -> pd.DataFrame:
1201 """The bundled demo's raw gaze, normalized — what the app overlays on it.
1203 OneStop ships no sample-level gaze, so this is **synthesized** from one of
1204 the demo's real trials and covers that trial alone."""
1205 return load_raw_gaze(_data.load_sample_raw_gaze(), names=names)
1208def check_data_health(
1209 words: pd.DataFrame | None = None,
1210 fixations: pd.DataFrame | None = None,
1211 raw_gaze: pd.DataFrame | None = None,
1212) -> pd.DataFrame:
1213 """Values that loaded as numbers but cannot be right — the Data Management page's *Data checks*.
1215 Checks the normalized tables (from
1216 [`load_scanpath_data`][scanpath_studio.api.load_scanpath_data] /
1217 [`load_raw_gaze`][scanpath_studio.api.load_raw_gaze]) for fixations lasting
1218 0 ms or less or with an infinite duration or onset, fixations and raw-gaze
1219 samples whose position is missing or infinite, word boxes with no area or no
1220 finite position, and per-screen screen sizes that are not finite and
1221 positive. One row per check that found
1222 anything: ``table``, ``check``, ``problem``, the ``columns`` it read (in the
1223 names the frames carry),
1224 ``rows`` of ``of_rows``, the ``trials`` they fall in, a ``breakdown`` by
1225 kind, ``severity`` (``"note"`` for raw-gaze gaps, which blinks and track
1226 loss make ordinary), ``what_happens`` to those rows in the app, and a few
1227 ``examples``. An empty frame means every check passed. Nothing is changed
1228 or dropped::
1230 words, fixations = sps.load_scanpath_data("ia.csv", "fixations.csv")
1231 print(sps.check_data_health(words, fixations))
1232 """
1233 from .data_health import findings_frame
1235 return findings_frame(_health_findings(words, fixations, raw_gaze))
1238def _health_findings(words, fixations, raw_gaze) -> list:
1239 """`data_health.check_data_health` on frames in either naming, its findings
1240 naming the columns as the frames did. The CLI's ``check`` prints
1241 these; :func:`check_data_health` tabulates them."""
1242 from dataclasses import replace
1244 from .data_health import check_data_health as _check
1246 named = {}
1247 for table, frame in (
1248 ("words", words),
1249 ("fixations", fixations),
1250 ("raw_gaze", raw_gaze),
1251 ):
1252 found = _cn.frame_names(frame) if frame is not None else None
1253 named[table] = (
1254 _cn.to_canonical_frame(frame) if frame is not None else None,
1255 found[1] if found else None,
1256 )
1257 findings = _check(*(frame for frame, _names in named.values()))
1259 def _in_own_names(finding):
1260 # DATA-66: name the columns and example fields as the frames did.
1261 names = named[finding.table][1]
1262 if names is None:
1263 return finding
1264 return replace(
1265 finding,
1266 columns=tuple(names.display(c) for c in finding.columns),
1267 examples=tuple(
1268 {names.display(k): v for k, v in row.items()}
1269 for row in finding.examples
1270 ),
1271 )
1273 return [_in_own_names(f) for f in findings]
1276def _require_computed_measures(name: str) -> None:
1277 """Refuse a held-back computation, naming the switch that enables it.
1279 These values are computed by Scanpath Studio rather than read from the
1280 dataset, and each needs checking by hand before it is released. A script
1281 gets an error rather than an unchecked number.
1282 """
1283 if not computed_measures_enabled():
1284 raise ValueError(
1285 f"{name} is not available in this release: its values are computed "
1286 "by Scanpath Studio and have not been validated yet. Set "
1287 f"{EXPERIMENTAL_ENV_VAR}=1 to use it anyway."
1288 )
1291def compute_word_metrics(words: pd.DataFrame, fixations: pd.DataFrame) -> pd.DataFrame:
1292 """Per-word reading measures (FFD/FPRT/RPD/TFD, skips, regressions, …).
1294 Experimental: raises unless ``SCANPATH_EXPERIMENTAL=1``. Pre-aggregated
1295 columns in ``words`` (EyeLink IA exports) are preserved; anything missing is
1296 computed from fixations + word bounding boxes. Takes the normalized frames
1297 from [`load_scanpath_data`][scanpath_studio.api.load_scanpath_data], and
1298 answers in the names they carry."""
1299 _require_computed_measures("compute_word_metrics")
1300 words, word_names = _named_in(words, "words")
1301 fixations, _fix_names = _named_in(fixations, "fixations")
1302 return _named_out(_data.compute_word_metrics(words, fixations), "words", word_names)
1305def trial_summary(words: pd.DataFrame, fixations: pd.DataFrame) -> pd.DataFrame:
1306 """Exportable one-row-per-trial reading summary.
1308 Experimental: raises unless ``SCANPATH_EXPERIMENTAL=1``."""
1309 _require_computed_measures("trial_summary")
1310 from .aggregation import trial_summary_table
1312 words, word_names = _named_in(words, "words", optional=True)
1313 fixations, fix_names = _named_in(fixations, "fixations", optional=True)
1314 names = _call_names(fixations=fix_names, words=word_names)
1315 return _named_out(
1316 trial_summary_table(words, fixations),
1317 "trial_summary",
1318 names.identity() if names else None,
1319 )
1322def reader_summary(words: pd.DataFrame, fixations: pd.DataFrame) -> pd.DataFrame:
1323 """Exportable one-row-per-reader reading summary.
1325 Experimental: raises unless ``SCANPATH_EXPERIMENTAL=1``."""
1326 _require_computed_measures("reader_summary")
1327 from .aggregation import reader_summary_table
1329 words, word_names = _named_in(words, "words", optional=True)
1330 fixations, fix_names = _named_in(fixations, "fixations", optional=True)
1331 names = _call_names(fixations=fix_names, words=word_names)
1332 return _named_out(
1333 reader_summary_table(words, fixations),
1334 "reader_summary",
1335 names.identity() if names else None,
1336 )
1339def preprocess_data(
1340 words: pd.DataFrame,
1341 fixations: pd.DataFrame,
1342 *,
1343 enabled: bool = False,
1344 short_policy: str = "Off",
1345 short_threshold_ms: float = 80.0,
1346 merge_distance_chars: float = 1.0,
1347 discard_blink_adjacent: bool = False,
1348) -> tuple[pd.DataFrame, pd.DataFrame, pd.DataFrame]:
1349 """Apply the optional preprocessing stage and return words/fixations/QA.
1351 Experimental: raises unless ``SCANPATH_EXPERIMENTAL=1``."""
1352 _require_computed_measures("preprocess_data")
1353 if not enabled:
1354 return words, fixations, pd.DataFrame()
1356 from .measures import assign_fixations_to_words, enrich_fixations
1357 from .preprocessing import preprocess_fixations
1359 given_words = words
1360 words, _word_names = _named_in(words, "words", optional=True)
1361 fixations, fix_names = _named_in(fixations, "fixations", optional=True)
1363 assigned = (
1364 enrich_fixations(assign_fixations_to_words(fixations, words), words)
1365 if not fixations.empty
1366 else fixations
1367 )
1368 processed, report = preprocess_fixations(
1369 assigned,
1370 words,
1371 settings={
1372 "enabled": enabled,
1373 "short_policy": short_policy,
1374 "short_threshold_ms": short_threshold_ms,
1375 "merge_distance_chars": merge_distance_chars,
1376 "discard_blink_adjacent": discard_blink_adjacent,
1377 },
1378 )
1379 # The QA report is derived: it names its ids as the fixations do.
1380 return (
1381 given_words,
1382 _named_out(processed, "fixations", fix_names),
1383 _named_out(report, "cleaning_qa", fix_names.identity() if fix_names else None),
1384 )
1387def analysis_tables(
1388 words: pd.DataFrame,
1389 fixations: pd.DataFrame,
1390 *,
1391 pixels_per_degree: float | None = None,
1392 raw_gaze: pd.DataFrame | None = None,
1393) -> dict[str, pd.DataFrame]:
1394 """The tables ``scanpath-studio analyze`` writes, as a dict of frames.
1396 Experimental: raises unless ``SCANPATH_EXPERIMENTAL=1``.
1398 ``fixations``, ``saccades``, ``word_measures``, ``sentence_measures``,
1399 ``trial_summary``, ``reader_summary``, ``characters`` and ``cleaning_qa``.
1401 ``word_measures`` is the words table with the reading measures it
1402 *brought*: none are computed here, and a words table that carries none
1403 leaves ``word_measures`` out.
1404 """
1405 _require_computed_measures("analysis_tables")
1406 from .aggregation import reader_summary_table, trial_summary_table
1407 from .measures import assign_fixations_to_words, enrich_fixations
1408 from .preprocessing import (
1409 character_grid,
1410 cleaning_report,
1411 saccade_table,
1412 sentence_measures,
1413 )
1415 words, word_names = _named_in(words, "words", optional=True)
1416 fixations, fix_names = _named_in(fixations, "fixations", optional=True)
1417 if raw_gaze is not None:
1418 raw_gaze, _gaze_names = _named_in(raw_gaze, "raw_gaze")
1419 analysis_fixations = (
1420 enrich_fixations(assign_fixations_to_words(fixations, words), words)
1421 if not fixations.empty and not words.empty
1422 else fixations
1423 )
1424 tables = {
1425 "fixations": analysis_fixations,
1426 "saccades": saccade_table(
1427 analysis_fixations,
1428 pixels_per_degree=pixels_per_degree,
1429 raw_gaze=raw_gaze,
1430 words=words,
1431 ),
1432 "word_measures": words,
1433 "sentence_measures": sentence_measures(words, analysis_fixations),
1434 "trial_summary": trial_summary_table(words, analysis_fixations),
1435 "reader_summary": reader_summary_table(words, analysis_fixations),
1436 "characters": character_grid(words),
1437 "cleaning_qa": cleaning_report(analysis_fixations),
1438 }
1439 if not _data.brought_reading_measures(words):
1440 del tables["word_measures"]
1441 # DATA-66: in the names the frames carried — the two tables that are the
1442 # dataset's own under their whole map, the derived ones by their ids only
1443 # (the export bundle's rule, `export._ARTIFACT_TABLE`).
1444 names = _call_names(fixations=fix_names, words=word_names)
1445 if names is None:
1446 return tables
1447 own = {"fixations": fix_names, "word_measures": word_names}
1448 return {
1449 artifact: _named_out(
1450 table,
1451 artifact,
1452 own[artifact] if artifact in own else names.identity(),
1453 )
1454 for artifact, table in tables.items()
1455 }
1458def alignment_sensitivity(
1459 words: pd.DataFrame,
1460 fixations: pd.DataFrame,
1461 methods: tuple[str, ...] = ("attach", "slice", "consensus"),
1462) -> tuple[pd.DataFrame, pd.DataFrame]:
1463 """Word-measure sensitivity and correction QA across line algorithms.
1465 A derived surface of vertical drift correction, so it is gated with
1466 it and raises rather than returning something that looks like a result.
1467 """
1468 if not drift_correction_enabled():
1469 raise ValueError(
1470 "alignment_sensitivity is not available in this release (vertical "
1471 "drift correction is not fully integrated yet). Set "
1472 f"{EXPERIMENTAL_ENV_VAR}=1 to enable it."
1473 )
1474 from .preprocessing import measure_sensitivity
1476 words, word_names = _named_in(words, "words")
1477 fixations, fix_names = _named_in(fixations, "fixations")
1478 names = _call_names(fixations=fix_names, words=word_names)
1479 tables = measure_sensitivity(words, fixations, methods)
1480 if names is None:
1481 return tables
1482 return tuple(
1483 _named_out(table, "alignment_sensitivity", names.identity()) for table in tables
1484 )
1487#: The columns each corpus-figure kind reads; ``"<value>"`` stands for
1488#: ``value_col``. EXP-13: without the check a table lacking one surfaced as a
1489#: bare ``KeyError: 'value'`` from inside the builder.
1490_CORPUS_COLUMNS = {
1491 "profile": ("word_id", "<value>"),
1492 "distribution": ("<value>",),
1493 # EXP-16: the builder draws its "no data" placeholder for a table with no
1494 # `diff` — right for the app's empty states, but headlessly it meant a
1495 # figure with nothing on it and an exit code of 0.
1496 "difference": ("word_id", "diff"),
1497}
1500def _require_corpus_columns(data: pd.DataFrame, kind: str, value_col: str) -> None:
1501 required = [
1502 value_col if column == "<value>" else column
1503 for column in _CORPUS_COLUMNS.get(kind, ())
1504 ]
1505 missing = [column for column in required if column not in data.columns]
1506 if not missing:
1507 return
1508 hint = (
1509 f" Name the measure column with value_col= (--value-col on the CLI); "
1510 f"it is {value_col!r} now."
1511 if value_col in missing
1512 else ""
1513 )
1514 raise ValueError(
1515 f"A {kind!r} corpus figure reads the column(s) "
1516 f"{', '.join(repr(column) for column in missing)}, which the table doesn't "
1517 f"have. Columns present ({len(data.columns)}): {_column_preview(data)}.{hint}"
1518 )
1521def plot_corpus_figure(
1522 data: pd.DataFrame,
1523 *,
1524 kind: str,
1525 measure_label: str = "Value",
1526 series_col: str = "series",
1527 value_col: str = "value",
1528 colors: tuple[str, ...] | None = None,
1529 canvas_width: int = 1000,
1530 base_font_size: int = 14,
1531 font_family: str = FONT_FAMILY,
1532) -> go.Figure:
1533 """Headless corpus profile/distribution/difference plot with shared colors.
1535 ``profile`` expects ``word_id`` plus ``value_col`` (and optional ``lo`` /
1536 ``hi``); ``distribution`` expects ``value_col``; ``difference`` expects
1537 ``word_id`` and ``diff``. When ``series_col`` is present, it defines the
1538 overlaid profile/distribution series. A table missing a column its ``kind``
1539 reads raises ``ValueError`` naming it and the columns present.
1540 """
1541 kind = str(kind).lower()
1542 _require_corpus_columns(data, kind, value_col)
1543 if kind == "profile":
1544 profiles = (
1545 {
1546 str(name): group.rename(columns={value_col: "value"})
1547 for name, group in data.groupby(series_col, sort=False)
1548 }
1549 if series_col in data
1550 else {measure_label: data.rename(columns={value_col: "value"})}
1551 )
1552 return make_word_profile_figure(
1553 profiles,
1554 measure_label=measure_label,
1555 canvas_width=canvas_width,
1556 base_font_size=base_font_size,
1557 font_family=font_family,
1558 colors=colors,
1559 )
1560 if kind == "distribution":
1561 groups = (
1562 {
1563 str(name): group[value_col].dropna().to_numpy()
1564 for name, group in data.groupby(series_col, sort=False)
1565 }
1566 if series_col in data
1567 else {measure_label: data[value_col].dropna().to_numpy()}
1568 )
1569 return make_distribution_figure(
1570 groups,
1571 metric_label=measure_label,
1572 canvas_width=canvas_width,
1573 base_font_size=base_font_size,
1574 font_family=font_family,
1575 colors=colors,
1576 )
1577 if kind == "difference":
1578 return make_difference_profile_figure(
1579 data,
1580 measure_label=measure_label,
1581 canvas_width=canvas_width,
1582 base_font_size=base_font_size,
1583 font_family=font_family,
1584 colors=colors,
1585 )
1586 raise ValueError("kind must be 'profile', 'distribution', or 'difference'.")
1589def _optional_frame(frame, label: str) -> pd.DataFrame:
1590 """``frame`` checked as normalized, or the empty canonical frame for ``None``.
1592 A dataset recorded as raw gaze alone has no words or fixations
1593 table, so the plotting entry points take ``None`` for either — the same
1594 empty canonical frame `load_scanpath_data` returns for a table it was not
1595 given."""
1596 if frame is None:
1597 return (
1598 _data.empty_words_frame()
1599 if label == "words"
1600 else _data.empty_fixations_frame()
1601 )
1602 return _require_normalized(frame, label)
1605def list_trials(
1606 words: pd.DataFrame | None = None,
1607 fixations: pd.DataFrame | None = None,
1608 *,
1609 raw_gaze: pd.DataFrame | None = None,
1610) -> pd.DataFrame:
1611 """One row per plottable trial: its participant id and trial id.
1613 Trials present in both frames when both are loaded; for single-report
1614 datasets (words-only or fixations-only), trials from whichever frame has
1615 data. ``raw_gaze`` (a frame from
1616 [`load_raw_gaze`][scanpath_studio.api.load_raw_gaze]) adds the trials that
1617 only its samples cover — every trial, for a dataset recorded as raw gaze
1618 alone (pass ``None`` for ``words`` and ``fixations`` then). The id columns
1619 take the names the frames carry."""
1620 words, word_names = _named_in(words, "words", optional=True)
1621 fixations, fix_names = _named_in(fixations, "fixations", optional=True)
1622 gaze_names = None
1623 if raw_gaze is not None:
1624 raw_gaze, gaze_names = _named_in(raw_gaze, "raw_gaze")
1625 names = _call_names(fixations=fix_names, words=word_names, raw_gaze=gaze_names)
1626 cols = ["participant_id", "trial_id"]
1627 if words.empty or fixations.empty:
1628 present = fixations if words.empty else words
1629 combos = present[cols].drop_duplicates()
1630 else:
1631 combos = words[cols].drop_duplicates().merge(fixations[cols].drop_duplicates())
1632 if raw_gaze is not None and not raw_gaze.empty:
1633 # The app's rule (`utils.combo_source`): a trial is listed when it has
1634 # fixations — or, in a dataset without any, words — or when it has raw
1635 # gaze. So a trial with words and samples but no fixations is listed,
1636 # while one the intersection above drops for having fixations but no
1637 # words stays dropped: its samples add nothing the rule is about.
1638 known = _data.trial_keys(fixations if not fixations.empty else words)
1639 samples = raw_gaze[cols].drop_duplicates()
1640 extra = samples[
1641 [
1642 (str(p), str(t)) not in known
1643 for p, t in zip(samples["participant_id"], samples["trial_id"])
1644 ]
1645 ]
1646 combos = pd.concat([combos, extra], ignore_index=True)
1647 combos = combos.sort_values(cols).reset_index(drop=True)
1648 return _named_out(combos, "trials", names.identity() if names else None)
1651def list_parts(
1652 words: pd.DataFrame | None,
1653 fixations: pd.DataFrame | None,
1654 participant: str | None = None,
1655 trial: str | None = None,
1656 *,
1657 raw_gaze: pd.DataFrame | None = None,
1658) -> pd.DataFrame:
1659 """Ordered screens in multipart data, optionally narrowed to one parent.
1661 Single-screen data returns an empty table. A trial recorded as raw gaze
1662 alone takes its screens from ``raw_gaze`` (its ``screen_id``), decided per
1663 trial — so a samples-only trial keeps its screens in a dataset whose other
1664 trials have fixations.
1665 """
1666 words, word_names = _named_in(words, "words", optional=True)
1667 fixations, fix_names = _named_in(fixations, "fixations", optional=True)
1668 gaze_names = None
1669 if raw_gaze is not None:
1670 raw_gaze, gaze_names = _named_in(raw_gaze, "raw_gaze")
1671 names = _call_names(fixations=fix_names, words=word_names, raw_gaze=gaze_names)
1672 catalog = part_catalog(words, fixations)
1673 if raw_gaze is not None and SCREEN_ID in raw_gaze.columns:
1674 # Per trial, as `_select_part` and the app decide it: a trial neither
1675 # words nor fixations cover takes its screens from its samples.
1676 samples = part_catalog(raw_gaze)
1677 covered = _data.trial_keys(words) | _data.trial_keys(fixations)
1678 own = [
1679 (str(p), str(t)) not in covered
1680 for p, t in zip(samples["participant_id"], samples["trial_id"])
1681 ]
1682 if any(own):
1683 catalog = pd.concat([catalog, samples[own]], ignore_index=True)
1684 if participant is not None or trial is not None:
1685 # An id spelled before composite ids escaped a `_` inside a part.
1686 participant, trial = (
1687 None if participant is None else str(participant),
1688 None if trial is None else str(trial),
1689 )
1690 respelled = _data.respell_reading(
1691 participant or "",
1692 trial or "",
1693 zip(catalog["participant_id"], catalog["trial_id"], strict=True),
1694 )
1695 participant = respelled[0] if participant is not None else None
1696 trial = respelled[1] if trial is not None else None
1697 if participant is not None:
1698 catalog = catalog[catalog["participant_id"].astype(str) == str(participant)]
1699 if trial is not None:
1700 catalog = catalog[catalog["trial_id"].astype(str) == str(trial)]
1701 return _named_out(
1702 catalog.reset_index(drop=True), "parts", names.identity() if names else None
1703 )
1706def _resolve_trial(
1707 words: pd.DataFrame,
1708 fixations: pd.DataFrame,
1709 participant: str | None,
1710 trial: str | None,
1711 *,
1712 default_first: bool = False,
1713 raw_gaze: pd.DataFrame | None = None,
1714) -> tuple[str, str]:
1715 """Resolve to one (participant_id, trial_id), validating what was given.
1717 A nonexistent participant/trial always raises — naming which of the two ids
1718 is unknown, a few valid values and the closest spellings. An underspecified
1719 selection matching several trials raises too, unless ``default_first`` picks
1720 the first match (the CLI's behavior, mirroring the app's default selection).
1721 ``raw_gaze`` makes the trials only its samples cover selectable.
1722 """
1723 combos = _cn.to_canonical_frame(list_trials(words, fixations, raw_gaze=raw_gaze))
1724 if combos.empty:
1725 raise ValueError(
1726 "The data holds no trial with both a participant and a trial id."
1727 )
1728 scoped = combos
1729 # Ids written before composite ids escaped a `_` inside a part
1730 # (`data.compose_id`) still find their reading when that is unambiguous.
1731 readings = list(zip(combos["participant_id"], combos["trial_id"], strict=True))
1732 if participant is not None:
1733 participant = _data.respell_reading(
1734 participant, trial if trial is not None else "", readings
1735 )[0]
1736 scoped = scoped[scoped["participant_id"] == str(participant)]
1737 if scoped.empty:
1738 raise ValueError(
1739 f"No trial matches participant={participant!r}: that participant "
1740 f"id is not in the data. {_value_hint(combos, 'participant_id', participant)}"
1741 )
1742 if trial is not None:
1743 trial = _data.respell_reading(
1744 participant if participant is not None else "", trial, readings
1745 )[1]
1746 narrowed = scoped[scoped["trial_id"] == str(trial)]
1747 if narrowed.empty:
1748 if participant is None:
1749 raise ValueError(
1750 f"No trial matches trial={trial!r}: that trial id is not in "
1751 f"the data. {_value_hint(combos, 'trial_id', trial)}"
1752 )
1753 raise ValueError(
1754 f"No trial matches participant={participant!r}, trial={trial!r}: "
1755 f"participant {str(participant)!r} has {len(scoped)} "
1756 f"trial{'' if len(scoped) == 1 else 's'}, none of them {str(trial)!r}. {_value_hint(scoped, 'trial_id', trial)}"
1757 )
1758 scoped = narrowed
1759 if len(scoped) > 1 and not default_first:
1760 preview = ", ".join(
1761 f"({pid!r}, {tid!r})"
1762 for pid, tid in scoped.head(5).itertuples(index=False, name=None)
1763 )
1764 if participant is None and trial is None:
1765 fix = "Pass participant= and trial=."
1766 elif participant is None:
1767 fix = (
1768 f"Trial {str(trial)!r} was read by {scoped['participant_id'].nunique()} "
1769 "participants — pass participant= too."
1770 )
1771 else:
1772 fix = (
1773 f"Participant {str(participant)!r} has {len(scoped)} trials — pass "
1774 "trial= too."
1775 )
1776 raise ValueError(
1777 f"Ambiguous selection: {len(scoped)} trials match "
1778 f"participant={participant!r}, trial={trial!r} (first few: {preview}). "
1779 f"{fix} list_trials(words, fixations) lists all "
1780 f"{len(combos)} trials."
1781 )
1782 row = scoped.iloc[0]
1783 return str(row["participant_id"]), str(row["trial_id"])
1786def _value_hint(combos: pd.DataFrame, column: str, wanted, limit: int = 5) -> str:
1787 """ "Closest / available ids" tail for a failed trial lookup."""
1788 values = [str(v) for v in combos[column].drop_duplicates()]
1789 close = difflib.get_close_matches(str(wanted), values, n=3, cutoff=0.6)
1790 shown = ", ".join(repr(v) for v in values[:limit])
1791 more = f", … (+{len(values) - limit} more)" if len(values) > limit else ""
1792 hint = f"Available: {shown}{more}."
1793 if close:
1794 hint += f" Closest: {', '.join(repr(v) for v in close)}."
1795 return hint
1798def _select_trial(
1799 words: pd.DataFrame,
1800 fixations: pd.DataFrame,
1801 participant: str | None,
1802 trial: str | None,
1803 *,
1804 raw_gaze: pd.DataFrame | None = None,
1805) -> tuple[pd.DataFrame, pd.DataFrame, str, str]:
1806 pid, tid = _resolve_trial(words, fixations, participant, trial, raw_gaze=raw_gaze)
1807 trial_words, trial_fixations = _data.filter_data(
1808 words, fixations, {"participants": [pid], "trials": [tid]}
1809 )
1810 if not trial_fixations.empty and trial_fixations["x"].isna().all():
1811 # AOI-sequence fixations whose coordinates couldn't be reconstructed:
1812 # either no words table was given, or the word/AoI ids matched no box.
1813 raise ValueError(
1814 f"Fixations for participant={pid!r}, trial={tid!r} have no usable "
1815 "coordinates. AOI-sequence datasets (no x/y) need a words table "
1816 "whose word/AOI ids match the fixations', so each fixation can be "
1817 "placed at its word box's center."
1818 )
1819 return trial_words, trial_fixations, pid, tid
1822def _select_part(
1823 words: pd.DataFrame,
1824 fixations: pd.DataFrame,
1825 participant: str | None,
1826 trial: str | None,
1827 screen: str | None,
1828 *,
1829 raw_gaze: pd.DataFrame | None = None,
1830 screen_param: str = "screen",
1831) -> tuple[pd.DataFrame, pd.DataFrame, str, str, str | None]:
1832 """Resolve one logical trial and, for multipart data, exactly one screen.
1834 ``screen_param`` is the keyword the caller took the screen as, so an error
1835 names the one to fix (``screen_b=`` for a comparison's second trial)."""
1836 trial_words, trial_fixations, pid, tid = _select_trial(
1837 words, fixations, participant, trial, raw_gaze=raw_gaze
1838 )
1839 catalog = part_catalog(trial_words, trial_fixations)
1840 if (
1841 catalog.empty
1842 and trial_words.empty
1843 and trial_fixations.empty
1844 and raw_gaze is not None
1845 and SCREEN_ID in raw_gaze.columns
1846 ):
1847 # VIZ-45: a trial recorded as raw gaze alone takes its screens from the
1848 # samples, so one screen's coordinate space is drawn at a time — as for
1849 # words and fixations — rather than every screen stacked into one.
1850 catalog = part_catalog(_data.filter_raw_gaze(raw_gaze, [pid], [tid]))
1851 if catalog.empty:
1852 if screen is not None:
1853 raise ValueError(
1854 f"{screen_param}= names a screen, but participant={pid!r}, "
1855 f"trial={tid!r} has only one; leave {screen_param}= out."
1856 )
1857 return trial_words, trial_fixations, pid, tid, None
1858 available = catalog[SCREEN_ID].astype(str).tolist()
1859 selected = str(screen) if screen is not None else available[0]
1860 if selected not in available:
1861 raise ValueError(
1862 f"Unknown {screen_param}={selected!r} for participant={pid!r}, "
1863 f"trial={tid!r}. "
1864 f"Available: {', '.join(repr(value) for value in available)}."
1865 )
1866 return (
1867 extract_part(trial_words, pid, tid, selected),
1868 extract_part(trial_fixations, pid, tid, selected),
1869 pid,
1870 tid,
1871 selected,
1872 )
1875def _apply_fix_index_range(
1876 trial_fixations: pd.DataFrame, fix_index_range, pid: str, tid: str
1877) -> pd.DataFrame:
1878 """Window the trial to fixations ``start..end`` of ``order_in_trial``.
1880 The headless form of the app's fixation-index slider: both bounds inclusive,
1881 1-based, and applied only to the frame that feeds the figure. Raises rather
1882 than silently drawing an empty scanpath when the window misses the trial."""
1883 if fix_index_range is None:
1884 return trial_fixations
1885 if not isinstance(fix_index_range, (tuple, list)) or len(fix_index_range) != 2:
1886 raise ValueError(
1887 f"fix_index_range must be a (start, end) pair of 1-based fixation "
1888 f"indices, got {fix_index_range!r}."
1889 )
1890 try:
1891 lo, hi = int(fix_index_range[0]), int(fix_index_range[1])
1892 except (TypeError, ValueError) as exc:
1893 raise ValueError(
1894 f"fix_index_range bounds must be integers, got {fix_index_range!r}."
1895 ) from exc
1896 if lo > hi:
1897 raise ValueError(
1898 f"fix_index_range={fix_index_range!r} is empty: start {lo} is after end {hi}."
1899 )
1900 if trial_fixations.empty:
1901 return trial_fixations
1902 if "order_in_trial" not in trial_fixations.columns:
1903 raise ValueError(
1904 "fix_index_range needs the 'order_in_trial' column, which "
1905 "load_scanpath_data() adds during normalization — pass the frames it "
1906 "returns."
1907 )
1908 order = trial_fixations["order_in_trial"]
1909 windowed = trial_fixations[(order >= lo) & (order <= hi)]
1910 if windowed.empty:
1911 raise ValueError(
1912 f"fix_index_range=({lo}, {hi}) selects no fixations: participant={pid!r}, "
1913 f"trial={tid!r} has {len(trial_fixations)} fixations "
1914 f"(order_in_trial {int(order.min())}–{int(order.max())})."
1915 )
1916 return windowed
1919def _figure_kwargs(overrides: dict) -> dict:
1920 settings = {**CANONICAL_FIGURE_DEFAULTS, **_expand_palette(overrides)}
1921 if settings.get("heatmap_metric") == "counts":
1922 settings["heatmap_metric"] = None
1923 return settings
1926_SPELLINGS = (("grey", "gray"), ("colour", "color"))
1929def _spelling_variants(text: str) -> list[str]:
1930 """``text`` as written, all-US and all-UK (#374: the palette names moved to
1931 US spelling, and both spellings must keep naming the same palette)."""
1932 import re
1934 out = [text]
1935 for pick in (1, 0):
1936 variant = text
1937 for pair in _SPELLINGS:
1938 variant = re.sub(pair[1 - pick], pair[pick], variant, flags=re.IGNORECASE)
1939 out.append(variant)
1940 return list(dict.fromkeys(out))
1943def resolve_palette(value: object) -> str:
1944 """The `constants.PALETTES` name ``value`` stands for: the app's name in
1945 either spelling (``"Print / grayscale"`` or ``"Print / greyscale"``), the
1946 short name (``"print"``), any case. Raises ``ValueError`` otherwise."""
1947 from .constants import PALETTES
1949 error = None
1950 for candidate in _spelling_variants(str(value)):
1951 try:
1952 name = normalize_palette(candidate)
1953 except ValueError as exc:
1954 error = error or exc
1955 continue
1956 if name in PALETTES:
1957 return name
1958 for spelled in _spelling_variants(name):
1959 if spelled in PALETTES:
1960 return spelled
1961 raise error or ValueError(f"Unknown palette {value!r}.")
1964def _check_colorscales(overrides: dict) -> None:
1965 """#374: a name Plotly doesn't know raised its own lower-cased
1966 ``PlotlyError`` from inside the builder; name the option instead."""
1967 from plotly.colors import get_colorscale
1968 from plotly.exceptions import PlotlyError
1970 for key in ("heatmap_colorscale", "fixation_colorscale"):
1971 value = overrides.get(key)
1972 if not isinstance(value, str):
1973 continue
1974 try:
1975 get_colorscale(value)
1976 except PlotlyError:
1977 raise ValueError(
1978 f"{key}={value!r} is not a Plotly color scale. Any named scale "
1979 "works, e.g. Viridis, Greens, Blues, Cividis; append _r to "
1980 "reverse one."
1981 ) from None
1984def _expand_palette(overrides: dict) -> dict:
1985 """Expand a ``palette=`` override into the color kwargs it stands for.
1987 ``palette`` names a set of color defaults tuned for a medium — screen,
1988 colorblind viewers, a black & white print, a projector. It's a *preset*, so
1989 any color the caller also passes explicitly wins over it::
1991 sps.plot_scanpath(w, f, palette="Print / grayscale")
1992 sps.plot_scanpath(w, f, palette="Default (colorblind-safe)", saccade_color="#000")
1994 The palette itself isn't a figure kwarg, so it's consumed here rather than
1995 forwarded. Raises on an unknown name — a silent fallback to the default
1996 palette would quietly produce the wrong figure for a print run.
1998 Every enumerated option is read here too (`plots.normalize_option_values`):
1999 ``heatmap_norm="log"`` is ``"Log"``, and a value that is none of the
2000 choices raises rather than drawing the default.
2001 """
2002 overrides = normalize_option_values(overrides)
2003 _check_colorscales(overrides)
2004 name = overrides.get("palette")
2005 if name is None:
2006 return overrides
2007 name = resolve_palette(name) # "print", "high-contrast", any case or spelling
2008 expanded = dict(overrides)
2009 expanded.pop("palette")
2010 # `word_label_color` is `text_color` on the figure builders.
2011 settings = palette_settings(name)
2012 settings["text_color"] = settings.pop("word_label_color")
2013 for key, value in settings.items():
2014 expanded.setdefault(key, value)
2015 return expanded
2018_NAMED_FIGURE_PARAMS = frozenset(
2019 {
2020 "canvas_size",
2021 "base_font_size",
2022 "font_family",
2023 "title",
2024 "caption",
2025 "screen",
2026 "raw_gaze",
2027 "illustration",
2028 "illustration_label",
2029 "fix_index_range",
2030 "column_names",
2031 }
2032)
2035def _reject_unknown_options(overrides: dict, valid, func_name: str) -> None:
2036 """Fail on a misspelled/unsupported keyword, naming the closest valid ones.
2038 Forwarding blindly would surface as ``make_scanpath_figure() got an
2039 unexpected keyword argument`` — an internal name the caller never typed."""
2040 unknown = sorted(set(overrides) - set(valid))
2041 if not unknown:
2042 return
2043 parts = []
2044 for key in unknown:
2045 # The builders' named parameters too: `canvas=` means `canvas_size=`.
2046 close = difflib.get_close_matches(
2047 key, sorted(set(valid) | _NAMED_FIGURE_PARAMS), n=3, cutoff=0.6
2048 )
2049 suffix = (
2050 f" (did you mean {', '.join(repr(c) for c in close)}?)" if close else ""
2051 )
2052 parts.append(f"{key!r}{suffix}")
2053 raise TypeError(
2054 f"{func_name}() got an unexpected keyword argument: {', '.join(parts)}. "
2055 f"help({func_name}) lists its parameters and figure_options() the "
2056 "figure options with their defaults."
2057 )
2060#: Figure options whose value names a column → (the table it is read from, its
2061#: CLI flag, the values that are not columns, what to do instead). EXP-17: the
2062#: builders look the column up and draw *nothing* when it is missing, so a
2063#: misspelling rendered a flat-coloured / unmarked figure without a word.
2064_COLUMN_OPTIONS = {
2065 "color_by": (
2066 "fixations",
2067 "--color-by",
2068 (UNIFORM_COLOR_FIELD, "line"),
2069 f"Use {UNIFORM_COLOR_FIELD!r} for one flat color, 'line' to color by "
2070 "text line, or one of the columns below.",
2071 ),
2072 "highlight_column": (
2073 "words",
2074 "--highlight-column",
2075 (),
2076 "It names the boolean words column marking the text to highlight; pass "
2077 "None ('' on the CLI) to highlight nothing.",
2078 ),
2079}
2082#: DATA-66: the figure options whose value names one column, and those naming a
2083#: list of them — each takes the dataset's own name as well as the internal one.
2084_ONE_COLUMN_OPTIONS = (
2085 "color_by",
2086 "highlight_column",
2087 "heatmap_metric",
2088 "word_hover_measure",
2089 "word_heatmap_col",
2090 "x_field",
2091 "y_field",
2092)
2093_COLUMN_LIST_OPTIONS = ("word_hover_fields", "fixation_hover_fields")
2094#: …and of those, the ones naming a column of the words table.
2095_WORD_OPTIONS = frozenset(
2096 ("highlight_column", "word_hover_measure", "word_heatmap_col", "word_hover_fields")
2097)
2100def _canonical_options(
2101 overrides: dict,
2102 names: ColumnNames | None,
2103 *,
2104 words: ColumnNames | None = None,
2105) -> dict:
2106 """``overrides`` with every column it names in the internal vocabulary —
2107 a word option in the words table's names (``words``) before the merged
2108 map's.
2110 ``heatmap_metric`` is checked here too: the heatmap weights by the fixation
2111 duration or counts fixations, and any other value used to count silently —
2112 which, once the dataset's own names are accepted, a misspelled name would."""
2113 out = dict(overrides)
2115 def canonical(option: str, value) -> str:
2116 if words is not None and option in _WORD_OPTIONS:
2117 found = words.to_canonical(value)
2118 if found != str(value):
2119 return found
2120 return names.to_canonical(value) if names is not None else value
2122 if names is not None or words is not None:
2123 for option in _ONE_COLUMN_OPTIONS:
2124 if isinstance(out.get(option), str):
2125 out[option] = canonical(option, out[option])
2126 for option in _COLUMN_LIST_OPTIONS:
2127 if out.get(option) is not None and not isinstance(out[option], str):
2128 out[option] = [canonical(option, value) for value in out[option]]
2129 metric = out.get("heatmap_metric")
2130 if metric not in (None, "duration_ms", "counts"):
2131 duration = (
2132 names.label("duration_ms")
2133 if names is not None and names.source("duration_ms")
2134 else "duration_ms"
2135 )
2136 raise ValueError(
2137 f"heatmap_metric={metric!r} (--heatmap-metric on the CLI) must be "
2138 f"the fixation duration ({duration!r}) or 'counts'."
2139 )
2140 return out
2143def _column_labels(
2144 names: ColumnNames | None,
2145 word_frame,
2146 fixation_frame,
2147 *,
2148 words: ColumnNames | None = None,
2149) -> dict | None:
2150 """`FigureSettings.column_labels` for frames that carried names — the
2151 figure's text in the dataset's own names, as the app writes it, a word
2152 column also under the words table's own name (`table_figure_labels`)."""
2153 if names is None:
2154 return None
2155 labels = names.figure_labels(
2156 [
2157 column
2158 for frame in (word_frame, fixation_frame)
2159 if frame is not None
2160 for column in frame
2161 ]
2162 )
2163 if words is not None and word_frame is not None:
2164 for column, label in words.figure_labels(word_frame.columns).items():
2165 if labels.get(column) != label:
2166 labels[f"words:{column}"] = label
2167 return labels
2170def _check_column_options(
2171 overrides: dict, *, words: pd.DataFrame, fixations: pd.DataFrame
2172) -> None:
2173 """Raise when an option the caller *named* points at no column.
2175 Only explicit values are checked: ``highlight_column`` defaults to OneStop's
2176 ``is_in_aspan``, which most corpora do not have and which the builder then
2177 rightly skips. An empty table is not checked — there is nothing to color."""
2178 frames = {"words": words, "fixations": fixations}
2179 for name, (kind, flag, synthetic, advice) in _COLUMN_OPTIONS.items():
2180 value = overrides.get(name)
2181 if value is None or value == "" or value in synthetic:
2182 continue
2183 frame = frames[kind]
2184 present = [str(column) for column in frame.columns]
2185 if frame.empty or str(value) in present:
2186 continue
2187 # Internal helper columns (`_text_id_mapped`) are not the user's to name.
2188 visible = [column for column in present if not column.startswith("_")]
2189 close = difflib.get_close_matches(str(value), visible, n=3, cutoff=0.6)
2190 hint = f" Closest: {', '.join(repr(c) for c in close)}." if close else ""
2191 raise ValueError(
2192 f"{name}={value!r} ({flag} on the CLI) names no column of the "
2193 f"{kind} table.{hint} {advice} Columns present ({len(visible)}): "
2194 f"{_column_preview(frame[visible])}."
2195 )
2198def figure_options(kind: str = "static", *, choices: bool = False) -> dict:
2199 """Every figure keyword a builder accepts → the default it renders with.
2201 With ``choices=True`` each name maps to ``{"default": …, "choices": …}``,
2202 where ``choices`` is the tuple of values an enumerated option takes
2203 (``heatmap_norm``: ``("Linear", "Log")``) and ``None`` for a free one. An
2204 enumerated option takes any spelling of a choice — case, spaces, ``-`` and
2205 ``_`` are ignored, so the CLI's ``"log"`` and ``"mark-border"`` work — and
2206 raises ``ValueError`` listing them for anything else.
2208 ``kind="static"`` covers [`plot_scanpath`][scanpath_studio.api.plot_scanpath],
2209 ``kind="animation"`` [`animate_scanpath`][scanpath_studio.api.animate_scanpath]
2210 (whose builder supports a subset), and ``kind="comparison"``
2211 [`compare_scanpaths`][scanpath_studio.api.compare_scanpaths]. The values are the
2212 defaults a call actually renders with (the app's Scanpath design) — so a scripted caller can diff its intended
2213 settings against what it would get::
2215 {k: v for k, v in sps.figure_options().items() if k.startswith("show_")}
2216 """
2217 if kind == "static":
2218 params = _STATIC_FIGURE_PARAMS
2219 defaults = FigureSettings.defaults(params) | CANONICAL_FIGURE_DEFAULTS
2220 elif kind == "animation":
2221 params = _ANIMATION_FIGURE_PARAMS
2222 defaults = FigureSettings.defaults(params) | _animation_defaults()
2223 elif kind == "comparison":
2224 # CMP-9: `compare_scanpaths` validates against this set, and its TypeError
2225 # points the caller here — so it has to be answerable.
2226 params = _COMPARISON_FIGURE_PARAMS
2227 defaults = FigureSettings.defaults(params) | {
2228 key: value
2229 for key, value in CANONICAL_FIGURE_DEFAULTS.items()
2230 if key in _COMPARISON_FIGURE_PARAMS
2231 }
2232 else:
2233 raise ValueError(
2234 f"Unknown kind {kind!r}; use 'static', 'animation' or 'comparison'."
2235 )
2236 options = {}
2237 for name in sorted(params):
2238 if name in defaults:
2239 # Some public defaults are ordered field lists. Return an independent
2240 # value so callers can edit the option reference without changing the
2241 # canonical defaults used by every later plot.
2242 options[name] = deepcopy(defaults[name])
2243 else: # pragma: no cover - every option is a FigureSettings field
2244 options[name] = None
2245 if choices:
2246 return {
2247 name: {"default": default, "choices": FIGURE_OPTION_CHOICES.get(name)}
2248 for name, default in options.items()
2249 }
2250 return options
2253def _animation_defaults() -> dict:
2254 """The canonical defaults the animation builder can actually take."""
2255 return {
2256 key: value
2257 for key, value in CANONICAL_FIGURE_DEFAULTS.items()
2258 if key in _ANIMATION_FIGURE_PARAMS
2259 }
2262def _apply_drift_correction(
2263 trial_words: pd.DataFrame,
2264 trial_fixations: pd.DataFrame,
2265 settings: dict,
2266 method: str | None,
2267 connectors: bool,
2268 explicit: dict,
2269) -> pd.DataFrame:
2270 """Snap fixations to their assigned text line, in place of the raw y.
2272 Mirrors what the app does on the static plot (``tabs.render_single_trial_tab``):
2273 run ``alignment.correct``, color the corrected fixations by line, and
2274 optionally draw original→corrected connectors. Returns the fixations to plot.
2275 """
2276 if method is None or str(method).lower() == "off":
2277 return trial_fixations
2278 # PRE-21: raise rather than ignore. A share link degrades silently because a
2279 # human can see the figure and the rail; a script cannot, so quietly
2280 # returning uncorrected fixations under a stated `drift_correction=` would
2281 # be a wrong result with no signal. Name the env var so it is one step to fix.
2282 if not drift_correction_enabled():
2283 raise ValueError(
2284 "drift_correction is not available in this release (vertical "
2285 f"drift correction is not fully integrated yet). Set "
2286 f"{EXPERIMENTAL_ENV_VAR}=1 to enable it, or pass drift_correction=None."
2287 )
2288 from . import alignment as _alignment # local: pulls in scipy
2290 name = str(method).lower()
2291 if name not in _alignment.ALGORITHMS:
2292 raise ValueError(
2293 f"Unknown drift_correction {method!r}; choose one of "
2294 f"{', '.join(_alignment.ALGORITHMS)} (or None to leave the fixations "
2295 "uncorrected)."
2296 )
2297 if trial_fixations.empty or trial_words.empty:
2298 return trial_fixations
2299 original_y = tuple(pd.to_numeric(trial_fixations["y"], errors="coerce"))
2300 corrected, _ = _alignment.correct(trial_fixations, trial_words, method=name)
2301 # Colouring by line is what makes the correction legible; an explicit
2302 # `color_by_line=` still wins.
2303 if "color_by_line" not in explicit:
2304 settings["color_by_line"] = True
2305 if connectors:
2306 settings["show_connectors"] = True
2307 settings["connector_y"] = original_y
2308 return corrected
2311def _check_canvas_size(canvas_size) -> None:
2312 """#374: ``canvas_size="1920x1080"`` was read character by character into a
2313 1 x 9 px canvas and drew an empty figure."""
2314 if canvas_size is None:
2315 return
2316 try:
2317 pair = not isinstance(canvas_size, str) and len(tuple(canvas_size)) == 2
2318 except TypeError:
2319 pair = False
2320 if not pair:
2321 raise ValueError(
2322 "canvas_size must be a (width, height) pair in pixels, e.g. "
2323 f"(2560, 1440); got {canvas_size!r} (--canvas WxH on the CLI)."
2324 )
2327def plot_scanpath(
2328 words: pd.DataFrame | None = None,
2329 fixations: pd.DataFrame | None = None,
2330 participant: str | None = None,
2331 trial: str | None = None,
2332 *,
2333 screen: str | None = None,
2334 canvas_size: tuple[int, int] | None = None,
2335 base_font_size: int = 16,
2336 font_family: str = FONT_FAMILY,
2337 raw_gaze: pd.DataFrame | None = None,
2338 drift_correction: str | None = None,
2339 drift_connectors: bool = False,
2340 fix_index_range: tuple[int, int] | None = None,
2341 illustration: bool = False,
2342 illustration_label: str = "auto",
2343 title: str = "",
2344 caption: str = "",
2345 column_names: dict | None = None,
2346 **figure_overrides,
2347) -> go.Figure:
2348 """Build one trial's scanpath figure (by default the app's Scanpath design).
2350 ``words`` / ``fixations`` are normalized frames from
2351 [`load_scanpath_data`][scanpath_studio.api.load_scanpath_data]. ``participant`` /
2352 ``trial`` may be omitted when the frames hold exactly one trial. ``canvas_size``
2353 is the monitor size in px; by default it is estimated from the data extents — pass
2354 the real monitor resolution (e.g. ``(2560, 1440)`` for OneStop) to keep coordinates
2355 true to scale. For a multipart trial, ``screen`` selects one child screen; omitting
2356 it selects the first recorded screen and never concatenates coordinate spaces.
2357 ``raw_gaze`` is a frame from [`load_raw_gaze`][scanpath_studio.api.load_raw_gaze],
2358 filtered to the selected trial and drawn as recorded. It can be the only table:
2359 for a dataset recorded as raw gaze alone pass ``None`` for ``words`` and
2360 ``fixations`` (``plot_scanpath(raw_gaze=samples, trial=…)``) — the trial is
2361 looked up in the samples, the canvas is estimated from their extent, and the
2362 figure is the samples alone. Nothing is derived from them: no fixations are
2363 detected, so the fixation, saccade and heatmap layers stay empty.
2365 ``drift_correction`` / ``drift_connectors`` are experimental: without
2366 ``SCANPATH_EXPERIMENTAL=1`` any ``drift_correction`` other than ``None`` /
2367 ``"off"`` raises ``ValueError``.
2369 ``fix_index_range=(start, end)`` draws only fixations ``start``
2370 through ``end`` (1-based, both inclusive) of the trial — the headless form of
2371 the app's fixation-index window.
2373 ``title`` / ``caption`` stamp a title/caption band onto the figure
2374 without shrinking the plot area, like the app's *Title & labels* — literal
2375 text here, not the app's ``{trial_id}``-style pattern, since the caller
2376 already knows which trial this is.
2378 ``illustration=True`` applies the Illustration preset (snapped fixations,
2379 arced saccades, uniform colors, no heatmap or word boxes); keywords you pass
2380 still win. ``illustration_label`` is ``"auto"`` (label the figure when it no
2381 longer shows the data as recorded), ``"show"`` or ``"hide"``. ``palette=``
2382 (``"default"``, ``"print"`` or ``"high-contrast"``, or the app's names) sets
2383 a group of colors at once; a color you pass explicitly wins.
2385 Remaining keywords override the app's defaults and are forwarded to
2386 `plots.make_scanpath_figure` (e.g. ``show_heatmap=True``,
2387 ``color_by="pass_index"``, ``x_field="order_in_trial"``); an unknown keyword raises
2388 a ``TypeError`` naming the closest valid options, and
2389 [`figure_options`][scanpath_studio.api.figure_options] lists them all with their
2390 defaults (``choices=True`` adds the values each enumerated option takes; a
2391 value is matched ignoring case, spaces, ``-`` and ``_``, and any other value
2392 raises a ``ValueError``). A ``color_by`` / ``highlight_column`` naming a column the trial's table
2393 doesn't have raises a ``ValueError`` naming the closest ones, rather than drawing
2394 without it.
2396 Frames under the dataset's own column names (what ``load_scanpath_data``
2397 returns by default) are read through the map they carry, and an option naming
2398 a column takes either name. ``column_names`` is that map for frames loaded with
2399 ``names="canonical"`` (``data.column_names``): the options then take the
2400 dataset's names too, and the figure's text uses them.
2401 """
2402 if illustration:
2403 figure_overrides = {
2404 "show_words": False,
2405 "show_word_labels": True,
2406 "show_fixations": True,
2407 "show_order": False,
2408 "show_saccades": True,
2409 "show_saccade_arrows": False,
2410 "show_heatmap": False,
2411 "color_by": UNIFORM_COLOR_FIELD,
2412 "saccade_color_mode": "Uniform",
2413 "saccade_render_mode": "Arc",
2414 "fixation_snap_to_word": True,
2415 "fixation_opacity": 1.0,
2416 **figure_overrides,
2417 }
2418 _reject_unknown_options(
2419 figure_overrides, _STATIC_FIGURE_PARAMS | {"palette"}, "plot_scanpath"
2420 )
2421 _check_canvas_size(canvas_size)
2422 words, word_names = _named_in(words, "words", optional=True)
2423 fixations, fix_names = _named_in(fixations, "fixations", optional=True)
2424 gaze_names = None
2425 if raw_gaze is not None:
2426 raw_gaze, gaze_names = _named_in(raw_gaze, "raw_gaze")
2427 names = _call_names(
2428 column_names, fixations=fix_names, words=word_names, raw_gaze=gaze_names
2429 )
2430 word_side = _table_names(column_names, "words", word_names)
2431 figure_overrides = _canonical_options(figure_overrides, names, words=word_side)
2432 trial_words, trial_fixations, pid, tid, selected_screen = _select_part(
2433 words, fixations, participant, trial, screen, raw_gaze=raw_gaze
2434 )
2435 if raw_gaze is not None:
2436 raw_gaze = _data.filter_raw_gaze(raw_gaze, [pid], [tid])
2437 if selected_screen is not None and SCREEN_ID in raw_gaze.columns:
2438 raw_gaze = extract_part(raw_gaze, pid, tid, selected_screen)
2439 _check_column_options(
2440 figure_overrides, words=trial_words, fixations=trial_fixations
2441 )
2442 full_fix_range = None
2443 if not trial_fixations.empty and "order_in_trial" in trial_fixations.columns:
2444 order = pd.to_numeric(
2445 trial_fixations["order_in_trial"], errors="coerce"
2446 ).dropna()
2447 if not order.empty:
2448 full_fix_range = (int(order.min()), int(order.max()))
2449 if canvas_size is None:
2450 canvas_size = screen_canvas_size(trial_words)
2451 if canvas_size is None:
2452 canvas_size = screen_canvas_size(trial_fixations)
2453 if canvas_size is None:
2454 canvas_size = _recorded_screen(words, fixations)
2455 if canvas_size is None:
2456 # VIZ-45: a trial with no fixations is sized from its samples, as the
2457 # app sizes a raw-gaze-only dataset's canvas.
2458 canvas_size = _data.compute_canvas_size(
2459 trial_words,
2460 trial_fixations
2461 if not trial_fixations.empty or raw_gaze is None
2462 else raw_gaze,
2463 )
2464 # Window first, correct second — the app's order (tabs._slice_fix_range runs
2465 # before alignment.correct), so a windowed correction sees only the kept
2466 # fixations.
2467 trial_fixations = _apply_fix_index_range(trial_fixations, fix_index_range, pid, tid)
2468 settings = _figure_kwargs(figure_overrides)
2469 label_mode = str(illustration_label).capitalize()
2470 if label_mode not in {"Auto", "Show", "Hide"}:
2471 raise ValueError("illustration_label must be 'auto', 'show', or 'hide'.")
2472 if "illustration_reasons" not in figure_overrides:
2473 from .illustration import illustration_reasons, resolve_label_reasons
2475 reasons = illustration_reasons(
2476 settings,
2477 fix_index_range=fix_index_range,
2478 full_fixation_range=full_fix_range,
2479 )
2480 settings["illustration_reasons"] = resolve_label_reasons(label_mode, reasons)
2481 # Spatial fields are explicit kwargs of make_scanpath_figure, so they can't
2482 # ride along in **settings without a "multiple values" TypeError.
2483 x_field = settings.pop("x_field", "x")
2484 y_field = settings.pop("y_field", "y")
2485 trial_fixations = _apply_drift_correction(
2486 trial_words,
2487 trial_fixations,
2488 settings,
2489 drift_correction,
2490 drift_connectors,
2491 figure_overrides,
2492 )
2493 if raw_gaze is not None:
2494 settings.setdefault("show_raw_gaze", True)
2495 render_settings = FigureSettings.from_mapping(
2496 settings,
2497 canvas_width=int(canvas_size[0]),
2498 canvas_height=int(canvas_size[1]),
2499 base_font_size=int(base_font_size),
2500 font_family=font_family,
2501 x_field=x_field,
2502 y_field=y_field,
2503 column_labels=_column_labels(
2504 names, trial_words, trial_fixations, words=word_side
2505 ),
2506 )
2507 fig = make_scanpath_figure(
2508 trial_words,
2509 trial_fixations,
2510 settings=render_settings,
2511 raw_gaze=raw_gaze,
2512 )
2513 annotate_figure(fig, title=title, caption=caption)
2514 return fig
2517def animate_scanpath(
2518 words: pd.DataFrame | None = None,
2519 fixations: pd.DataFrame | None = None,
2520 participant: str | None = None,
2521 trial: str | None = None,
2522 *,
2523 screen: str | None = None,
2524 screen_b: str | None = None,
2525 canvas_size: tuple[int, int] | None = None,
2526 base_font_size: int = 16,
2527 font_family: str = FONT_FAMILY,
2528 playback_speed: float = 1.0,
2529 autoplay: bool = True,
2530 fix_index_range: tuple[int, int] | None = None,
2531 fix_index_range_b: tuple[int, int] | None = None,
2532 illustration_label: str = "auto",
2533 title: str = "",
2534 caption: str = "",
2535 column_names: dict | None = None,
2536 trial_b: tuple[str, str] | None = None,
2537 dataset_b: str | None = None,
2538 setup: SetupSnapshot | None = None,
2539 setup_b: SetupSnapshot | None = None,
2540 **animation_overrides,
2541) -> go.Figure:
2542 """Build the animated scanpath replay for one trial.
2544 Same trial selection, canvas and column-name semantics as
2545 [`plot_scanpath`][scanpath_studio.api.plot_scanpath] (``column_names`` included),
2546 and ``screen`` selection for multipart trials. The replay takes the reading time divided by
2547 ``playback_speed``: save it as interactive HTML with
2548 [`save_figure`][scanpath_studio.api.save_figure], whose page keeps that clock
2549 itself, or rasterize it to GIF/MP4 with `animation_export.export_animation`, which
2550 lasts as long. (`fig.show()` plays it on Plotly's own frame queue, which runs
2551 slow.) ``fix_index_range=(start, end)`` replays only that window of the trial's
2552 fixations (1-based, inclusive), like
2553 [`plot_scanpath`][scanpath_studio.api.plot_scanpath].
2555 With ``autoplay`` (default ``True``) the saved interactive HTML auto-starts
2556 the replay on load *at ``playback_speed``* —
2557 [`save_figure`][scanpath_studio.api.save_figure] honors the marker the builder
2558 stamps on the figure. Pass ``autoplay=False`` to save a figure that opens paused
2559 (press ▶ Play to run it). Autoplay only affects the interactive HTML; a GIF/MP4
2560 always plays from its first frame.
2562 When ``playback_speed`` is not ``1``, the automatic Illustration label says
2563 the replay timing was changed. ``illustration_label`` accepts ``"auto"``,
2564 ``"show"``, or ``"hide"`` like [`plot_scanpath`][scanpath_studio.api.plot_scanpath].
2566 In a co-animation ``fix_index_range`` windows A only (the app's
2567 rule — A's slider never cuts B), ``fix_index_range_b`` windows B, and
2568 ``fixation_flags_b`` gives B flags of its own (``None``: A's
2569 ``fixation_flags``, or the ``fixation_flags`` of ``style_b`` when it
2570 names some).
2572 ``style_a`` / ``style_b`` style the two scanpaths of a co-animation as they
2573 style [`compare_scanpaths`][scanpath_studio.api.compare_scanpaths]' — the
2574 same keys (``fix_color``, ``marker_size_range``, ``opacity``, ``hollow``,
2575 ``saccade_color``, ``saccade_style``, ``saccade_width``), resolved the same
2576 way, so the replay and the static comparison draw each trial alike. The
2577 replay has no saccade-class filter, so a style naming ``saccade_classes``
2578 raises ``ValueError``. A lone replay ignores both.
2580 ``trial_b=(participant, trial)`` co-animates a second trial on the same
2581 clock, like the app's Animate + Compare. It is looked up in ``words_b`` /
2582 ``fixations_b`` when given, else in ``words`` / ``fixations`` — the way
2583 [`compare_scanpaths`][scanpath_studio.api.compare_scanpaths] takes it.
2584 Without ``trial_b``, ``words_b`` / ``fixations_b`` must hold one trial; B
2585 frames holding several raise ``ValueError`` rather than drawing them all. A
2586 multipart B is drawn at ``screen_b`` — looked up in B's own trial — or at
2587 its first recorded screen without it, as A is with ``screen``.
2589 **Two datasets.** Both readings are drawn in A's coordinates, so a
2590 co-animation is an overlay, and a trial from another dataset has to share
2591 A's screen. Name that dataset with ``dataset_b`` (or give its ``setup_b``)
2592 and the pair is checked the way `compare_scanpaths` checks an overlay: two
2593 different canvases raise ``IncomparableScreensError``, a ``ValueError``,
2594 rather than draw. ``setup`` / ``setup_b`` are
2595 `experimental_setup.SetupSnapshot` values; a side without one is read off
2596 its data — the extent of that one trial, which rarely spans the whole
2597 screen, so state both when you know them — and ``canvas_size`` covers A
2598 when you only have a resolution. ``dataset_b`` also prefixes B's
2599 participant ids with the dataset's name, as `compare_scanpaths` does, so a
2600 hover says whose participant it is. ``words_b`` / ``fixations_b`` passed without
2601 either are taken to be from A's dataset, as `render` passes them for
2602 ``--compare-with`` alone, and are not checked: two readings of one corpus
2603 can span different extents, and inferring a canvas from each would refuse
2604 pairs that shared a screen.
2606 The animation builder accepts a subset of the static figure's options
2607 (``show_words``, ``show_word_labels``, ``show_saccades``, ``show_order``, styling,
2608 and second-scanpath overlays) — see ``figure_options("animation")``; an unsupported
2609 key raises a ``ValueError`` naming the valid ones. The shared options default to the same values as
2610 [`plot_scanpath`][scanpath_studio.api.plot_scanpath] (`CANONICAL_FIGURE_DEFAULTS`),
2611 so the replay matches the static figure. ``palette=`` works here too; the
2612 colors it implies that the animation doesn't support are dropped rather than
2613 raising, since the caller named a look, not those individual keys.
2615 ``title`` / ``caption`` — same as
2616 [`plot_scanpath`][scanpath_studio.api.plot_scanpath].
2618 The replay is made of fixations, so a trial without any — one recorded as
2619 raw gaze alone, or a words-only one — raises ``ValueError`` rather than
2620 returning an empty replay, and ``raw_gaze=`` is refused: the replay draws no
2621 raw-gaze layer, and nothing detects fixations from samples. Draw samples with
2622 [`plot_scanpath`][scanpath_studio.api.plot_scanpath]`(raw_gaze=…)`.
2623 """
2624 _check_canvas_size(canvas_size)
2625 if "raw_gaze" in animation_overrides:
2626 raise ValueError(
2627 "animate_scanpath replays fixations and has no raw-gaze layer, and "
2628 "Scanpath Studio does not detect fixations from gaze samples. Draw the "
2629 "samples with plot_scanpath(..., raw_gaze=...) instead."
2630 )
2631 valid = set(_ANIMATION_FIGURE_PARAMS)
2632 explicit = set(animation_overrides) - {"palette"}
2633 animation_overrides = _expand_palette(animation_overrides)
2634 # Only the keys the caller named are held to the "is this supported?" rule;
2635 # a palette's extras (heatmap colorscale, highlight text colour, …) that the
2636 # animation has no parameter for are simply dropped.
2637 animation_overrides = {
2638 k: v for k, v in animation_overrides.items() if k in valid or k in explicit
2639 }
2640 unknown = explicit - valid
2641 if unknown:
2642 raise ValueError(
2643 f"animate_scanpath() does not support: {', '.join(sorted(unknown))}. "
2644 "figure_options('animation') lists the options it takes."
2645 )
2646 for side in ("style_a", "style_b"):
2647 style = animation_overrides.get(side)
2648 if isinstance(style, dict) and style.get("saccade_classes") is not None:
2649 raise ValueError(
2650 f"{side}['saccade_classes'] filters a comparison figure's "
2651 "saccades; the co-animation has no saccade-class filter. Drop "
2652 "it, or draw the pair with compare_scanpaths."
2653 )
2654 named = {k: v for k, v in animation_overrides.items() if k in explicit}
2655 # Same defaults as the static figure for every option both builders share, so
2656 # `plot_scanpath` and `animate_scanpath` don't render the same trial
2657 # differently (the app feeds both from one settings dict).
2658 animation_overrides = {**_animation_defaults(), **animation_overrides}
2659 words, word_names = _named_in(words, "words", optional=True)
2660 fixations, fix_names = _named_in(fixations, "fixations", optional=True)
2661 names = _call_names(column_names, fixations=fix_names, words=word_names)
2662 word_side = _table_names(column_names, "words", word_names)
2663 animation_overrides = _canonical_options(
2664 animation_overrides, names, words=word_side
2665 )
2666 named = _canonical_options(named, names, words=word_side)
2667 trial_words, trial_fixations, pid, tid, _selected_screen = _select_part(
2668 words, fixations, participant, trial, screen
2669 )
2670 if trial_fixations.empty:
2671 raise ValueError(
2672 f"participant={pid!r}, trial={tid!r} has no fixations to replay — the "
2673 "replay is built from fixations. A trial recorded as raw gaze alone "
2674 "can be drawn with plot_scanpath(..., raw_gaze=...); its samples are "
2675 "not turned into fixations."
2676 )
2677 _check_column_options(named, words=trial_words, fixations=trial_fixations)
2678 full_fix_range = None
2679 if not trial_fixations.empty and "order_in_trial" in trial_fixations.columns:
2680 full_order = pd.to_numeric(
2681 trial_fixations["order_in_trial"], errors="coerce"
2682 ).dropna()
2683 if not full_order.empty:
2684 full_fix_range = (int(full_order.min()), int(full_order.max()))
2685 trial_fixations = _apply_fix_index_range(trial_fixations, fix_index_range, pid, tid)
2686 # A's screen: a stated `setup`, else `canvas_size`, else read off the data —
2687 # the order `compare_scanpaths` resolves it in, and what CMP-21's gate reads.
2688 setup_a = _compare_setup(
2689 setup, canvas_size, trial_words, trial_fixations, side="setup"
2690 )
2691 passed_b = (
2692 animation_overrides.pop("words_b", None),
2693 animation_overrides.pop("fixations_b", None),
2694 )
2695 second_dataset = dataset_b is not None or setup_b is not None
2696 if second_dataset and all(frame is None for frame in passed_b):
2697 raise ValueError(
2698 "dataset_b / setup_b describe scanpath B's own dataset, but neither "
2699 "words_b nor fixations_b was passed. Pass B's frames too, or leave "
2700 "both out to draw trial_b from these frames."
2701 )
2702 if screen_b is not None and trial_b is None and all(f is None for f in passed_b):
2703 raise ValueError(
2704 "screen_b= picks scanpath B's screen, but there is no scanpath B. "
2705 "Pass trial_b=(participant, trial) too."
2706 )
2707 words_b, fixations_b = _second_reading(
2708 words, fixations, *passed_b, trial_b, screen_b=screen_b
2709 )
2710 if second_dataset and fixations_b is not None and not fixations_b.empty:
2711 _refuse_co_animation_across_screens(
2712 setup_a,
2713 setup_b,
2714 words_b,
2715 fixations_b,
2716 a_inferred=setup is None and canvas_size is None,
2717 )
2718 elif fixations_b is not None and not fixations_b.empty:
2719 # One dataset, two screen sizes it knows of: refused as the app and
2720 # `compare_scanpaths`' overlay refuse them.
2721 same_b = _same_dataset_setup_b(
2722 setup_a,
2723 setup_b,
2724 a_known=setup is not None or canvas_size is not None,
2725 words_a=trial_words,
2726 fixations_a=trial_fixations,
2727 words_b=words_b,
2728 fixations_b=fixations_b,
2729 )
2730 if same_b is not None:
2731 _refuse_co_animation_across_screens(
2732 setup_a, same_b, words_b, fixations_b, a_inferred=False
2733 )
2734 full_fix_range_b = None
2735 if (
2736 fixations_b is not None
2737 and not fixations_b.empty
2738 and "order_in_trial" in fixations_b.columns
2739 ):
2740 order_b = pd.to_numeric(fixations_b["order_in_trial"], errors="coerce").dropna()
2741 if not order_b.empty:
2742 full_fix_range_b = (int(order_b.min()), int(order_b.max()))
2743 if fix_index_range_b is not None and fixations_b is not None:
2744 pid_b, tid_b = (str(v) for v in (trial_b or ("B", "B")))
2745 fixations_b = _apply_fix_index_range(
2746 fixations_b, fix_index_range_b, pid_b, tid_b
2747 )
2748 if dataset_b is not None:
2749 # As `compare_scanpaths` and the app do: B's readers carry their
2750 # dataset's name, so a hover says whose reader it is.
2751 from .utils import qualify_for_compare
2753 words_b, fixations_b = (
2754 None if frame is None else qualify_for_compare(frame, dataset_b)
2755 for frame in (words_b, fixations_b)
2756 )
2757 label_mode = str(illustration_label).capitalize()
2758 if label_mode not in {"Auto", "Show", "Hide"}:
2759 raise ValueError("illustration_label must be 'auto', 'show', or 'hide'.")
2760 if "illustration_reasons" not in animation_overrides:
2761 from .illustration import illustration_reasons, resolve_label_reasons
2763 reasons = illustration_reasons(
2764 {**animation_overrides, "playback_speed": playback_speed},
2765 fix_index_range=fix_index_range,
2766 full_fixation_range=full_fix_range,
2767 # CMP-24: B's own flags and window, when it co-animates.
2768 fixation_flags_b=animation_overrides.get("fixation_flags_b")
2769 if fixations_b is not None
2770 else None,
2771 fix_index_range_b=fix_index_range_b,
2772 full_fixation_range_b=full_fix_range_b,
2773 )
2774 animation_overrides["illustration_reasons"] = resolve_label_reasons(
2775 label_mode, reasons
2776 )
2777 render_settings = FigureSettings.from_mapping(
2778 animation_overrides,
2779 canvas_width=int(setup_a.canvas_width),
2780 canvas_height=int(setup_a.canvas_height),
2781 base_font_size=int(base_font_size),
2782 font_family=font_family,
2783 playback_speed=playback_speed,
2784 autoplay=autoplay,
2785 column_labels=_column_labels(
2786 names, trial_words, trial_fixations, words=word_side
2787 ),
2788 )
2789 fig = make_scanpath_animation(
2790 trial_words,
2791 trial_fixations,
2792 settings=render_settings,
2793 fixations_b=fixations_b,
2794 words_b=words_b,
2795 )
2796 add_illustration_label(
2797 fig,
2798 animation_overrides.get("illustration_reasons"),
2799 text=render_settings.illustration_text,
2800 )
2801 annotate_figure(fig, title=title, caption=caption)
2802 return fig
2805def _second_reading(
2806 words: pd.DataFrame,
2807 fixations: pd.DataFrame,
2808 words_b: pd.DataFrame | None,
2809 fixations_b: pd.DataFrame | None,
2810 trial_b: tuple[str, str] | None,
2811 *,
2812 screen_b: str | None = None,
2813) -> tuple[pd.DataFrame | None, pd.DataFrame | None]:
2814 """Scanpath B's frames for a co-animation, cut to one reading.
2816 The animation builder draws every row it is handed, so frames passed the
2817 way `compare_scanpaths` takes them — B's whole corpus — drew every fixation
2818 in it. ``trial_b`` picks the reading, in B's own frames when given and A's
2819 otherwise, as `compare_scanpaths` does; without it B's frames must hold one
2820 trial, since guessing among several would draw somebody else's reading.
2821 A multipart B keeps one screen, never all of them — each is its own
2822 coordinate space: ``screen_b``, else its first, as A without ``screen=``
2823 and the app's B navigator start.
2824 """
2825 if trial_b is None:
2826 source = fixations_b if fixations_b is not None else words_b
2827 if source is None or source.empty:
2828 return words_b, fixations_b
2829 label = "fixations_b" if fixations_b is not None else "words_b"
2830 pairs = _require_normalized(source, label)[
2831 ["participant_id", "trial_id"]
2832 ].drop_duplicates()
2833 if len(pairs) > 1:
2834 raise ValueError(
2835 f"{label} holds {len(pairs)} trials, so the second scanpath is "
2836 "ambiguous. Pass trial_b=(participant, trial) to pick one — "
2837 "compare_scanpaths takes it the same way."
2838 )
2839 pid_b, tid_b = (str(value) for value in pairs.iloc[0])
2840 else:
2841 pid_b, tid_b = str(trial_b[0]), str(trial_b[1])
2842 words_b = words if words_b is None else words_b
2843 fixations_b = fixations if fixations_b is None else fixations_b
2845 def one_reading(frame: pd.DataFrame | None, label: str) -> pd.DataFrame | None:
2846 # `extract_part` masks afresh, as `_select_trial` slices A. Not
2847 # `utils.extract_trial`: its position cache is keyed by the frame's
2848 # identity, so an in-place edit between two calls handed back somebody
2849 # else's rows.
2850 if frame is None or frame.empty:
2851 return frame
2852 return extract_part(_require_normalized(frame, label), pid_b, tid_b)
2854 trial_words_b = one_reading(words_b, "words_b")
2855 trial_fix_b = one_reading(fixations_b, "fixations_b")
2856 if trial_b is not None and (trial_fix_b is None or trial_fix_b.empty):
2857 raise ValueError(
2858 f"No fixations for the second scanpath participant={pid_b!r}, "
2859 f"trial={tid_b!r}. list_trials() shows what the frames contain."
2860 )
2861 catalog = part_catalog(trial_words_b, trial_fix_b)
2862 if catalog.empty:
2863 if screen_b is not None:
2864 raise ValueError(
2865 "screen_b= was supplied for a single-screen trial "
2866 f"(participant={pid_b!r}, trial={tid_b!r})."
2867 )
2868 else:
2869 available = catalog[SCREEN_ID].astype(str).tolist()
2870 if screen_b is not None and str(screen_b) not in available:
2871 raise ValueError(
2872 f"Unknown screen_b={str(screen_b)!r} for participant={pid_b!r}, "
2873 f"trial={tid_b!r}. "
2874 f"Available: {', '.join(repr(value) for value in available)}."
2875 )
2876 screen_b = str(screen_b) if screen_b is not None else available[0]
2877 trial_words_b, trial_fix_b = (
2878 extract_part(frame, pid_b, tid_b, screen_b)
2879 if frame is not None and SCREEN_ID in frame.columns
2880 else frame
2881 for frame in (trial_words_b, trial_fix_b)
2882 )
2883 return trial_words_b, trial_fix_b
2886def _inferred_screen_hint(*, a_inferred: bool, b_inferred: bool) -> str:
2887 """How to state a screen that a refusal only read off the data.
2889 `setups_comparable` says the readings were *recorded* on different screens,
2890 but a screen nobody stated is the extent of that trial's data, which rarely
2891 spans the whole display — so the refusal names the parameter that states it.
2892 """
2893 if a_inferred and b_inferred:
2894 return (
2895 " Neither screen was stated, so both were read off the data, which "
2896 "rarely spans the whole screen; if they were shown on one, pass it as "
2897 "setup= and setup_b=."
2898 )
2899 if a_inferred:
2900 return (
2901 " A's screen was read off its data, which rarely spans the whole "
2902 "screen; if both were shown on one, pass A's as setup= or canvas_size=."
2903 )
2904 if b_inferred:
2905 return (
2906 " B's screen was read off its data, which rarely spans the whole "
2907 "screen; if both were shown on one, pass B's as setup_b=."
2908 )
2909 return ""
2912def _same_dataset_setup_b(
2913 setup_a: SetupSnapshot,
2914 setup_b: SetupSnapshot | None,
2915 *,
2916 a_known: bool,
2917 words_a: pd.DataFrame | None,
2918 fixations_a: pd.DataFrame | None,
2919 words_b: pd.DataFrame | None,
2920 fixations_b: pd.DataFrame | None,
2921) -> SetupSnapshot | None:
2922 """B's screen when it is known to differ from A's, within one dataset.
2924 One dataset can hold screens of different sizes, so a same-dataset pair is
2925 gated too — but only on screens either side actually *knows*: a stated
2926 setup, or the selected screen's own canvas columns. Two data extents say
2927 nothing (two readings of one screen rarely span the same area). Returns
2928 B's snapshot when both are known and the canvases differ — the pair an
2929 overlay or co-animation must refuse — else ``None``. ``a_known`` is whether
2930 the caller stated A's screen.
2931 """
2932 own_b = screen_canvas_size(words_b) or screen_canvas_size(fixations_b)
2933 a_known = (
2934 a_known
2935 or screen_canvas_size(words_a) is not None
2936 or screen_canvas_size(fixations_a) is not None
2937 )
2938 if setup_b is not None:
2939 resolved = setup_b
2940 elif own_b is not None:
2941 resolved = replace(
2942 setup_a, canvas_width=int(own_b[0]), canvas_height=int(own_b[1])
2943 )
2944 else:
2945 return None
2946 if not a_known or resolved.canvas == setup_a.canvas:
2947 return None
2948 return resolved
2951def _refuse_co_animation_across_screens(
2952 setup_a: SetupSnapshot,
2953 setup_b: SetupSnapshot | None,
2954 words_b: pd.DataFrame | None,
2955 fixations_b: pd.DataFrame,
2956 *,
2957 a_inferred: bool,
2958) -> None:
2959 """Refuse a co-animation of two datasets shown on different screens.
2961 A co-animation draws both readings on one clock in A's coordinates, which
2962 makes it an overlay, so it is held to `compare_scanpaths`'s overlay gate: the
2963 same `setups_comparable` predicate and the same error, with B's screen read
2964 off its data when the caller did not state it.
2965 """
2966 from .experimental_setup import IncomparableScreensError, setups_comparable
2968 resolved_b = _compare_setup(
2969 setup_b,
2970 None,
2971 words_b if words_b is not None else pd.DataFrame(),
2972 fixations_b,
2973 side="setup_b",
2974 )
2975 comparable, note = setups_comparable(setup_a, resolved_b)
2976 if not comparable:
2977 hint = _inferred_screen_hint(a_inferred=a_inferred, b_inferred=setup_b is None)
2978 raise IncomparableScreensError(
2979 f"{note} A co-animation replays both trials on one clock in one "
2980 "coordinate space, so none was drawn; compare them with "
2981 "compare_scanpaths(layout='side_by_side') (or 'stacked'), each drawn "
2982 f"to its own screen.{hint}",
2983 reason=note,
2984 )
2985 if note:
2986 # As in `compare_scanpaths`: matching canvases, but at least one screen
2987 # was never recorded — drawn, with the caveat where a script can see it.
2988 logging.getLogger(__name__).warning("animate_scanpath: %s", note)
2991def render_parent_trial(
2992 words: pd.DataFrame,
2993 fixations: pd.DataFrame,
2994 participant: str | None = None,
2995 trial: str | None = None,
2996 *,
2997 animate: bool = False,
2998 transition_mode: str = "instant",
2999 screens: Sequence[str] | None = None,
3000 **options,
3001) -> dict[str, go.Figure]:
3002 """Render every screen of one logical trial without stitching coordinates.
3004 ``screens`` renders only those screen ids (one id may be given as a
3005 string), in the trial's own order, as the app's Export → *Screens* does;
3006 an id the trial does not have raises ``ValueError``. ``None`` renders them
3007 all. ``screen_index`` in each figure's meta stays the screen's place in the
3008 trial.
3010 The ordered mapping is keyed by ``screen_id``. Each value is the same figure
3011 returned by [`plot_scanpath`][scanpath_studio.api.plot_scanpath] or
3012 [`animate_scanpath`][scanpath_studio.api.animate_scanpath]; callers can save them
3013 into deterministic per-screen files. ``transition_mode`` is ``"instant"`` or
3014 ``"recorded"``. For animated output, each figure's
3015 ``layout.meta['transition_after_ms']`` records the delay before the next screen
3016 (zero for instant mode, or the observed parent-clock gap). No visual saccade is ever
3017 drawn across the boundary.
3018 """
3019 if transition_mode not in {"instant", "recorded"}:
3020 raise ValueError("transition_mode must be 'instant' or 'recorded'.")
3021 raw_gaze = options.get("raw_gaze")
3022 pid, tid = _resolve_trial(
3023 _cn.to_canonical_frame(words),
3024 _cn.to_canonical_frame(fixations),
3025 participant,
3026 trial,
3027 raw_gaze=_cn.to_canonical_frame(raw_gaze),
3028 )
3029 # Read here under the internal names; the renderers take the frames as
3030 # given and name their figures as they were named (DATA-66).
3031 catalog = _cn.to_canonical_frame(
3032 list_parts(words, fixations, pid, tid, raw_gaze=raw_gaze)
3033 )
3034 canonical_fixations = _cn.to_canonical_frame(fixations)
3035 if catalog.empty:
3036 if screens is not None:
3037 raise ValueError(
3038 f"Trial {tid!r} of participant {pid!r} has no screens to choose from."
3039 )
3040 renderer = animate_scanpath if animate else plot_scanpath
3041 return {"screen-1": renderer(words, fixations, pid, tid, **options)}
3043 screen_ids = catalog[SCREEN_ID].astype(str).tolist()
3044 if isinstance(screens, str):
3045 screens = [screens]
3046 if screens is not None:
3047 missing = [str(s) for s in screens if str(s) not in screen_ids]
3048 if missing:
3049 raise ValueError(
3050 f"Trial {tid!r} of participant {pid!r} has no screen "
3051 f"{', '.join(map(repr, missing))}; its screens are "
3052 f"{', '.join(screen_ids)}."
3053 )
3054 chosen = None if screens is None else {str(s) for s in screens}
3055 # The screens drawn, in the trial's order; a recorded transition runs to
3056 # the next one *drawn*, and the last drawn has none.
3057 drawn = [s for s in screen_ids if chosen is None or s in chosen]
3058 rendered: dict[str, go.Figure] = {}
3059 for position, screen_id in enumerate(drawn):
3060 renderer = animate_scanpath if animate else plot_scanpath
3061 fig = renderer(words, fixations, pid, tid, screen=screen_id, **options)
3062 delay = 0.0
3063 if animate and transition_mode == "recorded" and position < len(drawn) - 1:
3064 current = extract_part(canonical_fixations, pid, tid, screen_id)
3065 following = extract_part(canonical_fixations, pid, tid, drawn[position + 1])
3066 if not current.empty and not following.empty:
3067 current_end = (
3068 pd.to_numeric(current["timestamp_ms"], errors="coerce")
3069 + pd.to_numeric(current["duration_ms"], errors="coerce").fillna(0)
3070 ).max()
3071 next_start = pd.to_numeric(
3072 following["timestamp_ms"], errors="coerce"
3073 ).min()
3074 if pd.notna(current_end) and pd.notna(next_start):
3075 delay = max(0.0, float(next_start - current_end))
3076 existing_meta = fig.layout.meta if isinstance(fig.layout.meta, dict) else {}
3077 fig.update_layout(
3078 meta={
3079 **existing_meta,
3080 "participant_id": pid,
3081 "trial_id": tid,
3082 "screen_id": screen_id,
3083 "screen_index": screen_ids.index(screen_id) + 1,
3084 "transition_mode": transition_mode,
3085 "transition_after_ms": delay,
3086 }
3087 )
3088 rendered[screen_id] = fig
3089 return rendered
3092#: Layout names `compare_scanpaths` accepts, mapped to the builder's spelling.
3093#: Hyphens are accepted so `cli.render --compare-layout side-by-side`, the share
3094#: link's `cmp_layout`, and this function all name the layout the same way.
3095_COMPARE_LAYOUTS = {
3096 "overlay": "overlay",
3097 "side_by_side": "side_by_side",
3098 "side-by-side": "side_by_side",
3099 "stacked": "stacked",
3100}
3103def _compare_setup(
3104 setup: SetupSnapshot | None,
3105 canvas_size: tuple[int, int] | None,
3106 words: pd.DataFrame,
3107 fixations: pd.DataFrame,
3108 *,
3109 side: str,
3110) -> SetupSnapshot:
3111 """One side's `SetupSnapshot`, from an explicit one, a canvas, or the data.
3113 The provenance is the point, because the overlay gate reads it: a canvas the
3114 caller *stated* is ``MEASURED``, one inferred from the data extents is
3115 ``ESTIMATED``. Both count as knowing the screen; neither claims a physical
3116 display, which a comparison never uses.
3117 """
3118 if setup is not None:
3119 if not isinstance(setup, SetupSnapshot):
3120 raise TypeError(
3121 f"{side} must be an experimental_setup.SetupSnapshot, got "
3122 f"{type(setup).__name__}."
3123 )
3124 return setup
3125 provenance = Provenance.MEASURED
3126 if canvas_size is None:
3127 canvas_size = _recorded_screen(words, fixations)
3128 if canvas_size is None:
3129 provenance = Provenance.ESTIMATED
3130 canvas_size = screen_canvas_size(words) or screen_canvas_size(fixations)
3131 if canvas_size is None:
3132 canvas_size = _data.compute_canvas_size(words, fixations)
3133 return SetupSnapshot(
3134 canvas_width=int(canvas_size[0]),
3135 canvas_height=int(canvas_size[1]),
3136 screen_provenance=provenance,
3137 )
3140def compare_scanpaths(
3141 words: pd.DataFrame,
3142 fixations: pd.DataFrame,
3143 trial_a: tuple[str, str],
3144 trial_b: tuple[str, str],
3145 *,
3146 screen: str | None = None,
3147 screen_b: str | None = None,
3148 words_b: pd.DataFrame | None = None,
3149 fixations_b: pd.DataFrame | None = None,
3150 dataset_b: str = "Dataset B",
3151 raw_gaze: pd.DataFrame | None = None,
3152 raw_gaze_b: pd.DataFrame | None = None,
3153 layout: str = "overlay",
3154 compare_stimulus: str = "both",
3155 setup: SetupSnapshot | None = None,
3156 setup_b: SetupSnapshot | None = None,
3157 canvas_size: tuple[int, int] | None = None,
3158 labels: tuple[str, str] | None = None,
3159 style_a: dict | None = None,
3160 style_b: dict | None = None,
3161 base_font_size: int = 16,
3162 font_family: str = FONT_FAMILY,
3163 fix_index_range: tuple[int, int] | None = None,
3164 fix_index_range_b: tuple[int, int] | None = None,
3165 drift_correction: str | None = None,
3166 title: str = "",
3167 caption: str = "",
3168 column_names: dict | None = None,
3169 **figure_overrides,
3170) -> go.Figure:
3171 """Build a two-scanpath comparison figure.
3173 The headless form of the app's **Compare** mode. ``trial_a`` / ``trial_b``
3174 are ``(participant, trial)`` pairs; ``layout`` is ``"overlay"``,
3175 ``"side_by_side"`` (``"side-by-side"`` also accepted) or ``"stacked"``.
3177 **Multipart trials.** Each scanpath is one screen, never a whole multipart
3178 trial: every screen is its own coordinate space, so pooling them would draw
3179 saccades across page boundaries. ``screen`` picks A's screen and
3180 ``screen_b`` B's, independently — B's is looked up in B's own frames, so it
3181 may be a later page or another dataset's. Either one left out is that
3182 trial's first recorded screen, as in
3183 [`plot_scanpath`][scanpath_studio.api.plot_scanpath];
3184 ``list_parts()`` lists them. A screen named for a single-screen trial, or
3185 one the trial does not have, raises ``ValueError``.
3187 **Two datasets.** Pass ``words_b`` / ``fixations_b`` to draw B from a
3188 *different* dataset. Two datasets can hold the same ``(participant_id,
3189 trial_id)`` and the builder slices by exactly that pair, so B's participant
3190 ids are namespaced with ``dataset_b`` inside the throwaway merged frames —
3191 without it one trial would silently render as two. The frames you pass in
3192 are never modified, and nothing in the returned figure's data depends on the
3193 namespace beyond the trace labels.
3195 **The overlay gate.** Across datasets an overlay needs both canvases to be
3196 the same size; otherwise this raises ``ValueError`` (the app falls back to
3197 side by side). One dataset can hold screens of different sizes too, so a
3198 same-dataset pair is refused the same way when the two selected screens
3199 carry different canvases (``canvas_width`` / ``canvas_height`` columns) or
3200 ``setup_b`` states another screen. Pass ``layout="side_by_side"`` or
3201 ``"stacked"`` to compare readings from different screens; each panel is
3202 then drawn to its own. Nothing is rescaled.
3204 ``setup`` / ``setup_b`` are `experimental_setup.SetupSnapshot`
3205 values — what the gate reads. ``canvas_size`` covers A when you only have a
3206 resolution; omit both and the canvas is read off the data.
3208 **Stimulus images.** ``background_image`` is A's page. A split layout draws
3209 B's panel over ``background_image_b`` (with ``background_image_size_b`` /
3210 ``background_image_origin_b``) and over nothing without it — never A's,
3211 since sharing a dataset says nothing about sharing a page.
3213 ``compare_stimulus`` picks whose word boxes and text an **overlay** draws —
3214 ``"both"`` (default), ``"a"`` or ``"b"``. Two datasets' AOIs coincide only
3215 when the text is identical. Split layouts ignore it; each panel owns its own
3216 stimulus.
3218 **Per-scanpath style.** ``style_a`` / ``style_b`` restyle one scanpath:
3219 ``fix_color``, ``marker_size_range``, ``opacity``, ``hollow``,
3220 ``saccade_color``, ``saccade_style``, ``saccade_width``, ``box_color`` —
3221 the outline of that reading's word boxes, its ``fix_color`` when left out —
3222 ``box_fill_color``, their fill, ``word_box_fill_color`` when left out — and
3223 ``raw_gaze_color``, that reading's raw-gaze samples, its ``fix_color`` when
3224 left out. These three are this figure's only: the co-animation draws one set
3225 of boxes, in ``word_box_color`` / ``word_box_fill_color``, and no raw gaze,
3226 and ignores them. ``heatmap_colorscale`` gives that reading's word-box
3227 heatmap its own color scale (``heatmap_colorscale`` when left out) on the
3228 range both share; when A's and B's differ, each gets its own color bar.
3230 **Filters, per scanpath.** ``fixation_flags`` and
3231 ``saccade_classes`` filter both scanpaths, as they filter
3232 [`plot_scanpath`][scanpath_studio.api.plot_scanpath]'s one; the same two keys
3233 in ``style_a`` / ``style_b`` give that scanpath its own, overriding them —
3234 e.g. ``style_b={"fixation_flags": {"short": {"mode": "Discard",
3235 "threshold_ms": 80}}, "saccade_classes": ["regression"]}``. The app's
3236 Compare mode draws A under the plot controls' filters and B under its own.
3237 ``fix_index_range`` windows both scanpaths; ``fix_index_range_b`` gives B a
3238 window of its own (the app's B slider).
3240 **Raw gaze.** ``raw_gaze`` is a frame from
3241 [`load_raw_gaze`][scanpath_studio.api.load_raw_gaze]; each reading's samples
3242 are drawn under its scanpath, in that scanpath's color (``raw_gaze_marker_size``
3243 / ``raw_gaze_opacity`` style them). It serves both readings of a
3244 same-dataset comparison; across datasets it is A's, and ``raw_gaze_b`` is
3245 B's. Passing either turns the layer on; ``show_raw_gaze=False`` keeps it off.
3247 Remaining keywords are forwarded to `plots.make_comparison_figure`
3248 (e.g. ``show_words=False``, ``color_by="duration_ms"``); an unknown one
3249 raises ``TypeError`` naming the closest valid options;
3250 ``figure_options("comparison")`` lists the accepted keywords. Column names
3251 follow [`plot_scanpath`][scanpath_studio.api.plot_scanpath]'s rule: A's
3252 names (or ``column_names``) name the options and the figure's text, and
3253 either dataset's frames may come under their own names.
3254 """
3255 _check_canvas_size(canvas_size)
3256 from .experimental_setup import IncomparableScreensError, setups_comparable
3257 from .utils import (
3258 align_compare_columns,
3259 qualify_for_compare,
3260 self_compare_participant,
3261 separate_self_compare,
3262 )
3264 resolved_layout = _COMPARE_LAYOUTS.get(str(layout).strip().lower())
3265 if resolved_layout is None:
3266 raise ValueError(
3267 f"Unknown compare layout {layout!r}; choose one of "
3268 f"{', '.join(sorted(set(_COMPARE_LAYOUTS.values())))}."
3269 )
3270 _reject_unknown_options(
3271 figure_overrides,
3272 _COMPARISON_FIGURE_PARAMS | {"palette"},
3273 "compare_scanpaths",
3274 )
3276 cross_dataset = words_b is not None or fixations_b is not None
3277 # DATA-66: A's names name the figure's text and its options; every frame,
3278 # A's or B's, is processed under the internal names.
3279 carried = {
3280 table: found[1]
3281 for table, frame in (("fixations", fixations), ("words", words))
3282 if (found := _cn.frame_names(frame)) is not None
3283 }
3284 names = _call_names(column_names, **carried)
3285 word_side = _table_names(column_names, "words", carried.get("words"))
3286 figure_overrides = _canonical_options(figure_overrides, names, words=word_side)
3287 words, fixations, words_b, fixations_b = (
3288 _cn.to_canonical_frame(frame)
3289 for frame in (words, fixations, words_b, fixations_b)
3290 )
3291 words_b = words if words_b is None else words_b
3292 fixations_b = fixations if fixations_b is None else fixations_b
3293 if raw_gaze is not None:
3294 raw_gaze = _require_normalized(raw_gaze, "raw_gaze")
3295 if raw_gaze_b is not None:
3296 raw_gaze_b = _require_normalized(raw_gaze_b, "raw_gaze_b")
3297 if raw_gaze_b is None and not cross_dataset:
3298 raw_gaze_b = raw_gaze
3300 for side, pair in (("trial_a", trial_a), ("trial_b", trial_b)):
3301 if isinstance(pair, str) or len(tuple(pair)) != 2:
3302 raise ValueError(
3303 f"{side} must be a (participant, trial) pair, e.g. "
3304 f"('l37_1129', 'l37_1129_2_1_1_Ele_r0'); got {pair!r}. "
3305 "list_trials(words, fixations) lists the pairs."
3306 )
3307 # Ids written before composite ids escaped a `_` inside a part still name
3308 # their reading when that is unambiguous (`data.respell_reading`).
3309 pid_a, tid_a = _data.respell_reading(*trial_a, _data.trial_keys(fixations))
3310 pid_b, tid_b = _data.respell_reading(*trial_b, _data.trial_keys(fixations_b))
3311 # One screen per side, each resolved in its own frames — the same contract
3312 # as `plot_scanpath`'s `screen`. Extracting whole parent trials pooled every
3313 # page of a multipart reading into one scanpath, saccades across pages and all.
3314 trial_words_a, trial_fix_a, pid_a, tid_a, _screen_a = _select_part(
3315 words, fixations, str(pid_a), str(tid_a), screen
3316 )
3317 trial_words_b, trial_fix_b, pid_b, tid_b, _screen_b = _select_part(
3318 words_b,
3319 fixations_b,
3320 str(pid_b),
3321 str(tid_b),
3322 screen_b,
3323 screen_param="screen_b",
3324 )
3325 trial_raw_a = _compare_raw_gaze(raw_gaze, pid_a, tid_a, trial_fix_a)
3326 trial_raw_b = _compare_raw_gaze(raw_gaze_b, pid_b, tid_b, trial_fix_b)
3327 for frame, (pid, tid) in ((trial_fix_a, trial_a), (trial_fix_b, trial_b)):
3328 if frame.empty:
3329 raise ValueError(
3330 f"No fixations for participant={pid!r}, trial={tid!r}. "
3331 f"list_trials() shows what the frames contain."
3332 )
3333 # Either reading may carry the column (two corpora need not share them), so
3334 # it is looked for across both.
3335 _check_column_options(
3336 figure_overrides,
3337 words=pd.concat([trial_words_a, trial_words_b], ignore_index=True),
3338 fixations=pd.concat([trial_fix_a, trial_fix_b], ignore_index=True),
3339 )
3341 setup_a = _compare_setup(
3342 setup, canvas_size, trial_words_a, trial_fix_a, side="setup"
3343 )
3344 resolved_setup_b = _compare_setup(
3345 setup_b, None, trial_words_b, trial_fix_b, side="setup_b"
3346 )
3347 gate = cross_dataset
3348 if not cross_dataset:
3349 same_b = _same_dataset_setup_b(
3350 setup_a,
3351 setup_b,
3352 a_known=setup is not None or canvas_size is not None,
3353 words_a=trial_words_a,
3354 fixations_a=trial_fix_a,
3355 words_b=trial_words_b,
3356 fixations_b=trial_fix_b,
3357 )
3358 gate = same_b is not None
3359 if same_b is not None:
3360 resolved_setup_b = same_b
3361 elif setup_b is None:
3362 # Not refused: B's split panel is drawn to its own screen's canvas,
3363 # else A's known one, else (neither known) its own data's extent.
3364 own_b = screen_canvas_size(trial_words_b) or screen_canvas_size(trial_fix_b)
3365 if own_b is None and (
3366 setup is not None
3367 or canvas_size is not None
3368 or screen_canvas_size(trial_words_a) is not None
3369 or screen_canvas_size(trial_fix_a) is not None
3370 ):
3371 resolved_setup_b = setup_a
3372 elif own_b is not None:
3373 resolved_setup_b = replace(
3374 setup_a, canvas_width=int(own_b[0]), canvas_height=int(own_b[1])
3375 )
3376 if resolved_layout == "overlay" and gate:
3377 comparable, note = setups_comparable(setup_a, resolved_setup_b)
3378 if not comparable:
3379 # BUG-85: the reason says why; this says what happened here and how
3380 # to ask for the split in Python. `render` rewords it in its flags.
3381 hint = (
3382 _inferred_screen_hint(
3383 a_inferred=setup is None and canvas_size is None,
3384 b_inferred=setup_b is None,
3385 )
3386 if cross_dataset
3387 else ""
3388 )
3389 raise IncomparableScreensError(
3390 f"{note} So no overlay was drawn; pass layout='side_by_side' (or "
3391 f"'stacked') to compare them in separate panels, each drawn to "
3392 f"its own screen.{hint}",
3393 reason=note,
3394 )
3395 if note:
3396 # The canvases match but at least one corpus never recorded a screen,
3397 # so the overlay is drawn with a caveat rather than refused. A script
3398 # has no caption to read it in, so it goes to the logger — loud enough
3399 # to appear in a pipeline's output, quiet enough not to be an error.
3400 logging.getLogger(__name__).warning("compare_scanpaths: %s", note)
3402 figure_pid_b = pid_b
3403 if cross_dataset:
3404 trial_words_b = qualify_for_compare(trial_words_b, dataset_b)
3405 trial_fix_b = qualify_for_compare(trial_fix_b, dataset_b)
3406 trial_raw_b = qualify_for_compare(trial_raw_b, dataset_b)
3407 figure_pid_b = (
3408 str(trial_fix_b["participant_id"].iloc[0])
3409 if not trial_fix_b.empty
3410 else pid_b
3411 )
3412 trial_fix_a = _apply_fix_index_range(trial_fix_a, fix_index_range, pid_a, tid_a)
3413 trial_fix_b = _apply_fix_index_range(
3414 trial_fix_b,
3415 fix_index_range if fix_index_range_b is None else fix_index_range_b,
3416 pid_b,
3417 tid_b,
3418 )
3419 if drift_correction:
3420 # PRE-21: same contract as plot_scanpath — raise, don't silently skip.
3421 if not drift_correction_enabled():
3422 raise ValueError(
3423 "drift_correction is not available in this release. Set "
3424 f"{EXPERIMENTAL_ENV_VAR}=1 to enable it, or pass "
3425 "drift_correction=None."
3426 )
3427 from .alignment import correct
3429 trial_fix_a, _ = correct(trial_fix_a, trial_words_a, drift_correction)
3430 trial_fix_b, _ = correct(trial_fix_b, trial_words_b, drift_correction)
3432 if not cross_dataset and (pid_a, tid_a) == (pid_b, tid_b):
3433 # CMP-22: a trial compared with itself — rename B's copy apart, or the
3434 # figure's (participant, trial) slice hands each side both copies.
3435 # The renamed id is for slicing only, so B's default legend name is
3436 # resolved here from the real one.
3437 if not labels:
3438 labels = tuple(
3439 _resolve_trial_display_name(pid_a, tid_a, trial_words_a, None, idx)
3440 for idx in (0, 1)
3441 )
3442 trial_words_b = separate_self_compare(trial_words_b, pid_b)
3443 trial_fix_b = separate_self_compare(trial_fix_b, pid_b)
3444 trial_raw_b = separate_self_compare(trial_raw_b, pid_b)
3445 figure_pid_b = self_compare_participant(pid_b)
3446 merged_words, merged_words_b, _ = align_compare_columns(
3447 trial_words_a, trial_words_b
3448 )
3449 merged_fix, merged_fix_b, _ = align_compare_columns(trial_fix_a, trial_fix_b)
3450 merged_raw = None
3451 if not (trial_raw_a.empty and trial_raw_b.empty):
3452 merged_raw = pd.concat(align_compare_columns(trial_raw_a, trial_raw_b)[:2])
3453 settings = _figure_kwargs(figure_overrides)
3454 settings.pop("illustration_reasons", None)
3455 if raw_gaze is not None or raw_gaze_b is not None:
3456 settings.setdefault("show_raw_gaze", True)
3457 render_settings = FigureSettings.from_mapping(
3458 {k: v for k, v in settings.items() if k in _COMPARISON_FIGURE_PARAMS},
3459 canvas_width=int(setup_a.canvas_width),
3460 canvas_height=int(setup_a.canvas_height),
3461 base_font_size=int(base_font_size),
3462 font_family=font_family,
3463 layout=resolved_layout,
3464 compare_stimulus=normalize_option_value("compare_stimulus", compare_stimulus),
3465 trial_labels=tuple(labels) if labels else None,
3466 style_a=style_a,
3467 style_b=style_b,
3468 column_labels=_column_labels(
3469 names, trial_words_a, trial_fix_a, words=word_side
3470 ),
3471 # Only the split layouts read this; an overlay that got here has two
3472 # equal canvases anyway, so it is the same value either way.
3473 canvas_b=resolved_setup_b.canvas,
3474 )
3475 fig = make_comparison_figure(
3476 pd.concat([merged_words, merged_words_b], ignore_index=True),
3477 pd.concat([merged_fix, merged_fix_b], ignore_index=True),
3478 (pid_a, tid_a),
3479 (figure_pid_b, tid_b),
3480 settings=render_settings,
3481 raw_gaze=merged_raw,
3482 )
3483 annotate_figure(fig, title=title, caption=caption)
3484 return fig
3487def _compare_raw_gaze(
3488 raw_gaze: pd.DataFrame | None, pid: str, tid: str, trial_fix: pd.DataFrame
3489) -> pd.DataFrame:
3490 """One comparison reading's samples — its trial's, and its screen's when the
3491 reading is one screen of a multipart trial."""
3492 if raw_gaze is None or raw_gaze.empty:
3493 return pd.DataFrame()
3494 samples = _data.filter_raw_gaze(raw_gaze, [pid], [tid])
3495 if SCREEN_ID in samples.columns and SCREEN_ID in trial_fix.columns:
3496 screens = trial_fix[SCREEN_ID].dropna().unique()
3497 if len(screens) == 1:
3498 samples = extract_part(samples, pid, tid, screens[0])
3499 return samples
3502def save_figure(
3503 fig: go.Figure,
3504 path: str | Path,
3505 *,
3506 scale: float = 2,
3507 width: int | None = None,
3508 height: int | None = None,
3509 width_mm: float | None = None,
3510 width_in: float | None = None,
3511 dpi: int | None = None,
3512) -> Path:
3513 """Save a figure by extension: ``.html`` (interactive, needs no browser) or
3514 ``.png``/``.svg``/``.pdf`` (static via Kaleido — needs Chrome, Chromium or
3515 Edge; run ``plotly_get_chrome -y`` once if none is installed). ``width`` /
3516 ``height`` set the image size in px (overriding the figure's own size);
3517 both ignored for ``.html``. Returns the written path.
3519 ``width_mm`` or ``width_in`` with ``dpi`` (default 300) sizes a PNG for
3520 print, as the app's Export → *Current figure* does: 180 mm at 600 dpi is
3521 a 4,252 px wide PNG, its height following the figure's aspect, with the
3522 dpi written into the file. They replace ``scale``."""
3523 path = Path(path)
3524 suffix = path.suffix.lower()
3525 if suffix not in (".html", ".png", ".svg", ".pdf"):
3526 raise ValueError(
3527 f"save_figure writes .html, .png, .svg or .pdf, not {suffix or path.name!r}. "
3528 "For a GIF or MP4 replay, use "
3529 "scanpath_studio.animation_export.export_animation."
3530 )
3531 if not path.parent.is_dir():
3532 raise FileNotFoundError(
3533 f"Can't write {path}: the folder {path.parent} does not exist."
3534 )
3535 if width_mm is not None or width_in is not None:
3536 if width_mm is not None and width_in is not None:
3537 raise ValueError("Pass width_mm or width_in, not both.")
3538 if suffix != ".png":
3539 raise ValueError(
3540 "width_mm / width_in / dpi size a PNG; save as .png, or set "
3541 "width / height / scale for other formats."
3542 )
3543 dpi = int(dpi or _export.DEFAULT_PRINT_DPI)
3544 unit, value = ("mm", width_mm) if width_mm is not None else ("in", width_in)
3545 base = int(width or fig.layout.width or 700)
3546 scale = _export.print_scale(base, float(value), unit, dpi)
3547 elif dpi is not None:
3548 raise ValueError(
3549 "dpi is the resolution of a print width: pass width_mm or width_in too."
3550 )
3551 if suffix == ".html":
3552 # BUG-93: an animation replays on the wall-clock player, which also
3553 # autoplays it at the configured speed when asked (VIZ-10). Plotly's own
3554 # `auto_play` stays off — it ignores the frame duration. PERF-17: the
3555 # frames are written packed and rebuilt by the page's own script, so a
3556 # long replay writes a fraction of the bytes. Static figures write
3557 # unchanged.
3558 page = replay_page(fig)
3559 if page is not None:
3560 figure_dict, script = page
3561 pio.write_html(
3562 figure_dict,
3563 str(path),
3564 validate=False,
3565 auto_play=False,
3566 post_script=script,
3567 config={**PLOTLY_CONFIG},
3568 )
3569 elif fig.frames:
3570 fig.write_html(str(path), auto_play=False, config={**PLOTLY_CONFIG})
3571 else:
3572 fig.write_html(str(path), config={**PLOTLY_CONFIG})
3573 return path
3574 if suffix in (".png", ".svg", ".pdf"):
3575 try:
3576 fig.write_image(str(path), scale=scale, width=width, height=height)
3577 if dpi is not None:
3578 _export.set_png_dpi(path, dpi)
3579 except OSError:
3580 raise # filesystem problem — the original error says it best
3581 except Exception as exc: # Kaleido raises various types
3582 raise RuntimeError(
3583 f"Static {suffix} export failed ({exc}). Kaleido needs Chrome, "
3584 "Chromium or Edge — install one, or run `plotly_get_chrome -y` "
3585 "once — or save as .html, which needs no browser."
3586 ) from exc
3587 return path
3590def save_figure_layers(
3591 fig: go.Figure,
3592 directory: str | Path,
3593 *,
3594 fmt: str = "svg",
3595 scale: int = 2,
3596 width: int | None = None,
3597 height: int | None = None,
3598) -> dict:
3599 """Split a scanpath figure into its layers and save one file per layer.
3601 Writes ``<directory>/<layer>.<fmt>`` for each *visible* layer (word boxes /
3602 fixations / saccades / heatmap / labels / stimulus image / frame) and returns
3603 ``{layer: Path}``. Each layer is the full figure with only that layer's elements
3604 and a transparent background, at the same size and axis ranges — so the files
3605 register perfectly when stacked in Illustrator / Inkscape. ``fmt`` is any
3606 [`save_figure`][scanpath_studio.api.save_figure] extension without the dot
3607 (``svg`` / ``pdf`` are vector and best for editing; ``png`` / ``html`` also
3608 work). ``scale`` / ``width`` / ``height`` are forwarded to
3609 [`save_figure`][scanpath_studio.api.save_figure]."""
3610 directory = Path(directory)
3611 # ENG-54: a failed render (most often Kaleido with no Chrome) used to leave
3612 # an empty `<output>_layers/` behind, which reads as "exported, but lost".
3613 # Whatever this call created is removed again if nothing was written to it.
3614 created = [path for path in (directory, *directory.parents) if not path.exists()]
3615 directory.mkdir(parents=True, exist_ok=True)
3616 written: dict = {}
3617 try:
3618 for layer, layer_fig in split_scanpath_layers(fig).items():
3619 path = directory / f"{layer}.{fmt.lstrip('.')}"
3620 written[layer] = save_figure(
3621 layer_fig, path, scale=scale, width=width, height=height
3622 )
3623 except Exception:
3624 for path in created: # deepest first
3625 if path.is_dir() and not any(path.iterdir()):
3626 path.rmdir()
3627 raise
3628 return written
3631def figure_code(
3632 *,
3633 kind: str = "static",
3634 source: str = "demo",
3635 source_options: dict | None = None,
3636 participant: str = "",
3637 trial: str = "",
3638 screen: str | None = None,
3639 compare: tuple[str, str] | None = None,
3640 compare_screen: str | None = None,
3641 compare_layout: str = "overlay",
3642 compare_stimulus: str = "both",
3643 compare_dataset: str = "",
3644 compare_canvas: tuple[int, int] | None = None,
3645 compare_labels: tuple[str, str] | None = None,
3646 canvas_size: tuple[int, int] | None = None,
3647 base_font_size: int = 16,
3648 font_family: str = FONT_FAMILY,
3649 title: str = "",
3650 caption: str = "",
3651 fix_index_range: tuple[int, int] | None = None,
3652 illustration_label: str = "auto",
3653 drift_correction: str | None = None,
3654 drift_connectors: bool = False,
3655 playback_speed: float = 1.0,
3656 autoplay: bool = True,
3657 flavor: str = "python",
3658 explicit: bool = False,
3659 output: str | None = None,
3660 **figure_overrides,
3661) -> str:
3662 """The API or CLI code that reproduces a figure.
3664 The headless twin of the app's 🔗 Share → *Reproduce this figure in code* block: give
3665 it the same arguments you would give
3666 [`plot_scanpath`][scanpath_studio.api.plot_scanpath] (``kind="static"``),
3667 [`animate_scanpath`][scanpath_studio.api.animate_scanpath] (``"animation"``) or
3668 [`compare_scanpaths`][scanpath_studio.api.compare_scanpaths] (``"comparison"``) and
3669 it returns the snippet that rebuilds that figure, rather than the figure::
3671 print(sps.figure_code(participant="l7_1090", trial="l7_1090_2_1_1_Ele_r0",
3672 show_heatmap=True, flavor="cli"))
3674 ``source`` names how the data is loaded — ``"demo"``, ``"synthetic"``, ``"files"``,
3675 ``"potec"``, ``"onestop"``, ``"author"``, or
3676 ``"unknown"`` for data a snippet can't name — with ``source_options`` carrying that
3677 loader's arguments (``{"root": …}``, ``{"words": [...], "fixations": [...]}``, and
3678 so on). With ``show_raw_gaze=True`` the raw-gaze table is read too: the demo's own,
3679 or the path(s) given as ``source_options["raw_gaze"]`` (plus an optional
3680 ``"raw_gaze_schema"``) — [`load_raw_gaze`][scanpath_studio.api.load_raw_gaze] in the
3681 Python form, ``--raw-gaze`` in the CLI one. ``source="raw_gaze"`` is a dataset
3682 recorded as raw gaze alone: the samples at ``source_options["raw_gaze"]`` are the
3683 data, and ``plot_scanpath`` is handed ``None`` for the words and fixations.
3685 ``screen`` / ``compare_screen`` are A's and B's screens of a multipart trial
3686 (``screen=`` / ``screen_b=``, ``--screen`` / ``--compare-screen``).
3688 ``compare_dataset`` names the dataset scanpath B was loaded from when it is a
3689 *second* one. B's participant id belongs to that dataset rather than
3690 the one the snippet loads, so both forms then load B's own tables and name
3691 B in them — ``words_b=`` / ``fixations_b=`` / ``dataset_b=``, and
3692 ``--compare-words`` / ``--compare-fixations`` beside ``--compare-with`` —
3693 from the placeholder paths ``B_WORDS`` / ``B_FIXATIONS``, which you point
3694 at its files. ``compare_canvas`` is B's screen, ``(width, height)``, when
3695 you know it: written as ``setup_b=`` and ``--compare-canvas``, which a
3696 co-animation across datasets needs.
3698 ``compare_labels`` is the pair you would pass
3699 [`compare_scanpaths`][scanpath_studio.api.compare_scanpaths] as ``labels=`` — the
3700 two trace labels, when they are not the composed defaults. Both forms
3701 carry them: ``labels=`` in the Python snippet, ``--label-a`` / ``--label-b`` in the
3702 CLI one.
3704 With ``participant`` / ``trial`` left empty the snippet renders the first
3705 available trial, as ``render`` does. ``canvas_size`` defaults to the screen
3706 ``render`` assumes for the source (the demo's 2560×1440, PoTeC's 1680×1050,
3707 …), so both flavors draw the same figure; ``output`` defaults to
3708 ``scanpath.html`` for an animation — ``render --animate`` writes only HTML —
3709 and to a PNG otherwise.
3711 Only the options that differ from
3712 [`figure_options`][scanpath_studio.api.figure_options] are written, so the snippet
3713 stays readable; ``explicit=True`` emits every option at its current value.
3714 ``flavor`` is ``"python"``, ``"cli"``, or ``"both"`` (the two separated by a blank
3715 line). Anything neither form can reproduce — a raw-gaze table with no path to
3716 name, an uploaded stimulus image, B's rows from a second corpus — follows as
3717 ``# Note:`` comments, matching the ⚠️ captions the app shows and the ``Note:`` lines
3718 `render --print-code` writes to stderr. See `code_snippet.ReproductionCode` for the
3719 structured form.
3720 """
3721 from . import code_snippet as _snippet
3723 if flavor not in ("python", "cli", "both"):
3724 raise ValueError(f"flavor must be 'python', 'cli' or 'both', got {flavor!r}.")
3725 _reject_unknown_options(
3726 figure_overrides,
3727 set(figure_options(kind)) | {"palette"},
3728 "figure_code",
3729 )
3730 if canvas_size is None:
3731 # EXP-14: `render` snaps these sources to their recorded screen while
3732 # `plot_scanpath` estimates one from the data, so leaving the canvas
3733 # unnamed made the two flavours of one recipe disagree.
3734 canvas_size = _snippet.source_canvas(source)
3735 state = _snippet.FigureState(
3736 kind=kind,
3737 settings={**figure_options(kind), **_expand_palette(figure_overrides)},
3738 participant=participant,
3739 trial=trial,
3740 screen=screen,
3741 canvas=canvas_size,
3742 base_font_size=base_font_size,
3743 font_family=font_family,
3744 title=title,
3745 caption=caption,
3746 fix_index_range=fix_index_range,
3747 illustration_label=illustration_label,
3748 drift_correction=drift_correction,
3749 drift_connectors=drift_connectors,
3750 playback_speed=playback_speed,
3751 autoplay=autoplay,
3752 compare=(
3753 _snippet.CompareTarget(
3754 participant=str(compare[0]),
3755 trial=str(compare[1]),
3756 screen=None if compare_screen is None else str(compare_screen),
3757 layout=compare_layout,
3758 compare_stimulus=compare_stimulus,
3759 dataset=str(compare_dataset),
3760 canvas=(
3761 (int(compare_canvas[0]), int(compare_canvas[1]))
3762 if compare_canvas and compare_dataset
3763 else None
3764 ),
3765 labels=(
3766 (str(compare_labels[0]), str(compare_labels[1]))
3767 if compare_labels
3768 else None
3769 ),
3770 )
3771 if compare is not None
3772 else None
3773 ),
3774 )
3775 code = _snippet.reproduction_code(
3776 _snippet.SnippetSource(
3777 kind=source, label=source, options=dict(source_options or {})
3778 ),
3779 state,
3780 explicit=explicit,
3781 output=output or _snippet.DEFAULT_OUTPUT.get(kind, "scanpath.png"),
3782 )
3783 cli = code.cli
3784 if code.cli_unsupported:
3785 cli += "\n# No `render` flag for: " + ", ".join(code.cli_unsupported)
3786 # The caveats apply to *both* snippets, so on "both" they are appended once
3787 # at the end rather than to each half — two identical blocks would read as
3788 # two different warnings.
3789 notes = "".join(f"\n# Note: {note}" for note in code.caveats)
3790 if flavor == "python":
3791 return code.python + notes
3792 if flavor == "cli":
3793 return cli + notes
3794 return f"{code.python}\n\n{cli}{notes}"
3797def cache_status() -> dict:
3798 """Describe the on-device recovery cache a local app run keeps.
3800 The app stores completed uploaded datasets, column mappings, view settings,
3801 saved designs, metadata tables and annotations under the user's cache directory so a refresh or restart resumes
3802 where it left off — on localhost/desktop only, never on a hosted deployment. This
3803 reports that store without launching the app: ``enabled``, ``directory``,
3804 ``datasets`` (name + per-frame row counts), ``rows``, ``annotations``,
3805 ``designs``, ``metadata``, ``settings``, ``bytes``, ``saved_at``, plus ``exists`` / ``readable`` for a
3806 missing or unreadable manifest, ``damaged`` (name + reason) for a stored
3807 dataset whose entry or files are broken — the app restores the others and
3808 keeps that one in the cache rather than dropping it — and
3809 ``damaged_metadata``, the reason the stored metadata tables would not
3810 restore (``""`` when they would). Delete it with
3811 [`clear_cache`][scanpath_studio.api.clear_cache]; the same information is in the
3812 app's 🗂️ Data Management → *Saved on this computer* section and in
3813 ``scanpath-studio cache``."""
3814 from .persistence import cache_status as _cache_status
3816 return _cache_status(url="http://localhost")
3819def clear_cache() -> dict:
3820 """Delete the on-device recovery cache and return its status afterwards.
3822 Removes only the files this app wrote (``manifest.json`` and the dataset
3823 Parquet files); anything else in the folder is left alone. A *running* local
3824 app writes its session back out at the end of its next change — start it
3825 with ``scanpath-studio run --no-persist`` or ``SCANPATH_STUDIO_PERSIST=0`` to
3826 stop that."""
3827 from .persistence import clear_local_state
3829 clear_local_state()
3830 return cache_status()
3833def version_info() -> BuildInfo:
3834 """Which build of Scanpath Studio this is — no network access.
3836 ``version`` is what ``scanpath_studio.__version__`` holds: the release itself
3837 (``"0.35.0"``), or between releases a PEP 440 version that sorts after it —
3838 ``"0.35.0.post3+g8f18219"`` is three commits after v0.35.0, at commit
3839 ``8f18219``, and it ends ``.dirty`` with uncommitted changes. ``release`` is
3840 the release it descends from (``scanpath_studio.__release__``), ``distance``
3841 the commits since (``None`` when unknown), ``commit``, ``dirty``, and
3842 ``source`` — how it was worked out: ``"checkout"`` (``git describe``),
3843 ``"stamp"`` (a desktop bundle's build stamp), ``"vcs"`` (a
3844 ``pip install git+…``) or ``"release"``. ``describe()`` says it in a
3845 sentence. The same is in Help → About and ``scanpath-studio version``."""
3846 from .build_info import build_info
3848 return build_info()
3851def check_for_updates(timeout: float = 5.0) -> UpdateCheck:
3852 """Ask GitHub whether a newer release than this build is out.
3854 The one call here that uses the network, and only when made: it reads the
3855 latest release from ``api.github.com`` (drafts and pre-releases excluded)
3856 and compares it with [`version_info`][scanpath_studio.api.version_info]. It
3857 never raises. ``status`` is ``"up_to_date"``, ``"update_available"``,
3858 ``"ahead"`` (a development build past the latest release) or ``"error"``
3859 (offline, no answer within ``timeout`` seconds, rate-limited, …), and
3860 ``message`` says it in a sentence. With an update available, ``command`` is
3861 the shell command that updates this install (``pip install -U
3862 scanpath-studio``, ``uv tool upgrade scanpath-studio``, ``git pull``, …),
3863 ``latest.url`` the release notes, and in the desktop app ``download`` the
3864 archive for this computer. The same check is Help → About → *Check for
3865 updates* and ``scanpath-studio version --check``."""
3866 from .updates import check_for_updates as _check
3868 return _check(timeout)