Coverage for scanpath_studio/api.py: 93%

1106 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""Headless programmatic API for scanpath-studio. 

2 

3The Streamlit app and this module share one pipeline (``data`` → ``measures`` 

4→ ``plots``), so a figure produced here goes through the exact same builders as 

5the app and is pixel-identical *given the same settings*. The headless defaults 

6(``CANONICAL_FIGURE_DEFAULTS``) are the app's default *Scanpath* design — 

7fixations, saccades and the text — so a bare call draws the figure the app 

8opens on. 

9Typical use:: 

10 

11 import scanpath_studio as sps 

12 

13 words, fixations = sps.load_scanpath_data("ia.csv", "fixations.csv") 

14 print(sps.list_trials(words, fixations)) 

15 fig = sps.plot_scanpath(words, fixations, participant="p1", trial="t1") 

16 sps.save_figure(fig, "scanpath.html") # or .png/.svg/.pdf (needs Chrome) 

17 

18Every keyword accepted by :func:`plots.make_scanpath_figure` / 

19:func:`plots.make_scanpath_animation` can be overridden through 

20``plot_scanpath`` / ``animate_scanpath`` (e.g. ``show_heatmap=True``); 

21:func:`figure_options` lists them with their effective defaults. ``docs/agents.md`` 

22is the task-oriented guide to this module for scripted / agent use. 

23""" 

24 

25from __future__ import annotations 

26 

27import difflib 

28import logging 

29from collections.abc import Iterable, Sequence 

30from copy import deepcopy 

31from dataclasses import replace 

32from pathlib import Path 

33 

34import pandas as pd 

35import plotly.graph_objects as go 

36import plotly.io as pio 

37 

38# Outside a Streamlit runtime the @st.cache_data decorators in `data` fall 

39# back to bare-mode caching and log a "No runtime found" warning per cached 

40# function — harmless but noisy for library/CLI users, so quiet those loggers. 

41# Order matters twice over: streamlit must be imported first (its get_logger() 

42# sets each module logger's level at import, clobbering anything set earlier), 

43# and `.data` must be imported after (its decorators fire the warnings at 

44# import time). Inside the app a runtime exists and these warnings never fire. 

45import streamlit as _st # noqa: F401 (imported for its logging side effect) 

46 

47for _name in ( 

48 "streamlit.runtime.caching.cache_data_api", 

49 "streamlit.runtime.scriptrunner_utils.script_run_context", 

50): 

51 logging.getLogger(_name).setLevel(logging.ERROR) 

52 

53from . import column_names as _cn # noqa: E402 

54from . import data as _data # noqa: E402 

55from . import export as _export # noqa: E402 

56from .build_info import BuildInfo # noqa: E402 

57from .column_names import ColumnNames # noqa: E402 

58from .constants import ( # noqa: E402 

59 DEFAULT_BACKGROUND_COLOR, 

60 DEFAULT_ORDER_FONT_COLOR, 

61 EXPERIMENTAL_ENV_VAR, 

62 FONT_FAMILY, 

63 PLOTLY_CONFIG, 

64 SACCADE_CLASS_ORDER, 

65 UNIFORM_COLOR_FIELD, 

66 computed_measures_enabled, 

67 drift_correction_enabled, 

68 palette_settings, 

69) 

70from .experimental_setup import Provenance, SetupSnapshot # noqa: E402 

71from .export import annotate_figure # noqa: E402 

72from .multipart import ( # noqa: E402 

73 SCREEN_ID, 

74 apply_trial_parts_manifest, 

75 extract_part, 

76 part_catalog, 

77 screen_canvas_size, 

78) 

79from .plots import ( # noqa: E402 

80 ANIMATION_FIGURE_OPTIONS, 

81 COMPARISON_FIGURE_OPTIONS, 

82 FIGURE_OPTION_CHOICES, 

83 STATIC_FIGURE_OPTIONS, 

84 FigureSettings, 

85 _resolve_trial_display_name, 

86 add_illustration_label, 

87 make_comparison_figure, 

88 make_difference_profile_figure, 

89 make_distribution_figure, 

90 make_scanpath_animation, 

91 make_scanpath_figure, 

92 make_word_profile_figure, 

93 normalize_option_value, 

94 normalize_option_values, 

95 normalize_palette, 

96 replay_page, 

97 split_scanpath_layers, 

98) 

99from .updates import UpdateCheck # noqa: E402 

100 

101 

102def build_authored_scanpath( 

103 text: str, events: pd.DataFrame | None = None, **layout_options 

104) -> tuple[pd.DataFrame, pd.DataFrame]: 

105 """Build normalized word/fixation frames from hand-authored reading events. 

106 

107 When ``events`` is omitted, one centered fixation per laid-out word is used. 

108 ``layout_options`` are forwarded to `authoring.layout_text`. 

109 """ 

110 from .authoring import authored_fixations, default_events, layout_text 

111 

112 words = layout_text(text, **layout_options) 

113 if events is None: 

114 events = default_events(words) 

115 return words, authored_fixations(words, events) 

116 

117 

118def load_authored_scanpath( 

119 source: str | Path, 

120) -> tuple[pd.DataFrame, pd.DataFrame]: 

121 """Load an authoring file — the JSON the app's **Download authoring file** 

122 saves — or its text, as normalized word/fixation frames.""" 

123 from .authoring import parse_authoring_document 

124 

125 raw = str(source) 

126 if raw.lstrip().startswith("{"): 

127 payload = raw 

128 else: 

129 payload = Path(source).read_text(encoding="utf-8") 

130 document = parse_authoring_document(payload) 

131 return build_authored_scanpath( 

132 document.text, 

133 document.events, 

134 **document.layout, 

135 ) 

136 

137 

138TableLike = pd.DataFrame | str | Path 

139TablesLike = TableLike | list["TableLike"] 

140 

141# The headless rendering is the app's default *Scanpath* design (#374, F21): 

142# fixations, saccades and the text, with word boxes, the heatmap and fixation 

143# numbers off (controls._VIZ_WIDGET_DEFAULTS), so a bare `plot_scanpath` / 

144# `render` draws the figure the app opens on. Turn a layer on with its keyword 

145# (`show_heatmap=True`). `heatmap_metric="counts"` is translated to the 

146# figure-level `None` in _figure_kwargs, like tabs._build_figure_settings. 

147# 

148# Every option tracks the app's own default (controls._VIZ_WIDGET_DEFAULTS → 

149# controls._collect_viz_settings → tabs._build_figure_settings), so the same 

150# call renders the same picture headless as on screen. `figure_options()` 

151# prints the merged result. 

152_FIGURE_CONTEXT_FIELDS = frozenset( 

153 {"canvas_width", "canvas_height", "base_font_size", "font_family"} 

154) 

155_STATIC_FIGURE_PARAMS = frozenset(STATIC_FIGURE_OPTIONS) - _FIGURE_CONTEXT_FIELDS 

156#: CMP-9. `layout`, `compare_stimulus`, `trial_labels` and `canvas_b` are named 

157#: parameters of `compare_scanpaths`, so they are not also loose keywords. 

158_COMPARISON_FIGURE_PARAMS = frozenset(COMPARISON_FIGURE_OPTIONS) - ( 

159 _FIGURE_CONTEXT_FIELDS | {"layout", "compare_stimulus", "trial_labels", "canvas_b"} 

160) 

161_ANIMATION_FIGURE_PARAMS = ( 

162 frozenset(ANIMATION_FIGURE_OPTIONS) 

163 - _FIGURE_CONTEXT_FIELDS 

164 - {"playback_speed", "autoplay"} 

165) | {"fixations_b", "words_b"} 

166 

167_CANONICAL_OPTION_NAMES = { 

168 "show_words", 

169 "illustration_text", 

170 "word_box_color", 

171 "word_box_line_opacity", 

172 "word_box_fill_color", 

173 "word_box_fill_opacity", 

174 "show_word_labels", 

175 "show_fixations", 

176 "show_order", 

177 "show_saccades", 

178 "show_saccade_arrows", 

179 "show_heatmap", 

180 "heatmap_style", 

181 "heatmap_norm", 

182 "x_field", 

183 "y_field", 

184 "color_by", 

185 "heatmap_metric", 

186 "marker_size_range", 

187 "marker_size_scale", 

188 "marker_duration_range", 

189 "duration_size_legend", 

190 "legend_layout", 

191 "order_font_size", 

192 "order_font_color", 

193 "show_fixation_colorbar", 

194 "fixation_colorbar_orientation", 

195 "fixation_colorbar_tickangle", 

196 "fixation_colorbar_tickfont_size", 

197 "show_heatmap_colorbar", 

198 "heatmap_colorbar_orientation", 

199 "heatmap_colorbar_tickangle", 

200 "heatmap_colorbar_tickfont_size", 

201 "fixation_color_range", 

202 "heatmap_range", 

203 "fixation_colorscale", 

204 "heatmap_colorscale", 

205 "critical_span_style", 

206 "highlight_column", 

207 "saccade_color", 

208 "saccade_style", 

209 "saccade_width", 

210 "saccade_color_mode", 

211 "saccade_class_colors", 

212 "saccade_type_legend", 

213 "saccade_classes", 

214 "saccade_render_mode", 

215 "fixation_snap_to_word", 

216 "fixation_color", 

217 "fixation_symbol", 

218 "fixation_opacity", 

219 "background_color", 

220 "color_by_line", 

221 "fit_to_monitor", 

222 "show_coordinate_grid", 

223 "coordinate_grid_spacing", 

224 "line_spacing", 

225 "scale_text_to_boxes", 

226 "background_image", 

227 "background_image_size", 

228 "background_image_origin", 

229 "background_image_opacity", 

230 "word_hover_fields", 

231 "fixation_hover_fields", 

232} 

233 

234CANONICAL_FIGURE_DEFAULTS: dict = FigureSettings.defaults( 

235 _CANONICAL_OPTION_NAMES 

236) | dict( 

237 show_words=False, 

238 show_order=False, 

239 show_heatmap=False, 

240 heatmap_metric="duration_ms", 

241 order_font_color=DEFAULT_ORDER_FONT_COLOR, 

242 saccade_classes=list(SACCADE_CLASS_ORDER), 

243 fixation_opacity=0.7, 

244 background_color=DEFAULT_BACKGROUND_COLOR, 

245 fit_to_monitor=True, 

246 # #374 F26: the A/B legend is on, as in the app. 

247 show_legend=True, 

248 word_hover_fields=["text", "word_id", "line_idx", "total_fixation_duration_ms"], 

249 fixation_hover_fields=["order_in_trial", "duration_ms", "word_id"], 

250) 

251 

252 

253def _as_dataframe( 

254 table: TablesLike, label: str, *, plan_for=None, kind: str | None = None 

255) -> pd.DataFrame: 

256 if isinstance(table, pd.DataFrame): 

257 # DATA-66: a frame this API returned under the dataset's own names goes 

258 # back to the internal names it was normalized under, so loading it 

259 # again is the round-trip it was before (a converted width sits beside 

260 # a renamed left edge, which no detection would pair up). 

261 return _cn.to_canonical_frame(table) 

262 items = _data.expand_table_inputs(table) 

263 for item in items: 

264 if not isinstance(item, pd.DataFrame) and not Path(item).is_file(): 

265 raise FileNotFoundError( 

266 f"{label} table not found: {item} (looked in {Path.cwd()})" 

267 ) 

268 # #374 F3: a zip holding both EyeLink reports gives each table its own. 

269 return _data.read_tables(items, plan_for=plan_for, kind=kind) 

270 

271 

272def _metadata_id_plan(id_column, infer, *extra): 

273 """``plan_for`` for a metadata table: read its id column(s) as text. 

274 

275 The same protection the data tables get — read as numbers, readers ``1`` 

276 and ``01`` become one reader before the metadata ever sees them. The id 

277 column is the caller's, else the one ``infer`` would pick from the header. 

278 """ 

279 

280 def plan(header) -> _data.ReadPlan: 

281 names = [str(name) for name in header] 

282 # `infer_*` treats a row-less frame as "no table" — give it one row. 

283 resolved = id_column or infer( 

284 pd.DataFrame([[None] * len(names)], columns=names) 

285 ) 

286 columns = [*_data.trial_mapping_columns(resolved or []), *extra] 

287 return _data.ReadPlan( 

288 identity=tuple(c for c in dict.fromkeys(columns) if c and c in names) 

289 ) 

290 

291 return plan 

292 

293 

294# --------------------------------------------------------------------------- 

295# Schema diagnostics 

296# 

297# `data.validate_*_schema` says *what* is missing ("missing Trial ID"). A caller 

298# scripting against an unfamiliar table also needs *why*: which column names 

299# auto-detection looked for, which columns the table actually has, and the exact 

300# override to pass. These tables mirror `data.propose_*_schema` field for field — 

301# add a field there, add it here. 

302# --------------------------------------------------------------------------- 

303 

304_SCHEMA_SPECS: dict = { 

305 "words": { 

306 "title": "Words", 

307 "noun": "words", 

308 "param": "word_schema", 

309 # (schema key, human label, candidate column names) — the fields whose 

310 # absence makes `validate_word_schema` fail. 

311 "required": ( 

312 ("trial", "Trial ID", _data.TRIAL_CANDIDATES), 

313 ("word_id", "Word ID", _data.WORD_ID_CANDIDATES), 

314 ), 

315 # …plus one "either group A or group B" requirement. 

316 "group_label": "Word box", 

317 "groups": (("x", "y", "width", "height"), ("left", "right", "top", "bottom")), 

318 "group_candidates": { 

319 "x": _data.WORD_X_CANDIDATES, 

320 "y": _data.WORD_Y_CANDIDATES, 

321 "width": _data.WORD_WIDTH_CANDIDATES, 

322 "height": _data.WORD_HEIGHT_CANDIDATES, 

323 "left": _data.WORD_LEFT_CANDIDATES, 

324 "right": _data.WORD_RIGHT_CANDIDATES, 

325 "top": _data.WORD_TOP_CANDIDATES, 

326 "bottom": _data.WORD_BOTTOM_CANDIDATES, 

327 }, 

328 "propose": _data.propose_word_schema, 

329 }, 

330 "fixations": { 

331 "title": "Fixations", 

332 "noun": "fixations", 

333 "param": "fix_schema", 

334 "required": ( 

335 ("trial", "Trial ID", _data.TRIAL_CANDIDATES), 

336 ("duration", "Duration", _data.FIX_DURATION_CANDIDATES), 

337 ), 

338 "group_label": "Fixation location", 

339 "groups": (("x", "y"), ("word_id",)), 

340 "group_candidates": { 

341 "x": _data.FIX_X_CANDIDATES, 

342 "y": _data.FIX_Y_CANDIDATES, 

343 "word_id": _data.FIX_WORD_ID_CANDIDATES, 

344 }, 

345 "propose": _data.propose_fix_schema, 

346 }, 

347 "raw_gaze": { 

348 "title": "Raw gaze", 

349 "noun": "raw gaze", 

350 "param": "raw_gaze_schema", 

351 "required": ( 

352 ("trial", "Trial ID", _data.TRIAL_CANDIDATES), 

353 ("x", "X", _data.RAW_GAZE_X_CANDIDATES), 

354 ("y", "Y", _data.RAW_GAZE_Y_CANDIDATES), 

355 ), 

356 "group_label": None, 

357 "groups": (), 

358 "group_candidates": {}, 

359 "propose": _data.propose_raw_gaze_schema, 

360 }, 

361} 

362 

363 

364def _column_preview(frame: pd.DataFrame, limit: int = 40) -> str: 

365 """Comma-separated column names, truncated so a 100-column IA report stays 

366 readable in a traceback.""" 

367 cols = [str(c) for c in frame.columns] 

368 shown = ", ".join(cols[:limit]) 

369 if len(cols) > limit: 

370 shown += f", … (+{len(cols) - limit} more)" 

371 return shown 

372 

373 

374class SchemaError(ValueError): 

375 """A table whose columns don't resolve onto the canonical fields. 

376 

377 Still a ``ValueError`` with the same message, so ``except ValueError`` 

378 callers are unaffected. The parts are kept apart for the CLI, whose 

379 users cannot pass ``word_schema=``: it keeps :attr:`detail` and replaces 

380 :attr:`hint` — the API-vocabulary "pass ``word_schema={…}``" line — with its 

381 own ``--word-schema`` one, built from :attr:`mapping` (the mapping skeleton, 

382 ``None`` when the fix is to correct a mapping rather than write one).""" 

383 

384 def __init__( 

385 self, lines: list[str], hint: str, *, param: str, mapping: dict | None = None 

386 ) -> None: 

387 self.detail = "\n".join(lines) 

388 self.hint = hint 

389 self.param = param 

390 self.mapping = mapping 

391 super().__init__(f"{self.detail}\n{hint}") 

392 

393 

394def _schema_skeleton_mapping(kind: str, schema: dict) -> dict: 

395 """What was detected, ``'<column>'`` for the rest. An explicit schema 

396 replaces auto-detection wholesale, so every required key has to be in it — 

397 not just the ones that failed.""" 

398 spec = _SCHEMA_SPECS[kind] 

399 keys = [key for key, _, _ in spec["required"]] 

400 if spec["groups"]: 

401 # Suggest whichever coordinate convention is closest to complete. 

402 best = min( 

403 spec["groups"], 

404 key=lambda group: sum(1 for key in group if not schema.get(key)), 

405 ) 

406 keys += [key for key in best if key not in keys] 

407 return {key: schema[key] if schema.get(key) else "<column>" for key in keys} 

408 

409 

410def _schema_skeleton(kind: str, schema: dict) -> str: 

411 """:func:`_schema_skeleton_mapping` as a copy-pasteable Python literal.""" 

412 items = ", ".join( 

413 f"{key!r}: {value!r}" 

414 for key, value in _schema_skeleton_mapping(kind, schema).items() 

415 ) 

416 return "{" + items + "}" 

417 

418 

419def _schema_columns(schema: dict) -> list[tuple[str, str]]: 

420 """``(schema key, column name)`` for every column a mapping names. 

421 

422 Multi-column (composite) mappings — the trial / participant / text id may be 

423 a list, see :func:`data.trial_id_series` — expand to one pair per column.""" 

424 pairs: list = [] 

425 for key, value in schema.items(): 

426 if value is None: 

427 continue 

428 if isinstance(value, (list, tuple, set)): 

429 pairs.extend((key, str(col)) for col in value) 

430 else: 

431 pairs.append((key, str(value))) 

432 return pairs 

433 

434 

435def _check_mapped_columns(kind: str, frame: pd.DataFrame, schema: dict) -> None: 

436 """Reject a schema that maps a column the table doesn't have. 

437 

438 Only reachable with a caller-supplied schema — auto-detection only ever 

439 picks columns that exist. Without this check a mistyped mapping raises a 

440 bare ``KeyError: '<column>'`` from inside ``normalize_*``. (It could also be 

441 silently ignored until BUG-58: ``normalize_words`` used to prefer a literal 

442 ``unique_trial_id`` column over the mapped trial id; the mapping wins now.)""" 

443 spec = _SCHEMA_SPECS[kind] 

444 present = {str(c) for c in frame.columns} 

445 missing = [(key, col) for key, col in _schema_columns(schema) if col not in present] 

446 if not missing: 

447 return 

448 plural = "" if len(missing) == 1 else "s" 

449 lines = [ 

450 f"{spec['title']} schema maps {len(missing)} column name{plural} the " 

451 f"{spec['noun']} table doesn't have:" 

452 ] 

453 for key, column in missing: 

454 close = difflib.get_close_matches(column, sorted(present), n=3, cutoff=0.6) 

455 hint = f" (closest: {', '.join(repr(c) for c in close)})" if close else "" 

456 lines.append(f" - {spec['param']}[{key!r}] = {column!r}: no such column{hint}") 

457 lines.append( 

458 f"Columns present in the {spec['noun']} table ({len(frame.columns)}): " 

459 f"{_column_preview(frame)}" 

460 ) 

461 raise SchemaError( 

462 lines, 

463 f"api.propose_schema(table, {kind!r}) returns the auto-detected mapping to " 

464 "start from.", 

465 param=spec["param"], 

466 ) 

467 

468 

469def _known_schema_keys(kind: str, frame: pd.DataFrame) -> set: 

470 """Every key a ``kind`` column mapping can set.""" 

471 spec = _SCHEMA_SPECS[kind] 

472 keys = set(spec["propose"](frame.iloc[:0])) | {"block"} 

473 keys |= {key for key, _, _ in spec["required"]} 

474 keys |= {key for group in spec["groups"] for key in group} 

475 return keys 

476 

477 

478def _unknown_schema_keys(kind: str, frame: pd.DataFrame, schema: dict) -> list: 

479 """The keys of ``schema`` that name no field (#374: ``x_pos`` for ``x``).""" 

480 known = _known_schema_keys(kind, frame) 

481 return [key for key in schema if key not in known] 

482 

483 

484def _schema_error( 

485 kind: str, frame: pd.DataFrame, schema: dict, problems: list, explicit: bool = False 

486) -> SchemaError: 

487 """Build the ``ValueError`` for a table whose canonical fields don't resolve. 

488 

489 Names every canonical field that could not be resolved, the candidate column 

490 names auto-detection tried for it, the columns the table actually has, and 

491 the explicit mapping to pass instead. ``explicit`` marks a schema the caller 

492 supplied — nothing was auto-detected, so the message points at the keys 

493 missing from *their* mapping rather than at failed detection.""" 

494 spec = _SCHEMA_SPECS[kind] 

495 param = spec["param"] 

496 lines = [f"{spec['title']} column mapping problems: {'; '.join(problems)}"] 

497 unknown = _unknown_schema_keys(kind, frame, schema) if explicit else [] 

498 if unknown: 

499 import difflib 

500 

501 known = sorted(_known_schema_keys(kind, frame)) 

502 for key in unknown: 

503 close = difflib.get_close_matches(key, known, n=1) or [ 

504 k for k in known if key.startswith(f"{k}_") or key.endswith(f"_{k}") 

505 ] 

506 hint = f" — did you mean {close[0]!r}?" if close else "" 

507 lines.append(f"{param} has a key that is not a field: {key!r}{hint}") 

508 lines.append(f"Fields: {', '.join(known)}") 

509 

510 if explicit: 

511 bullets = [ 

512 f" - {label} ({param} key {key!r}): not set in the {param} you passed. " 

513 f"Auto-detection (used when {param} is omitted) looks for: " 

514 f"{', '.join(candidates)}" 

515 for key, label, candidates in spec["required"] 

516 if not schema.get(key) 

517 ] 

518 else: 

519 bullets = [ 

520 f" - {label} ({param} key {key!r}): no column matched. " 

521 f"Looked for: {', '.join(candidates)}" 

522 for key, label, candidates in spec["required"] 

523 if not schema.get(key) 

524 ] 

525 groups_missing = [ 

526 (group, [key for key in group if not schema.get(key)]) 

527 for group in spec["groups"] 

528 ] 

529 # A group requirement only fails when *every* alternative is incomplete. 

530 if groups_missing and all(missing for _, missing in groups_missing): 

531 alternatives = " or ".join( 

532 f"({', '.join(group)})" for group, _ in groups_missing 

533 ) 

534 detail = "; ".join( 

535 f"({', '.join(group)}) is missing {', '.join(missing)}" 

536 for group, missing in groups_missing 

537 ) 

538 unresolved = dict.fromkeys( 

539 key for _, missing in groups_missing for key in missing 

540 ) 

541 looked = " | ".join( 

542 f"{key}: {', '.join(spec['group_candidates'][key])}" for key in unresolved 

543 ) 

544 looked_label = "Auto-detection looks for" if explicit else "Looked for" 

545 bullets.append( 

546 f" - {spec['group_label']} ({param} keys): need either {alternatives} " 

547 f"— {detail}.\n {looked_label} → {looked}" 

548 ) 

549 if bullets: 

550 lines.append( 

551 f"Missing from the {param} you passed:" 

552 if explicit 

553 else f"Could not find these columns in the {spec['noun']} table:" 

554 ) 

555 lines.extend(bullets) 

556 resolved = ", ".join( 

557 f"{key}={value!r}" 

558 for key, value in schema.items() 

559 if value is not None and key not in unknown 

560 ) 

561 lines.append( 

562 f"Fields the {param} does set: {resolved or '(none)'}" 

563 if explicit 

564 else f"Fields that did resolve: {resolved or '(none)'}" 

565 ) 

566 lines.append( 

567 f"Columns present in the {spec['noun']} table ({len(frame.columns)}): " 

568 f"{_column_preview(frame)}" 

569 ) 

570 if not explicit: 

571 lines.append( 

572 "Matching ignores case and separators (IA_LEFT == ia_left == 'Ia Left') " 

573 "and takes the first candidate that matches; failing that, a vendor " 

574 "prefix or suffix on a known name (AOI_LEFT, LEFT_px) is tried next, " 

575 "accepted only when exactly one column qualifies." 

576 ) 

577 hint = ( 

578 f"An explicit {param} replaces auto-detection wholesale, so it needs every " 

579 f"required key, e.g. {param}={_schema_skeleton(kind, schema)} — " 

580 f"api.propose_schema(df, {kind!r}) returns the auto-detected mapping." 

581 if explicit 

582 else f"To override auto-detection pass the full mapping, e.g. " 

583 f"{param}={_schema_skeleton(kind, schema)} — " 

584 f"api.propose_schema(df, {kind!r}) returns what was detected." 

585 ) 

586 return SchemaError( 

587 lines, hint, param=param, mapping=_schema_skeleton_mapping(kind, schema) 

588 ) 

589 

590 

591def propose_schema(table: TablesLike, kind: str = "words") -> dict: 

592 """Auto-detected column mapping for a **raw** (un-normalized) table. 

593 

594 ``kind`` is ``"words"``, ``"fixations"`` or ``"raw_gaze"``. Returns 

595 ``{canonical field: source column or None}`` — the same mapping 

596 [`load_scanpath_data`][scanpath_studio.api.load_scanpath_data] infers internally, 

597 so it's the place to start when detection got a field wrong or couldn't find one: 

598 edit the dict and pass it back as ``word_schema=`` / ``fix_schema=``:: 

599 

600 from scanpath_studio import api 

601 

602 schema = api.propose_schema("ia.csv", "words") 

603 schema["trial"] = "TRIAL_LABEL" 

604 words, fixations = api.load_scanpath_data("ia.csv", "fix.csv", 

605 word_schema=schema) 

606 

607 ``table`` is a DataFrame, path, glob or list of paths, like the loader's. 

608 """ 

609 if kind not in _SCHEMA_SPECS: 

610 raise ValueError( 

611 f"Unknown kind {kind!r}; choose one of {', '.join(_SCHEMA_SPECS)}." 

612 ) 

613 frame = _as_dataframe( 

614 table, 

615 _SCHEMA_SPECS[kind]["noun"], 

616 kind=kind if kind in ("words", "fixations") else None, 

617 ) 

618 return _SCHEMA_SPECS[kind]["propose"](frame) 

619 

620 

621_NORMALIZED_ID_COLUMNS = ("participant_id", "trial_id") 

622 

623 

624#: DATA-66: the column vocabularies a loader can return frames in. 

625NAMES_SOURCE = "source" 

626NAMES_CANONICAL = "canonical" 

627 

628 

629class ScanpathData(tuple): 

630 """What [`load_scanpath_data`][scanpath_studio.api.load_scanpath_data] 

631 returns: the ``(words, fixations)`` frames — so 

632 ``words, fixations = load_scanpath_data(…)`` unpacks it — plus 

633 ``column_names``, the dataset's own name for every canonical column, per 

634 table (``{"words": ColumnNames, "fixations": ColumnNames}``). 

635 

636 With ``names="source"`` (the default) the frames' columns are the dataset's 

637 own names and each frame carries its map, so any API function takes it — 

638 sliced, filtered or merged — and the names it writes are yours. With 

639 ``names="canonical"`` they are the internal names, the same for every 

640 dataset.""" 

641 

642 def __new__(cls, words, fixations, column_names=None): 

643 data = super().__new__(cls, (words, fixations)) 

644 data.column_names = dict(column_names or {}) 

645 return data 

646 

647 def __getnewargs__(self): 

648 return (self[0], self[1], self.column_names) 

649 

650 @property 

651 def words(self) -> pd.DataFrame: 

652 return self[0] 

653 

654 @property 

655 def fixations(self) -> pd.DataFrame: 

656 return self[1] 

657 

658 

659def _check_names_choice(names: str) -> None: 

660 if names not in (NAMES_SOURCE, NAMES_CANONICAL): 

661 raise ValueError( 

662 f'names must be "{NAMES_SOURCE}" (the dataset\'s own column names) ' 

663 f'or "{NAMES_CANONICAL}" (the internal ones), got {names!r}.' 

664 ) 

665 

666 

667def _named_in( 

668 frame, label: str, *, optional: bool = False 

669) -> tuple[pd.DataFrame, ColumnNames | None]: 

670 """DATA-66: ``(canonical frame, its map)`` for a frame handed to the API. 

671 

672 A frame :func:`load_scanpath_data` returned under the dataset's own names 

673 carries its map (`column_names.attach`); it is renamed back here, and the 

674 map is returned for the call's options, figure text and output frames. A 

675 canonical frame passes as it is, with no map. ``optional`` takes ``None`` 

676 as the empty table.""" 

677 found = _cn.frame_names(frame) 

678 frame = _cn.to_canonical_frame(frame) 

679 frame = ( 

680 _optional_frame(frame, label) if optional else _require_normalized(frame, label) 

681 ) 

682 return frame, (found[1] if found else None) 

683 

684 

685def _named_out(frame, table: str, names: ColumnNames | None): 

686 """An output frame in the names its inputs carried: a table that 

687 is the dataset's own under its whole map, else (``names`` already 

688 restricted by the caller) only its ids. Canonical inputs, canonical out.""" 

689 if names is None or frame is None or not isinstance(frame, pd.DataFrame): 

690 return frame 

691 return _cn.attach(frame, table, names) 

692 

693 

694def _call_names( 

695 given: dict | None = None, **maps: ColumnNames | None 

696) -> ColumnNames | None: 

697 """One map across the frames of a call (fixations', then words', then raw 

698 gaze's), or ``None`` when every frame was canonical. ``given`` is a 

699 ``column_names=`` argument (``ScanpathData.column_names``), which names 

700 canonical frames and wins over what the frames carry.""" 

701 if given: 

702 return _cn.across_tables( 

703 { 

704 table: names 

705 if isinstance(names, ColumnNames) 

706 else ColumnNames.from_payload(names) 

707 for table, names in given.items() 

708 } 

709 ) 

710 present = {table: names for table, names in maps.items() if names is not None} 

711 return _cn.across_tables(present) if present else None 

712 

713 

714def _table_names( 

715 given: dict | None, table: str, carried: ColumnNames | None 

716) -> ColumnNames | None: 

717 """``table``'s own map for a call — from ``column_names=`` when given, else 

718 what its frame carried. A word option is read in the words table's names 

719 first: the merged map gives a shared column (``word_id``) the fixations'.""" 

720 if given: 

721 names = given.get(table) 

722 if names is None: 

723 return None 

724 return ( 

725 names if isinstance(names, ColumnNames) else ColumnNames.from_payload(names) 

726 ) 

727 return carried 

728 

729 

730def _require_normalized(frame, label: str) -> pd.DataFrame: 

731 """Guard the plotting entry points against raw / wrongly-typed input. 

732 

733 The builders consume the *normalized* frames :func:`load_scanpath_data` 

734 returns; handing them a path or a raw table otherwise fails deep inside with 

735 a ``KeyError: 'participant_id'``.""" 

736 if not isinstance(frame, pd.DataFrame): 

737 raise TypeError( 

738 f"{label} must be the normalized pandas DataFrame returned by " 

739 f"load_scanpath_data(), got {type(frame).__name__}. " 

740 "Call words, fixations = load_scanpath_data(words=…, fixations=…) " 

741 "first — it reads paths/globs and normalizes column names." 

742 ) 

743 # DATA-66: a frame under the dataset's own names is processed canonically. 

744 frame = _cn.to_canonical_frame(frame) 

745 missing = [col for col in _NORMALIZED_ID_COLUMNS if col not in frame.columns] 

746 if missing: 

747 raise ValueError( 

748 f"{label} frame is not normalized: missing the canonical column(s) " 

749 f"{', '.join(missing)}. Its columns are: {_column_preview(frame)}. " 

750 "Pass the frames returned by load_scanpath_data(...) (raw tables have " 

751 "to go through it first). A frame it returned under the dataset's own " 

752 "names loses that map when merged or concatenated with another table " 

753 "(pandas drops DataFrame.attrs there): pass the result through " 

754 "load_scanpath_data(...) again, or load with names='canonical'." 

755 ) 

756 return frame 

757 

758 

759def load_scanpath_data( 

760 words: TablesLike | None = None, 

761 fixations: TablesLike | None = None, 

762 *, 

763 word_schema: dict | None = None, 

764 fix_schema: dict | None = None, 

765 trial_parts_manifest: dict | None = None, 

766 image_root: str | Path | None = None, 

767 image_pattern: str = "{text_id}.png", 

768 keep_columns: Iterable[str] | None = None, 

769 names: str = NAMES_SOURCE, 

770) -> ScanpathData: 

771 """Load and normalize a words table and/or a fixations table. 

772 

773 The columns keep the names your files give them: 

774 ``CURRENT_FIX_DURATION``, not ``duration_ms``. A column Scanpath Studio 

775 built, converted, computed or changed keeps its internal name, and 

776 ``data.column_names`` (a [`ScanpathData`][scanpath_studio.api.ScanpathData]) 

777 records what every column was called. Every API function takes these 

778 frames, and every column option (``color_by=``, hover fields …) takes 

779 either name. ``names="canonical"`` returns the internal names instead — 

780 the same for every dataset, for code that works across them. 

781 

782 ``words`` / ``fixations`` may be DataFrames, paths to ``.csv`` / ``.tsv`` / 

783 ``.txt`` / ``.tab`` / ``.parquet`` / ``.feather`` / ``.xlsx`` / ``.xls`` files (or 

784 a ``.zip`` of them), glob patterns, or lists of paths — multi-file datasets (one 

785 file per participant and/or text) are concatenated, with each file's stem kept in 

786 a ``source_file`` column. Column schemas are auto-detected (EyeLink, Gazepoint, 

787 Tobii, SMI, Pupil Labs, and snake_case names); pass ``word_schema`` / 

788 ``fix_schema`` mappings (field → column name; see 

789 [`propose_schema`][scanpath_studio.api.propose_schema]) 

790 to override detection. 

791 

792 ``trial_parts_manifest`` accepts a nested parent-trial/parts definition for 

793 datasets whose source tables identify screens through arbitrary selector 

794 columns; explicit ``screen_id`` / ``screen_index`` columns can instead be 

795 mapped directly in each schema. Either table may be omitted for datasets 

796 that ship only one report: the 

797 missing side comes back as an empty canonical frame and the plots simply 

798 skip that layer. Words without a participant column (stimulus-level AOIs) 

799 are copied onto every trial in the fixations — each trial matched by 

800 its trial id, else the trial id it had before a repeat's ``_r2`` suffix, 

801 else its ``text_id`` (trial ids that embed the participant), with a 

802 ``data.StimulusJoinWarning`` (a ``UserWarning``) when some trials match 

803 none — and fixations without x/y but with a word/AOI ID are placed at 

804 word-box centers. Columns named in ``data.INTERNAL_COLUMNS`` are the 

805 pipeline's bookkeeping (``data.drop_internal_columns`` removes them). 

806 

807 Normalization keeps the mapped fields and the recognized optional ones 

808 (eye, EyeLink's interest-area measures, linguistic features …) and drops the 

809 rest. ``keep_columns`` names further columns of your own to carry through 

810 under their own names — a pupil size, a detection confidence — from 

811 whichever table has them, so a figure can color, hover or plot by them 

812 (the app's *Extra fields to keep*; ``render --keep-columns`` on the command line). 

813 

814 Returns the normalized ``(words, fixations)`` frames the plotting 

815 functions expect. Raises ``ValueError`` if a required field can't be found — 

816 the message names the canonical field, the column names auto-detection 

817 looked for, and the columns the table actually has — and 

818 ``data.StimulusJoinError`` (a ``ValueError``) when a stimulus-level words 

819 table shares neither a trial id nor a ``text_id`` with any trial (or, 

820 multipart, with every screen a trial has fixations on). 

821 """ 

822 _check_names_choice(names) 

823 if words is None and fixations is None: 

824 raise ValueError("Provide at least one of words= or fixations=.") 

825 # DATA-66: a frame this API already named is loaded under its internal 

826 # names; its own map then renames the new one back to the user's. 

827 prior = { 

828 table: found[1] 

829 for table, frame in (("words", words), ("fixations", fixations)) 

830 if (found := _cn.frame_names(frame)) is not None 

831 } 

832 

833 if words is not None: 

834 # BUG-53: a word spelled "None" or "NA" is a word, not a missing cell. 

835 words_df = _as_dataframe( 

836 words, 

837 "words", 

838 plan_for=lambda header: _data.verbatim_text_plan(header, word_schema), 

839 kind="words", 

840 ) 

841 explicit = word_schema is not None 

842 word_schema = word_schema or _data.propose_word_schema(words_df) 

843 _check_mapped_columns("words", words_df, word_schema) 

844 problems = _data.validate_word_schema(word_schema) 

845 if problems: 

846 raise _schema_error("words", words_df, word_schema, problems, explicit) 

847 words_norm = _data.normalize_words( 

848 words_df, 

849 word_schema, 

850 keep_columns=_with_optional_fields( 

851 keep_columns, _data.WORD_OPTIONAL_FIELDS 

852 ), 

853 ) 

854 if trial_parts_manifest is not None: 

855 words_norm = apply_trial_parts_manifest( 

856 words_norm, words_df, trial_parts_manifest, kind="words" 

857 ) 

858 else: 

859 words_norm = _data.empty_words_frame() 

860 

861 if fixations is not None: 

862 fixations_df = _as_dataframe( 

863 fixations, 

864 "fixations", 

865 plan_for=lambda header: _data.identity_text_plan(header, fix_schema), 

866 kind="fixations", 

867 ) 

868 explicit = fix_schema is not None 

869 fix_schema = fix_schema or _data.propose_fix_schema(fixations_df) 

870 _check_mapped_columns("fixations", fixations_df, fix_schema) 

871 problems = _data.validate_fix_schema(fix_schema) 

872 if problems: 

873 raise _schema_error( 

874 "fixations", fixations_df, fix_schema, problems, explicit 

875 ) 

876 fixations_norm = _data.normalize_fixations( 

877 fixations_df, 

878 fix_schema, 

879 keep_columns=_with_optional_fields(keep_columns, _data.FIX_OPTIONAL_FIELDS), 

880 ) 

881 if trial_parts_manifest is not None: 

882 fixations_norm = apply_trial_parts_manifest( 

883 fixations_norm, 

884 fixations_df, 

885 trial_parts_manifest, 

886 kind="fixations", 

887 ) 

888 else: 

889 fixations_norm = _data.empty_fixations_frame() 

890 

891 words_norm, fixations_norm, _join, rewrites = _data.harmonize_frames_reporting( 

892 words_norm, fixations_norm 

893 ) 

894 if image_root is not None: 

895 words_norm = _data.resolve_stimulus_image_paths( 

896 words_norm, image_root, image_pattern 

897 ) 

898 fixations_norm = _data.resolve_stimulus_image_paths( 

899 fixations_norm, image_root, image_pattern 

900 ) 

901 # DATA-66: what each column was called in these files — from the schemas 

902 # and raw columns normalization read, with the columns the fixups rewrote 

903 # marked converted. 

904 maps = { 

905 table: ColumnNames.from_payload(payload) 

906 for table, payload in _cn.for_tables( 

907 {"words": word_schema, "fixations": fix_schema}, 

908 { 

909 "words": words_df if words is not None else None, 

910 "fixations": fixations_df if fixations is not None else None, 

911 }, 

912 rewrites=rewrites, 

913 ).items() 

914 } 

915 maps = { 

916 table: names_map.through(prior[table]) if table in prior else names_map 

917 for table, names_map in maps.items() 

918 } 

919 if names == NAMES_SOURCE: 

920 words_norm = _cn.attach(words_norm, "words", maps.get("words")) 

921 fixations_norm = _cn.attach(fixations_norm, "fixations", maps.get("fixations")) 

922 return ScanpathData(words_norm, fixations_norm, maps) 

923 

924 

925def _with_optional_fields( 

926 keep_columns: Iterable[str] | None, registry: list 

927) -> set | None: 

928 """``keep_columns`` as the normalizers take it: ``None`` (every recognized 

929 optional field, nothing else) when none are named, else those names *plus* 

930 every optional field — a non-``None`` set would otherwise limit them.""" 

931 if not keep_columns: 

932 return None 

933 if isinstance(keep_columns, str): 

934 keep_columns = [keep_columns] 

935 return {str(c) for c in keep_columns} | {entry[0] for entry in registry} 

936 

937 

938def load_participant_metadata( 

939 table: TablesLike, 

940 *, 

941 id_column: str | None = None, 

942 participants: pd.DataFrame | list | None = None, 

943): 

944 """Load a participant-level metadata table. 

945 

946 ``table`` is a DataFrame or a path/glob to a CSV/TSV/Parquet/Excel file with 

947 **one row per participant**: an id column plus anything known about them 

948 (``native_language``, ``age``, a comprehension score). ``id_column`` 

949 defaults to the first recognized spelling (``participant_id``, ``subject``, 

950 ``RECORDING_SESSION_LABEL``, …). 

951 

952 Pass ``participants`` — a normalized frame or a list of ids — to have the 

953 join validated against the data you actually loaded; the returned object's 

954 ``.report`` then names the participants missing from either side. 

955 

956 Returns a 

957 `ParticipantMetadata`: the cleaned frame, 

958 a field registry (name, label, grain, dtype, missingness), and the join 

959 report. Nothing is broadcast onto the words/fixations frames — use 

960 `scanpath_studio.metadata.project` to attach chosen columns to a 

961 per-trial frame, or ``.values_for(pid)`` for one participant. 

962 

963 >>> words, fixations = load_sample_data() 

964 >>> meta = load_participant_metadata( 

965 ... "readers.csv", participants=fixations 

966 ... ) # doctest: +SKIP 

967 >>> meta.names # doctest: +SKIP 

968 ('native_language', 'age') 

969 """ 

970 from scanpath_studio import metadata as _metadata 

971 

972 frame = _as_dataframe( 

973 table, 

974 "participant metadata", 

975 plan_for=_metadata_id_plan(id_column, _metadata.infer_participant_id_column), 

976 ) 

977 resolved = id_column or _metadata.infer_participant_id_column(frame) 

978 if not resolved or resolved not in frame.columns: 

979 raise ValueError( 

980 "Could not find the participant-id column in the metadata table. " 

981 f"Columns: {_column_preview(frame)}. Pass id_column= explicitly." 

982 ) 

983 if isinstance(participants, pd.DataFrame): 

984 participants = _metadata.participant_ids(_cn.to_canonical_frame(participants)) 

985 return _metadata.build_participant_metadata( 

986 frame, 

987 resolved, 

988 source_name=getattr(table, "name", None) or "participant metadata", 

989 participants=participants, 

990 ) 

991 

992 

993def load_trial_metadata( 

994 table: TablesLike, 

995 *, 

996 id_column: str | None = None, 

997 participant_column: str | None = None, 

998 trials: pd.DataFrame | None = None, 

999): 

1000 """Load a trial-level metadata table. 

1001 

1002 The sibling of 

1003 [`load_participant_metadata`][scanpath_studio.api.load_participant_metadata], one 

1004 grain down: ``table`` has **one row per trial** — a trial-id column plus anything 

1005 known about that trial (a list name, a condition, a per-trial comprehension 

1006 score). 

1007 

1008 **The key is yours to state, and it changes what the table means.** Keyed by 

1009 trial id alone, a row describes a *text*, and every trial of it 

1010 inherits that row; pass ``participant_column`` to key by participant **and** 

1011 trial, so a row describes one *trial*. Nothing in a file says which world 

1012 a corpus is in, so this is never inferred — unlike ``id_column``, which 

1013 defaults to the first recognized spelling (``trial_id``, ``item_id``, 

1014 ``TRIAL_INDEX``, …). 

1015 

1016 Pass ``trials`` — a normalized fixations/words frame, or any frame with 

1017 ``participant_id`` + ``trial_id`` — to have the join validated against the 

1018 data you actually loaded; the returned ``.report`` then names the trials 

1019 missing from either side. 

1020 

1021 Returns a `TrialMetadata`: the cleaned 

1022 frame, a field registry, and the join report. As with the participant 

1023 table, nothing is broadcast onto the words/fixations frames. 

1024 

1025 >>> words, fixations = load_sample_data() 

1026 >>> meta = load_trial_metadata( 

1027 ... "readings.csv", trials=fixations 

1028 ... ) # doctest: +SKIP 

1029 >>> meta.names # doctest: +SKIP 

1030 ('list_name', 'comprehension_score') 

1031 """ 

1032 from scanpath_studio import metadata as _metadata 

1033 

1034 frame = _as_dataframe( 

1035 table, 

1036 "trial metadata", 

1037 plan_for=_metadata_id_plan( 

1038 id_column, _metadata.infer_trial_id_column, participant_column 

1039 ), 

1040 ) 

1041 resolved = id_column or _metadata.infer_trial_id_column(frame) 

1042 if not resolved or resolved not in frame.columns: 

1043 raise ValueError( 

1044 "Could not find the trial-id column in the metadata table. " 

1045 f"Columns: {_column_preview(frame)}. Pass id_column= explicitly." 

1046 ) 

1047 if participant_column and participant_column not in frame.columns: 

1048 raise ValueError( 

1049 f"participant_column={participant_column!r} is not in the metadata " 

1050 f"table. Columns: {_column_preview(frame)}." 

1051 ) 

1052 keys = ( 

1053 _metadata.trial_keys(_cn.to_canonical_frame(trials)) 

1054 if trials is not None 

1055 else None 

1056 ) 

1057 return _metadata.build_trial_metadata( 

1058 frame, 

1059 resolved, 

1060 participant_column, 

1061 source_name=getattr(table, "name", None) or "trial metadata", 

1062 keys=keys, 

1063 ) 

1064 

1065 

1066def load_text_metadata( 

1067 table: TablesLike, 

1068 *, 

1069 id_column: str | list[str] | None = None, 

1070 texts: pd.DataFrame | list | None = None, 

1071): 

1072 """Load a text-level metadata table — the third grain. 

1073 

1074 ``table`` has **one row per text** — a text-id column plus anything known about that 

1075 text (genre, difficulty, a stimulus-level comprehension score). Flat grain, like 

1076 [`load_participant_metadata`][scanpath_studio.api.load_participant_metadata]: never 

1077 keyed by participant, since a text is a stimulus rather than something one participant owns. 

1078 ``id_column`` defaults to the first recognized spelling (``text_id``, 

1079 ``paragraph_id``, ``stimulus_id``, …) and may be several columns to build a 

1080 composite id, the same way the uploaded data's own Text ID mapping does. 

1081 

1082 Pass ``texts`` — a normalized fixations/words frame, or any iterable of 

1083 text ids — to have the join validated against the data you actually 

1084 loaded; the returned ``.report`` then names the texts missing from either 

1085 side. 

1086 

1087 Returns a `TextMetadata`: the cleaned 

1088 frame, a field registry, and the join report. As with the other two 

1089 grains, nothing is broadcast onto the words/fixations frames. 

1090 

1091 >>> words, fixations = load_sample_data() 

1092 >>> meta = load_text_metadata( 

1093 ... "texts.csv", texts=words 

1094 ... ) # doctest: +SKIP 

1095 >>> meta.names # doctest: +SKIP 

1096 ('genre', 'difficulty') 

1097 """ 

1098 from scanpath_studio import metadata as _metadata 

1099 

1100 frame = _as_dataframe( 

1101 table, 

1102 "text metadata", 

1103 plan_for=_metadata_id_plan(id_column, _metadata.infer_text_id_column), 

1104 ) 

1105 resolved = id_column or _metadata.infer_text_id_column(frame) 

1106 if not resolved or any( 

1107 c not in frame.columns for c in _metadata.trial_mapping_columns(resolved) 

1108 ): 

1109 raise ValueError( 

1110 "Could not find the text-id column in the metadata table. " 

1111 f"Columns: {_column_preview(frame)}. Pass id_column= explicitly." 

1112 ) 

1113 if isinstance(texts, pd.DataFrame): 

1114 texts = _metadata.text_keys(_cn.to_canonical_frame(texts)) 

1115 return _metadata.build_text_metadata( 

1116 frame, 

1117 resolved, 

1118 source_name=getattr(table, "name", None) or "text metadata", 

1119 keys=texts, 

1120 ) 

1121 

1122 

1123def load_sample_data(*, names: str = NAMES_SOURCE) -> ScanpathData: 

1124 """Return the bundled OneStop demo, normalized and ready to plot: two 

1125 participants, twelve trials each, every one of them with fixations. Under 

1126 the demo's own column names; ``names="canonical"`` for the internal ones 

1127 (see [`load_scanpath_data`][scanpath_studio.api.load_scanpath_data]). 

1128 

1129 The frames carry the demo's recorded screen (OneStop's 2560×1440), so 

1130 `plot_scanpath` draws them on it without a ``canvas_size``, as 

1131 ``scanpath-studio render --sample`` does.""" 

1132 from .code_snippet import SOURCE_DEMO, source_canvas 

1133 

1134 data = load_scanpath_data(*_data.load_sample_data(), names=names) 

1135 screen = source_canvas(SOURCE_DEMO) 

1136 for frame in data: 

1137 frame.attrs[RECORDED_SCREEN_ATTR] = screen 

1138 return data 

1139 

1140 

1141#: The `DataFrame.attrs` key a frame carries its dataset's recorded screen in, 

1142#: ``(width, height)`` px — read when no ``canvas_size`` is passed. 

1143RECORDED_SCREEN_ATTR = "scanpath_studio.recorded_screen" 

1144 

1145 

1146def _recorded_screen(*frames) -> tuple[int, int] | None: 

1147 """The recorded screen one of ``frames`` carries (`load_sample_data`).""" 

1148 for frame in frames: 

1149 screen = getattr(frame, "attrs", {}).get(RECORDED_SCREEN_ATTR) 

1150 if screen: 

1151 return int(screen[0]), int(screen[1]) 

1152 return None 

1153 

1154 

1155def load_raw_gaze( 

1156 table: TablesLike, 

1157 *, 

1158 raw_gaze_schema: dict | None = None, 

1159 names: str = NAMES_SOURCE, 

1160) -> pd.DataFrame: 

1161 """Load and normalize a raw (sample-level) gaze table for ``raw_gaze=``. 

1162 

1163 The third table [`plot_scanpath`][scanpath_studio.api.plot_scanpath] can draw, under 

1164 the fixations: one row per eye-tracker sample, with a participant, a trial, ``x`` / 

1165 ``y`` and usually a timestamp. ``table`` is a DataFrame, path, glob or list of 

1166 paths, like [`load_scanpath_data`][scanpath_studio.api.load_scanpath_data]'s, and 

1167 the columns are auto-detected the same way; pass ``raw_gaze_schema`` (field → 

1168 column, see ``api.propose_schema(table, "raw_gaze")``) to override the detection. 

1169 ``plot_scanpath`` keeps only the plotted trial's (and screen's) samples, so one 

1170 table can serve a whole corpus:: 

1171 

1172 raw_gaze = sps.load_raw_gaze("gaze_samples.csv") 

1173 fig = sps.plot_scanpath(words, fixations, "p1", "t3", raw_gaze=raw_gaze) 

1174 

1175 Under the table's own column names, like ``load_scanpath_data``'s; 

1176 ``names="canonical"`` for the internal ones. 

1177 """ 

1178 _check_names_choice(names) 

1179 frame = _as_dataframe( 

1180 table, 

1181 "raw gaze", 

1182 plan_for=lambda header: _data.identity_text_plan( 

1183 header, raw_gaze_schema, kind="raw_gaze" 

1184 ), 

1185 ) 

1186 explicit = raw_gaze_schema is not None 

1187 schema = raw_gaze_schema or _data.propose_raw_gaze_schema(frame) 

1188 _check_mapped_columns("raw_gaze", frame, schema) 

1189 problems = _data.validate_raw_gaze_schema(schema) 

1190 if problems: 

1191 raise _schema_error("raw_gaze", frame, schema, problems, explicit) 

1192 normalized = _data.normalize_raw_gaze(frame, schema) 

1193 if names == NAMES_CANONICAL: 

1194 return normalized 

1195 return _cn.attach( 

1196 normalized, "raw_gaze", _cn.from_schema("raw_gaze", schema, frame.columns) 

1197 ) 

1198 

1199 

1200def load_sample_raw_gaze(*, names: str = NAMES_SOURCE) -> pd.DataFrame: 

1201 """The bundled demo's raw gaze, normalized — what the app overlays on it. 

1202 

1203 OneStop ships no sample-level gaze, so this is **synthesized** from one of 

1204 the demo's real trials and covers that trial alone.""" 

1205 return load_raw_gaze(_data.load_sample_raw_gaze(), names=names) 

1206 

1207 

1208def check_data_health( 

1209 words: pd.DataFrame | None = None, 

1210 fixations: pd.DataFrame | None = None, 

1211 raw_gaze: pd.DataFrame | None = None, 

1212) -> pd.DataFrame: 

1213 """Values that loaded as numbers but cannot be right — the Data Management page's *Data checks*. 

1214 

1215 Checks the normalized tables (from 

1216 [`load_scanpath_data`][scanpath_studio.api.load_scanpath_data] / 

1217 [`load_raw_gaze`][scanpath_studio.api.load_raw_gaze]) for fixations lasting 

1218 0 ms or less or with an infinite duration or onset, fixations and raw-gaze 

1219 samples whose position is missing or infinite, word boxes with no area or no 

1220 finite position, and per-screen screen sizes that are not finite and 

1221 positive. One row per check that found 

1222 anything: ``table``, ``check``, ``problem``, the ``columns`` it read (in the 

1223 names the frames carry), 

1224 ``rows`` of ``of_rows``, the ``trials`` they fall in, a ``breakdown`` by 

1225 kind, ``severity`` (``"note"`` for raw-gaze gaps, which blinks and track 

1226 loss make ordinary), ``what_happens`` to those rows in the app, and a few 

1227 ``examples``. An empty frame means every check passed. Nothing is changed 

1228 or dropped:: 

1229 

1230 words, fixations = sps.load_scanpath_data("ia.csv", "fixations.csv") 

1231 print(sps.check_data_health(words, fixations)) 

1232 """ 

1233 from .data_health import findings_frame 

1234 

1235 return findings_frame(_health_findings(words, fixations, raw_gaze)) 

1236 

1237 

1238def _health_findings(words, fixations, raw_gaze) -> list: 

1239 """`data_health.check_data_health` on frames in either naming, its findings 

1240 naming the columns as the frames did. The CLI's ``check`` prints 

1241 these; :func:`check_data_health` tabulates them.""" 

1242 from dataclasses import replace 

1243 

1244 from .data_health import check_data_health as _check 

1245 

1246 named = {} 

1247 for table, frame in ( 

1248 ("words", words), 

1249 ("fixations", fixations), 

1250 ("raw_gaze", raw_gaze), 

1251 ): 

1252 found = _cn.frame_names(frame) if frame is not None else None 

1253 named[table] = ( 

1254 _cn.to_canonical_frame(frame) if frame is not None else None, 

1255 found[1] if found else None, 

1256 ) 

1257 findings = _check(*(frame for frame, _names in named.values())) 

1258 

1259 def _in_own_names(finding): 

1260 # DATA-66: name the columns and example fields as the frames did. 

1261 names = named[finding.table][1] 

1262 if names is None: 

1263 return finding 

1264 return replace( 

1265 finding, 

1266 columns=tuple(names.display(c) for c in finding.columns), 

1267 examples=tuple( 

1268 {names.display(k): v for k, v in row.items()} 

1269 for row in finding.examples 

1270 ), 

1271 ) 

1272 

1273 return [_in_own_names(f) for f in findings] 

1274 

1275 

1276def _require_computed_measures(name: str) -> None: 

1277 """Refuse a held-back computation, naming the switch that enables it. 

1278 

1279 These values are computed by Scanpath Studio rather than read from the 

1280 dataset, and each needs checking by hand before it is released. A script 

1281 gets an error rather than an unchecked number. 

1282 """ 

1283 if not computed_measures_enabled(): 

1284 raise ValueError( 

1285 f"{name} is not available in this release: its values are computed " 

1286 "by Scanpath Studio and have not been validated yet. Set " 

1287 f"{EXPERIMENTAL_ENV_VAR}=1 to use it anyway." 

1288 ) 

1289 

1290 

1291def compute_word_metrics(words: pd.DataFrame, fixations: pd.DataFrame) -> pd.DataFrame: 

1292 """Per-word reading measures (FFD/FPRT/RPD/TFD, skips, regressions, …). 

1293 

1294 Experimental: raises unless ``SCANPATH_EXPERIMENTAL=1``. Pre-aggregated 

1295 columns in ``words`` (EyeLink IA exports) are preserved; anything missing is 

1296 computed from fixations + word bounding boxes. Takes the normalized frames 

1297 from [`load_scanpath_data`][scanpath_studio.api.load_scanpath_data], and 

1298 answers in the names they carry.""" 

1299 _require_computed_measures("compute_word_metrics") 

1300 words, word_names = _named_in(words, "words") 

1301 fixations, _fix_names = _named_in(fixations, "fixations") 

1302 return _named_out(_data.compute_word_metrics(words, fixations), "words", word_names) 

1303 

1304 

1305def trial_summary(words: pd.DataFrame, fixations: pd.DataFrame) -> pd.DataFrame: 

1306 """Exportable one-row-per-trial reading summary. 

1307 

1308 Experimental: raises unless ``SCANPATH_EXPERIMENTAL=1``.""" 

1309 _require_computed_measures("trial_summary") 

1310 from .aggregation import trial_summary_table 

1311 

1312 words, word_names = _named_in(words, "words", optional=True) 

1313 fixations, fix_names = _named_in(fixations, "fixations", optional=True) 

1314 names = _call_names(fixations=fix_names, words=word_names) 

1315 return _named_out( 

1316 trial_summary_table(words, fixations), 

1317 "trial_summary", 

1318 names.identity() if names else None, 

1319 ) 

1320 

1321 

1322def reader_summary(words: pd.DataFrame, fixations: pd.DataFrame) -> pd.DataFrame: 

1323 """Exportable one-row-per-reader reading summary. 

1324 

1325 Experimental: raises unless ``SCANPATH_EXPERIMENTAL=1``.""" 

1326 _require_computed_measures("reader_summary") 

1327 from .aggregation import reader_summary_table 

1328 

1329 words, word_names = _named_in(words, "words", optional=True) 

1330 fixations, fix_names = _named_in(fixations, "fixations", optional=True) 

1331 names = _call_names(fixations=fix_names, words=word_names) 

1332 return _named_out( 

1333 reader_summary_table(words, fixations), 

1334 "reader_summary", 

1335 names.identity() if names else None, 

1336 ) 

1337 

1338 

1339def preprocess_data( 

1340 words: pd.DataFrame, 

1341 fixations: pd.DataFrame, 

1342 *, 

1343 enabled: bool = False, 

1344 short_policy: str = "Off", 

1345 short_threshold_ms: float = 80.0, 

1346 merge_distance_chars: float = 1.0, 

1347 discard_blink_adjacent: bool = False, 

1348) -> tuple[pd.DataFrame, pd.DataFrame, pd.DataFrame]: 

1349 """Apply the optional preprocessing stage and return words/fixations/QA. 

1350 

1351 Experimental: raises unless ``SCANPATH_EXPERIMENTAL=1``.""" 

1352 _require_computed_measures("preprocess_data") 

1353 if not enabled: 

1354 return words, fixations, pd.DataFrame() 

1355 

1356 from .measures import assign_fixations_to_words, enrich_fixations 

1357 from .preprocessing import preprocess_fixations 

1358 

1359 given_words = words 

1360 words, _word_names = _named_in(words, "words", optional=True) 

1361 fixations, fix_names = _named_in(fixations, "fixations", optional=True) 

1362 

1363 assigned = ( 

1364 enrich_fixations(assign_fixations_to_words(fixations, words), words) 

1365 if not fixations.empty 

1366 else fixations 

1367 ) 

1368 processed, report = preprocess_fixations( 

1369 assigned, 

1370 words, 

1371 settings={ 

1372 "enabled": enabled, 

1373 "short_policy": short_policy, 

1374 "short_threshold_ms": short_threshold_ms, 

1375 "merge_distance_chars": merge_distance_chars, 

1376 "discard_blink_adjacent": discard_blink_adjacent, 

1377 }, 

1378 ) 

1379 # The QA report is derived: it names its ids as the fixations do. 

1380 return ( 

1381 given_words, 

1382 _named_out(processed, "fixations", fix_names), 

1383 _named_out(report, "cleaning_qa", fix_names.identity() if fix_names else None), 

1384 ) 

1385 

1386 

1387def analysis_tables( 

1388 words: pd.DataFrame, 

1389 fixations: pd.DataFrame, 

1390 *, 

1391 pixels_per_degree: float | None = None, 

1392 raw_gaze: pd.DataFrame | None = None, 

1393) -> dict[str, pd.DataFrame]: 

1394 """The tables ``scanpath-studio analyze`` writes, as a dict of frames. 

1395 

1396 Experimental: raises unless ``SCANPATH_EXPERIMENTAL=1``. 

1397 

1398 ``fixations``, ``saccades``, ``word_measures``, ``sentence_measures``, 

1399 ``trial_summary``, ``reader_summary``, ``characters`` and ``cleaning_qa``. 

1400 

1401 ``word_measures`` is the words table with the reading measures it 

1402 *brought*: none are computed here, and a words table that carries none 

1403 leaves ``word_measures`` out. 

1404 """ 

1405 _require_computed_measures("analysis_tables") 

1406 from .aggregation import reader_summary_table, trial_summary_table 

1407 from .measures import assign_fixations_to_words, enrich_fixations 

1408 from .preprocessing import ( 

1409 character_grid, 

1410 cleaning_report, 

1411 saccade_table, 

1412 sentence_measures, 

1413 ) 

1414 

1415 words, word_names = _named_in(words, "words", optional=True) 

1416 fixations, fix_names = _named_in(fixations, "fixations", optional=True) 

1417 if raw_gaze is not None: 

1418 raw_gaze, _gaze_names = _named_in(raw_gaze, "raw_gaze") 

1419 analysis_fixations = ( 

1420 enrich_fixations(assign_fixations_to_words(fixations, words), words) 

1421 if not fixations.empty and not words.empty 

1422 else fixations 

1423 ) 

1424 tables = { 

1425 "fixations": analysis_fixations, 

1426 "saccades": saccade_table( 

1427 analysis_fixations, 

1428 pixels_per_degree=pixels_per_degree, 

1429 raw_gaze=raw_gaze, 

1430 words=words, 

1431 ), 

1432 "word_measures": words, 

1433 "sentence_measures": sentence_measures(words, analysis_fixations), 

1434 "trial_summary": trial_summary_table(words, analysis_fixations), 

1435 "reader_summary": reader_summary_table(words, analysis_fixations), 

1436 "characters": character_grid(words), 

1437 "cleaning_qa": cleaning_report(analysis_fixations), 

1438 } 

1439 if not _data.brought_reading_measures(words): 

1440 del tables["word_measures"] 

1441 # DATA-66: in the names the frames carried — the two tables that are the 

1442 # dataset's own under their whole map, the derived ones by their ids only 

1443 # (the export bundle's rule, `export._ARTIFACT_TABLE`). 

1444 names = _call_names(fixations=fix_names, words=word_names) 

1445 if names is None: 

1446 return tables 

1447 own = {"fixations": fix_names, "word_measures": word_names} 

1448 return { 

1449 artifact: _named_out( 

1450 table, 

1451 artifact, 

1452 own[artifact] if artifact in own else names.identity(), 

1453 ) 

1454 for artifact, table in tables.items() 

1455 } 

1456 

1457 

1458def alignment_sensitivity( 

1459 words: pd.DataFrame, 

1460 fixations: pd.DataFrame, 

1461 methods: tuple[str, ...] = ("attach", "slice", "consensus"), 

1462) -> tuple[pd.DataFrame, pd.DataFrame]: 

1463 """Word-measure sensitivity and correction QA across line algorithms. 

1464 

1465 A derived surface of vertical drift correction, so it is gated with 

1466 it and raises rather than returning something that looks like a result. 

1467 """ 

1468 if not drift_correction_enabled(): 

1469 raise ValueError( 

1470 "alignment_sensitivity is not available in this release (vertical " 

1471 "drift correction is not fully integrated yet). Set " 

1472 f"{EXPERIMENTAL_ENV_VAR}=1 to enable it." 

1473 ) 

1474 from .preprocessing import measure_sensitivity 

1475 

1476 words, word_names = _named_in(words, "words") 

1477 fixations, fix_names = _named_in(fixations, "fixations") 

1478 names = _call_names(fixations=fix_names, words=word_names) 

1479 tables = measure_sensitivity(words, fixations, methods) 

1480 if names is None: 

1481 return tables 

1482 return tuple( 

1483 _named_out(table, "alignment_sensitivity", names.identity()) for table in tables 

1484 ) 

1485 

1486 

1487#: The columns each corpus-figure kind reads; ``"<value>"`` stands for 

1488#: ``value_col``. EXP-13: without the check a table lacking one surfaced as a 

1489#: bare ``KeyError: 'value'`` from inside the builder. 

1490_CORPUS_COLUMNS = { 

1491 "profile": ("word_id", "<value>"), 

1492 "distribution": ("<value>",), 

1493 # EXP-16: the builder draws its "no data" placeholder for a table with no 

1494 # `diff` — right for the app's empty states, but headlessly it meant a 

1495 # figure with nothing on it and an exit code of 0. 

1496 "difference": ("word_id", "diff"), 

1497} 

1498 

1499 

1500def _require_corpus_columns(data: pd.DataFrame, kind: str, value_col: str) -> None: 

1501 required = [ 

1502 value_col if column == "<value>" else column 

1503 for column in _CORPUS_COLUMNS.get(kind, ()) 

1504 ] 

1505 missing = [column for column in required if column not in data.columns] 

1506 if not missing: 

1507 return 

1508 hint = ( 

1509 f" Name the measure column with value_col= (--value-col on the CLI); " 

1510 f"it is {value_col!r} now." 

1511 if value_col in missing 

1512 else "" 

1513 ) 

1514 raise ValueError( 

1515 f"A {kind!r} corpus figure reads the column(s) " 

1516 f"{', '.join(repr(column) for column in missing)}, which the table doesn't " 

1517 f"have. Columns present ({len(data.columns)}): {_column_preview(data)}.{hint}" 

1518 ) 

1519 

1520 

1521def plot_corpus_figure( 

1522 data: pd.DataFrame, 

1523 *, 

1524 kind: str, 

1525 measure_label: str = "Value", 

1526 series_col: str = "series", 

1527 value_col: str = "value", 

1528 colors: tuple[str, ...] | None = None, 

1529 canvas_width: int = 1000, 

1530 base_font_size: int = 14, 

1531 font_family: str = FONT_FAMILY, 

1532) -> go.Figure: 

1533 """Headless corpus profile/distribution/difference plot with shared colors. 

1534 

1535 ``profile`` expects ``word_id`` plus ``value_col`` (and optional ``lo`` / 

1536 ``hi``); ``distribution`` expects ``value_col``; ``difference`` expects 

1537 ``word_id`` and ``diff``. When ``series_col`` is present, it defines the 

1538 overlaid profile/distribution series. A table missing a column its ``kind`` 

1539 reads raises ``ValueError`` naming it and the columns present. 

1540 """ 

1541 kind = str(kind).lower() 

1542 _require_corpus_columns(data, kind, value_col) 

1543 if kind == "profile": 

1544 profiles = ( 

1545 { 

1546 str(name): group.rename(columns={value_col: "value"}) 

1547 for name, group in data.groupby(series_col, sort=False) 

1548 } 

1549 if series_col in data 

1550 else {measure_label: data.rename(columns={value_col: "value"})} 

1551 ) 

1552 return make_word_profile_figure( 

1553 profiles, 

1554 measure_label=measure_label, 

1555 canvas_width=canvas_width, 

1556 base_font_size=base_font_size, 

1557 font_family=font_family, 

1558 colors=colors, 

1559 ) 

1560 if kind == "distribution": 

1561 groups = ( 

1562 { 

1563 str(name): group[value_col].dropna().to_numpy() 

1564 for name, group in data.groupby(series_col, sort=False) 

1565 } 

1566 if series_col in data 

1567 else {measure_label: data[value_col].dropna().to_numpy()} 

1568 ) 

1569 return make_distribution_figure( 

1570 groups, 

1571 metric_label=measure_label, 

1572 canvas_width=canvas_width, 

1573 base_font_size=base_font_size, 

1574 font_family=font_family, 

1575 colors=colors, 

1576 ) 

1577 if kind == "difference": 

1578 return make_difference_profile_figure( 

1579 data, 

1580 measure_label=measure_label, 

1581 canvas_width=canvas_width, 

1582 base_font_size=base_font_size, 

1583 font_family=font_family, 

1584 colors=colors, 

1585 ) 

1586 raise ValueError("kind must be 'profile', 'distribution', or 'difference'.") 

1587 

1588 

1589def _optional_frame(frame, label: str) -> pd.DataFrame: 

1590 """``frame`` checked as normalized, or the empty canonical frame for ``None``. 

1591 

1592 A dataset recorded as raw gaze alone has no words or fixations 

1593 table, so the plotting entry points take ``None`` for either — the same 

1594 empty canonical frame `load_scanpath_data` returns for a table it was not 

1595 given.""" 

1596 if frame is None: 

1597 return ( 

1598 _data.empty_words_frame() 

1599 if label == "words" 

1600 else _data.empty_fixations_frame() 

1601 ) 

1602 return _require_normalized(frame, label) 

1603 

1604 

1605def list_trials( 

1606 words: pd.DataFrame | None = None, 

1607 fixations: pd.DataFrame | None = None, 

1608 *, 

1609 raw_gaze: pd.DataFrame | None = None, 

1610) -> pd.DataFrame: 

1611 """One row per plottable trial: its participant id and trial id. 

1612 

1613 Trials present in both frames when both are loaded; for single-report 

1614 datasets (words-only or fixations-only), trials from whichever frame has 

1615 data. ``raw_gaze`` (a frame from 

1616 [`load_raw_gaze`][scanpath_studio.api.load_raw_gaze]) adds the trials that 

1617 only its samples cover — every trial, for a dataset recorded as raw gaze 

1618 alone (pass ``None`` for ``words`` and ``fixations`` then). The id columns 

1619 take the names the frames carry.""" 

1620 words, word_names = _named_in(words, "words", optional=True) 

1621 fixations, fix_names = _named_in(fixations, "fixations", optional=True) 

1622 gaze_names = None 

1623 if raw_gaze is not None: 

1624 raw_gaze, gaze_names = _named_in(raw_gaze, "raw_gaze") 

1625 names = _call_names(fixations=fix_names, words=word_names, raw_gaze=gaze_names) 

1626 cols = ["participant_id", "trial_id"] 

1627 if words.empty or fixations.empty: 

1628 present = fixations if words.empty else words 

1629 combos = present[cols].drop_duplicates() 

1630 else: 

1631 combos = words[cols].drop_duplicates().merge(fixations[cols].drop_duplicates()) 

1632 if raw_gaze is not None and not raw_gaze.empty: 

1633 # The app's rule (`utils.combo_source`): a trial is listed when it has 

1634 # fixations — or, in a dataset without any, words — or when it has raw 

1635 # gaze. So a trial with words and samples but no fixations is listed, 

1636 # while one the intersection above drops for having fixations but no 

1637 # words stays dropped: its samples add nothing the rule is about. 

1638 known = _data.trial_keys(fixations if not fixations.empty else words) 

1639 samples = raw_gaze[cols].drop_duplicates() 

1640 extra = samples[ 

1641 [ 

1642 (str(p), str(t)) not in known 

1643 for p, t in zip(samples["participant_id"], samples["trial_id"]) 

1644 ] 

1645 ] 

1646 combos = pd.concat([combos, extra], ignore_index=True) 

1647 combos = combos.sort_values(cols).reset_index(drop=True) 

1648 return _named_out(combos, "trials", names.identity() if names else None) 

1649 

1650 

1651def list_parts( 

1652 words: pd.DataFrame | None, 

1653 fixations: pd.DataFrame | None, 

1654 participant: str | None = None, 

1655 trial: str | None = None, 

1656 *, 

1657 raw_gaze: pd.DataFrame | None = None, 

1658) -> pd.DataFrame: 

1659 """Ordered screens in multipart data, optionally narrowed to one parent. 

1660 

1661 Single-screen data returns an empty table. A trial recorded as raw gaze 

1662 alone takes its screens from ``raw_gaze`` (its ``screen_id``), decided per 

1663 trial — so a samples-only trial keeps its screens in a dataset whose other 

1664 trials have fixations. 

1665 """ 

1666 words, word_names = _named_in(words, "words", optional=True) 

1667 fixations, fix_names = _named_in(fixations, "fixations", optional=True) 

1668 gaze_names = None 

1669 if raw_gaze is not None: 

1670 raw_gaze, gaze_names = _named_in(raw_gaze, "raw_gaze") 

1671 names = _call_names(fixations=fix_names, words=word_names, raw_gaze=gaze_names) 

1672 catalog = part_catalog(words, fixations) 

1673 if raw_gaze is not None and SCREEN_ID in raw_gaze.columns: 

1674 # Per trial, as `_select_part` and the app decide it: a trial neither 

1675 # words nor fixations cover takes its screens from its samples. 

1676 samples = part_catalog(raw_gaze) 

1677 covered = _data.trial_keys(words) | _data.trial_keys(fixations) 

1678 own = [ 

1679 (str(p), str(t)) not in covered 

1680 for p, t in zip(samples["participant_id"], samples["trial_id"]) 

1681 ] 

1682 if any(own): 

1683 catalog = pd.concat([catalog, samples[own]], ignore_index=True) 

1684 if participant is not None or trial is not None: 

1685 # An id spelled before composite ids escaped a `_` inside a part. 

1686 participant, trial = ( 

1687 None if participant is None else str(participant), 

1688 None if trial is None else str(trial), 

1689 ) 

1690 respelled = _data.respell_reading( 

1691 participant or "", 

1692 trial or "", 

1693 zip(catalog["participant_id"], catalog["trial_id"], strict=True), 

1694 ) 

1695 participant = respelled[0] if participant is not None else None 

1696 trial = respelled[1] if trial is not None else None 

1697 if participant is not None: 

1698 catalog = catalog[catalog["participant_id"].astype(str) == str(participant)] 

1699 if trial is not None: 

1700 catalog = catalog[catalog["trial_id"].astype(str) == str(trial)] 

1701 return _named_out( 

1702 catalog.reset_index(drop=True), "parts", names.identity() if names else None 

1703 ) 

1704 

1705 

1706def _resolve_trial( 

1707 words: pd.DataFrame, 

1708 fixations: pd.DataFrame, 

1709 participant: str | None, 

1710 trial: str | None, 

1711 *, 

1712 default_first: bool = False, 

1713 raw_gaze: pd.DataFrame | None = None, 

1714) -> tuple[str, str]: 

1715 """Resolve to one (participant_id, trial_id), validating what was given. 

1716 

1717 A nonexistent participant/trial always raises — naming which of the two ids 

1718 is unknown, a few valid values and the closest spellings. An underspecified 

1719 selection matching several trials raises too, unless ``default_first`` picks 

1720 the first match (the CLI's behavior, mirroring the app's default selection). 

1721 ``raw_gaze`` makes the trials only its samples cover selectable. 

1722 """ 

1723 combos = _cn.to_canonical_frame(list_trials(words, fixations, raw_gaze=raw_gaze)) 

1724 if combos.empty: 

1725 raise ValueError( 

1726 "The data holds no trial with both a participant and a trial id." 

1727 ) 

1728 scoped = combos 

1729 # Ids written before composite ids escaped a `_` inside a part 

1730 # (`data.compose_id`) still find their reading when that is unambiguous. 

1731 readings = list(zip(combos["participant_id"], combos["trial_id"], strict=True)) 

1732 if participant is not None: 

1733 participant = _data.respell_reading( 

1734 participant, trial if trial is not None else "", readings 

1735 )[0] 

1736 scoped = scoped[scoped["participant_id"] == str(participant)] 

1737 if scoped.empty: 

1738 raise ValueError( 

1739 f"No trial matches participant={participant!r}: that participant " 

1740 f"id is not in the data. {_value_hint(combos, 'participant_id', participant)}" 

1741 ) 

1742 if trial is not None: 

1743 trial = _data.respell_reading( 

1744 participant if participant is not None else "", trial, readings 

1745 )[1] 

1746 narrowed = scoped[scoped["trial_id"] == str(trial)] 

1747 if narrowed.empty: 

1748 if participant is None: 

1749 raise ValueError( 

1750 f"No trial matches trial={trial!r}: that trial id is not in " 

1751 f"the data. {_value_hint(combos, 'trial_id', trial)}" 

1752 ) 

1753 raise ValueError( 

1754 f"No trial matches participant={participant!r}, trial={trial!r}: " 

1755 f"participant {str(participant)!r} has {len(scoped)} " 

1756 f"trial{'' if len(scoped) == 1 else 's'}, none of them {str(trial)!r}. {_value_hint(scoped, 'trial_id', trial)}" 

1757 ) 

1758 scoped = narrowed 

1759 if len(scoped) > 1 and not default_first: 

1760 preview = ", ".join( 

1761 f"({pid!r}, {tid!r})" 

1762 for pid, tid in scoped.head(5).itertuples(index=False, name=None) 

1763 ) 

1764 if participant is None and trial is None: 

1765 fix = "Pass participant= and trial=." 

1766 elif participant is None: 

1767 fix = ( 

1768 f"Trial {str(trial)!r} was read by {scoped['participant_id'].nunique()} " 

1769 "participants — pass participant= too." 

1770 ) 

1771 else: 

1772 fix = ( 

1773 f"Participant {str(participant)!r} has {len(scoped)} trials — pass " 

1774 "trial= too." 

1775 ) 

1776 raise ValueError( 

1777 f"Ambiguous selection: {len(scoped)} trials match " 

1778 f"participant={participant!r}, trial={trial!r} (first few: {preview}). " 

1779 f"{fix} list_trials(words, fixations) lists all " 

1780 f"{len(combos)} trials." 

1781 ) 

1782 row = scoped.iloc[0] 

1783 return str(row["participant_id"]), str(row["trial_id"]) 

1784 

1785 

1786def _value_hint(combos: pd.DataFrame, column: str, wanted, limit: int = 5) -> str: 

1787 """ "Closest / available ids" tail for a failed trial lookup.""" 

1788 values = [str(v) for v in combos[column].drop_duplicates()] 

1789 close = difflib.get_close_matches(str(wanted), values, n=3, cutoff=0.6) 

1790 shown = ", ".join(repr(v) for v in values[:limit]) 

1791 more = f", … (+{len(values) - limit} more)" if len(values) > limit else "" 

1792 hint = f"Available: {shown}{more}." 

1793 if close: 

1794 hint += f" Closest: {', '.join(repr(v) for v in close)}." 

1795 return hint 

1796 

1797 

1798def _select_trial( 

1799 words: pd.DataFrame, 

1800 fixations: pd.DataFrame, 

1801 participant: str | None, 

1802 trial: str | None, 

1803 *, 

1804 raw_gaze: pd.DataFrame | None = None, 

1805) -> tuple[pd.DataFrame, pd.DataFrame, str, str]: 

1806 pid, tid = _resolve_trial(words, fixations, participant, trial, raw_gaze=raw_gaze) 

1807 trial_words, trial_fixations = _data.filter_data( 

1808 words, fixations, {"participants": [pid], "trials": [tid]} 

1809 ) 

1810 if not trial_fixations.empty and trial_fixations["x"].isna().all(): 

1811 # AOI-sequence fixations whose coordinates couldn't be reconstructed: 

1812 # either no words table was given, or the word/AoI ids matched no box. 

1813 raise ValueError( 

1814 f"Fixations for participant={pid!r}, trial={tid!r} have no usable " 

1815 "coordinates. AOI-sequence datasets (no x/y) need a words table " 

1816 "whose word/AOI ids match the fixations', so each fixation can be " 

1817 "placed at its word box's center." 

1818 ) 

1819 return trial_words, trial_fixations, pid, tid 

1820 

1821 

1822def _select_part( 

1823 words: pd.DataFrame, 

1824 fixations: pd.DataFrame, 

1825 participant: str | None, 

1826 trial: str | None, 

1827 screen: str | None, 

1828 *, 

1829 raw_gaze: pd.DataFrame | None = None, 

1830 screen_param: str = "screen", 

1831) -> tuple[pd.DataFrame, pd.DataFrame, str, str, str | None]: 

1832 """Resolve one logical trial and, for multipart data, exactly one screen. 

1833 

1834 ``screen_param`` is the keyword the caller took the screen as, so an error 

1835 names the one to fix (``screen_b=`` for a comparison's second trial).""" 

1836 trial_words, trial_fixations, pid, tid = _select_trial( 

1837 words, fixations, participant, trial, raw_gaze=raw_gaze 

1838 ) 

1839 catalog = part_catalog(trial_words, trial_fixations) 

1840 if ( 

1841 catalog.empty 

1842 and trial_words.empty 

1843 and trial_fixations.empty 

1844 and raw_gaze is not None 

1845 and SCREEN_ID in raw_gaze.columns 

1846 ): 

1847 # VIZ-45: a trial recorded as raw gaze alone takes its screens from the 

1848 # samples, so one screen's coordinate space is drawn at a time — as for 

1849 # words and fixations — rather than every screen stacked into one. 

1850 catalog = part_catalog(_data.filter_raw_gaze(raw_gaze, [pid], [tid])) 

1851 if catalog.empty: 

1852 if screen is not None: 

1853 raise ValueError( 

1854 f"{screen_param}= names a screen, but participant={pid!r}, " 

1855 f"trial={tid!r} has only one; leave {screen_param}= out." 

1856 ) 

1857 return trial_words, trial_fixations, pid, tid, None 

1858 available = catalog[SCREEN_ID].astype(str).tolist() 

1859 selected = str(screen) if screen is not None else available[0] 

1860 if selected not in available: 

1861 raise ValueError( 

1862 f"Unknown {screen_param}={selected!r} for participant={pid!r}, " 

1863 f"trial={tid!r}. " 

1864 f"Available: {', '.join(repr(value) for value in available)}." 

1865 ) 

1866 return ( 

1867 extract_part(trial_words, pid, tid, selected), 

1868 extract_part(trial_fixations, pid, tid, selected), 

1869 pid, 

1870 tid, 

1871 selected, 

1872 ) 

1873 

1874 

1875def _apply_fix_index_range( 

1876 trial_fixations: pd.DataFrame, fix_index_range, pid: str, tid: str 

1877) -> pd.DataFrame: 

1878 """Window the trial to fixations ``start..end`` of ``order_in_trial``. 

1879 

1880 The headless form of the app's fixation-index slider: both bounds inclusive, 

1881 1-based, and applied only to the frame that feeds the figure. Raises rather 

1882 than silently drawing an empty scanpath when the window misses the trial.""" 

1883 if fix_index_range is None: 

1884 return trial_fixations 

1885 if not isinstance(fix_index_range, (tuple, list)) or len(fix_index_range) != 2: 

1886 raise ValueError( 

1887 f"fix_index_range must be a (start, end) pair of 1-based fixation " 

1888 f"indices, got {fix_index_range!r}." 

1889 ) 

1890 try: 

1891 lo, hi = int(fix_index_range[0]), int(fix_index_range[1]) 

1892 except (TypeError, ValueError) as exc: 

1893 raise ValueError( 

1894 f"fix_index_range bounds must be integers, got {fix_index_range!r}." 

1895 ) from exc 

1896 if lo > hi: 

1897 raise ValueError( 

1898 f"fix_index_range={fix_index_range!r} is empty: start {lo} is after end {hi}." 

1899 ) 

1900 if trial_fixations.empty: 

1901 return trial_fixations 

1902 if "order_in_trial" not in trial_fixations.columns: 

1903 raise ValueError( 

1904 "fix_index_range needs the 'order_in_trial' column, which " 

1905 "load_scanpath_data() adds during normalization — pass the frames it " 

1906 "returns." 

1907 ) 

1908 order = trial_fixations["order_in_trial"] 

1909 windowed = trial_fixations[(order >= lo) & (order <= hi)] 

1910 if windowed.empty: 

1911 raise ValueError( 

1912 f"fix_index_range=({lo}, {hi}) selects no fixations: participant={pid!r}, " 

1913 f"trial={tid!r} has {len(trial_fixations)} fixations " 

1914 f"(order_in_trial {int(order.min())}–{int(order.max())})." 

1915 ) 

1916 return windowed 

1917 

1918 

1919def _figure_kwargs(overrides: dict) -> dict: 

1920 settings = {**CANONICAL_FIGURE_DEFAULTS, **_expand_palette(overrides)} 

1921 if settings.get("heatmap_metric") == "counts": 

1922 settings["heatmap_metric"] = None 

1923 return settings 

1924 

1925 

1926_SPELLINGS = (("grey", "gray"), ("colour", "color")) 

1927 

1928 

1929def _spelling_variants(text: str) -> list[str]: 

1930 """``text`` as written, all-US and all-UK (#374: the palette names moved to 

1931 US spelling, and both spellings must keep naming the same palette).""" 

1932 import re 

1933 

1934 out = [text] 

1935 for pick in (1, 0): 

1936 variant = text 

1937 for pair in _SPELLINGS: 

1938 variant = re.sub(pair[1 - pick], pair[pick], variant, flags=re.IGNORECASE) 

1939 out.append(variant) 

1940 return list(dict.fromkeys(out)) 

1941 

1942 

1943def resolve_palette(value: object) -> str: 

1944 """The `constants.PALETTES` name ``value`` stands for: the app's name in 

1945 either spelling (``"Print / grayscale"`` or ``"Print / greyscale"``), the 

1946 short name (``"print"``), any case. Raises ``ValueError`` otherwise.""" 

1947 from .constants import PALETTES 

1948 

1949 error = None 

1950 for candidate in _spelling_variants(str(value)): 

1951 try: 

1952 name = normalize_palette(candidate) 

1953 except ValueError as exc: 

1954 error = error or exc 

1955 continue 

1956 if name in PALETTES: 

1957 return name 

1958 for spelled in _spelling_variants(name): 

1959 if spelled in PALETTES: 

1960 return spelled 

1961 raise error or ValueError(f"Unknown palette {value!r}.") 

1962 

1963 

1964def _check_colorscales(overrides: dict) -> None: 

1965 """#374: a name Plotly doesn't know raised its own lower-cased 

1966 ``PlotlyError`` from inside the builder; name the option instead.""" 

1967 from plotly.colors import get_colorscale 

1968 from plotly.exceptions import PlotlyError 

1969 

1970 for key in ("heatmap_colorscale", "fixation_colorscale"): 

1971 value = overrides.get(key) 

1972 if not isinstance(value, str): 

1973 continue 

1974 try: 

1975 get_colorscale(value) 

1976 except PlotlyError: 

1977 raise ValueError( 

1978 f"{key}={value!r} is not a Plotly color scale. Any named scale " 

1979 "works, e.g. Viridis, Greens, Blues, Cividis; append _r to " 

1980 "reverse one." 

1981 ) from None 

1982 

1983 

1984def _expand_palette(overrides: dict) -> dict: 

1985 """Expand a ``palette=`` override into the color kwargs it stands for. 

1986 

1987 ``palette`` names a set of color defaults tuned for a medium — screen, 

1988 colorblind viewers, a black & white print, a projector. It's a *preset*, so 

1989 any color the caller also passes explicitly wins over it:: 

1990 

1991 sps.plot_scanpath(w, f, palette="Print / grayscale") 

1992 sps.plot_scanpath(w, f, palette="Default (colorblind-safe)", saccade_color="#000") 

1993 

1994 The palette itself isn't a figure kwarg, so it's consumed here rather than 

1995 forwarded. Raises on an unknown name — a silent fallback to the default 

1996 palette would quietly produce the wrong figure for a print run. 

1997 

1998 Every enumerated option is read here too (`plots.normalize_option_values`): 

1999 ``heatmap_norm="log"`` is ``"Log"``, and a value that is none of the 

2000 choices raises rather than drawing the default. 

2001 """ 

2002 overrides = normalize_option_values(overrides) 

2003 _check_colorscales(overrides) 

2004 name = overrides.get("palette") 

2005 if name is None: 

2006 return overrides 

2007 name = resolve_palette(name) # "print", "high-contrast", any case or spelling 

2008 expanded = dict(overrides) 

2009 expanded.pop("palette") 

2010 # `word_label_color` is `text_color` on the figure builders. 

2011 settings = palette_settings(name) 

2012 settings["text_color"] = settings.pop("word_label_color") 

2013 for key, value in settings.items(): 

2014 expanded.setdefault(key, value) 

2015 return expanded 

2016 

2017 

2018_NAMED_FIGURE_PARAMS = frozenset( 

2019 { 

2020 "canvas_size", 

2021 "base_font_size", 

2022 "font_family", 

2023 "title", 

2024 "caption", 

2025 "screen", 

2026 "raw_gaze", 

2027 "illustration", 

2028 "illustration_label", 

2029 "fix_index_range", 

2030 "column_names", 

2031 } 

2032) 

2033 

2034 

2035def _reject_unknown_options(overrides: dict, valid, func_name: str) -> None: 

2036 """Fail on a misspelled/unsupported keyword, naming the closest valid ones. 

2037 

2038 Forwarding blindly would surface as ``make_scanpath_figure() got an 

2039 unexpected keyword argument`` — an internal name the caller never typed.""" 

2040 unknown = sorted(set(overrides) - set(valid)) 

2041 if not unknown: 

2042 return 

2043 parts = [] 

2044 for key in unknown: 

2045 # The builders' named parameters too: `canvas=` means `canvas_size=`. 

2046 close = difflib.get_close_matches( 

2047 key, sorted(set(valid) | _NAMED_FIGURE_PARAMS), n=3, cutoff=0.6 

2048 ) 

2049 suffix = ( 

2050 f" (did you mean {', '.join(repr(c) for c in close)}?)" if close else "" 

2051 ) 

2052 parts.append(f"{key!r}{suffix}") 

2053 raise TypeError( 

2054 f"{func_name}() got an unexpected keyword argument: {', '.join(parts)}. " 

2055 f"help({func_name}) lists its parameters and figure_options() the " 

2056 "figure options with their defaults." 

2057 ) 

2058 

2059 

2060#: Figure options whose value names a column → (the table it is read from, its 

2061#: CLI flag, the values that are not columns, what to do instead). EXP-17: the 

2062#: builders look the column up and draw *nothing* when it is missing, so a 

2063#: misspelling rendered a flat-coloured / unmarked figure without a word. 

2064_COLUMN_OPTIONS = { 

2065 "color_by": ( 

2066 "fixations", 

2067 "--color-by", 

2068 (UNIFORM_COLOR_FIELD, "line"), 

2069 f"Use {UNIFORM_COLOR_FIELD!r} for one flat color, 'line' to color by " 

2070 "text line, or one of the columns below.", 

2071 ), 

2072 "highlight_column": ( 

2073 "words", 

2074 "--highlight-column", 

2075 (), 

2076 "It names the boolean words column marking the text to highlight; pass " 

2077 "None ('' on the CLI) to highlight nothing.", 

2078 ), 

2079} 

2080 

2081 

2082#: DATA-66: the figure options whose value names one column, and those naming a 

2083#: list of them — each takes the dataset's own name as well as the internal one. 

2084_ONE_COLUMN_OPTIONS = ( 

2085 "color_by", 

2086 "highlight_column", 

2087 "heatmap_metric", 

2088 "word_hover_measure", 

2089 "word_heatmap_col", 

2090 "x_field", 

2091 "y_field", 

2092) 

2093_COLUMN_LIST_OPTIONS = ("word_hover_fields", "fixation_hover_fields") 

2094#: …and of those, the ones naming a column of the words table. 

2095_WORD_OPTIONS = frozenset( 

2096 ("highlight_column", "word_hover_measure", "word_heatmap_col", "word_hover_fields") 

2097) 

2098 

2099 

2100def _canonical_options( 

2101 overrides: dict, 

2102 names: ColumnNames | None, 

2103 *, 

2104 words: ColumnNames | None = None, 

2105) -> dict: 

2106 """``overrides`` with every column it names in the internal vocabulary — 

2107 a word option in the words table's names (``words``) before the merged 

2108 map's. 

2109 

2110 ``heatmap_metric`` is checked here too: the heatmap weights by the fixation 

2111 duration or counts fixations, and any other value used to count silently — 

2112 which, once the dataset's own names are accepted, a misspelled name would.""" 

2113 out = dict(overrides) 

2114 

2115 def canonical(option: str, value) -> str: 

2116 if words is not None and option in _WORD_OPTIONS: 

2117 found = words.to_canonical(value) 

2118 if found != str(value): 

2119 return found 

2120 return names.to_canonical(value) if names is not None else value 

2121 

2122 if names is not None or words is not None: 

2123 for option in _ONE_COLUMN_OPTIONS: 

2124 if isinstance(out.get(option), str): 

2125 out[option] = canonical(option, out[option]) 

2126 for option in _COLUMN_LIST_OPTIONS: 

2127 if out.get(option) is not None and not isinstance(out[option], str): 

2128 out[option] = [canonical(option, value) for value in out[option]] 

2129 metric = out.get("heatmap_metric") 

2130 if metric not in (None, "duration_ms", "counts"): 

2131 duration = ( 

2132 names.label("duration_ms") 

2133 if names is not None and names.source("duration_ms") 

2134 else "duration_ms" 

2135 ) 

2136 raise ValueError( 

2137 f"heatmap_metric={metric!r} (--heatmap-metric on the CLI) must be " 

2138 f"the fixation duration ({duration!r}) or 'counts'." 

2139 ) 

2140 return out 

2141 

2142 

2143def _column_labels( 

2144 names: ColumnNames | None, 

2145 word_frame, 

2146 fixation_frame, 

2147 *, 

2148 words: ColumnNames | None = None, 

2149) -> dict | None: 

2150 """`FigureSettings.column_labels` for frames that carried names — the 

2151 figure's text in the dataset's own names, as the app writes it, a word 

2152 column also under the words table's own name (`table_figure_labels`).""" 

2153 if names is None: 

2154 return None 

2155 labels = names.figure_labels( 

2156 [ 

2157 column 

2158 for frame in (word_frame, fixation_frame) 

2159 if frame is not None 

2160 for column in frame 

2161 ] 

2162 ) 

2163 if words is not None and word_frame is not None: 

2164 for column, label in words.figure_labels(word_frame.columns).items(): 

2165 if labels.get(column) != label: 

2166 labels[f"words:{column}"] = label 

2167 return labels 

2168 

2169 

2170def _check_column_options( 

2171 overrides: dict, *, words: pd.DataFrame, fixations: pd.DataFrame 

2172) -> None: 

2173 """Raise when an option the caller *named* points at no column. 

2174 

2175 Only explicit values are checked: ``highlight_column`` defaults to OneStop's 

2176 ``is_in_aspan``, which most corpora do not have and which the builder then 

2177 rightly skips. An empty table is not checked — there is nothing to color.""" 

2178 frames = {"words": words, "fixations": fixations} 

2179 for name, (kind, flag, synthetic, advice) in _COLUMN_OPTIONS.items(): 

2180 value = overrides.get(name) 

2181 if value is None or value == "" or value in synthetic: 

2182 continue 

2183 frame = frames[kind] 

2184 present = [str(column) for column in frame.columns] 

2185 if frame.empty or str(value) in present: 

2186 continue 

2187 # Internal helper columns (`_text_id_mapped`) are not the user's to name. 

2188 visible = [column for column in present if not column.startswith("_")] 

2189 close = difflib.get_close_matches(str(value), visible, n=3, cutoff=0.6) 

2190 hint = f" Closest: {', '.join(repr(c) for c in close)}." if close else "" 

2191 raise ValueError( 

2192 f"{name}={value!r} ({flag} on the CLI) names no column of the " 

2193 f"{kind} table.{hint} {advice} Columns present ({len(visible)}): " 

2194 f"{_column_preview(frame[visible])}." 

2195 ) 

2196 

2197 

2198def figure_options(kind: str = "static", *, choices: bool = False) -> dict: 

2199 """Every figure keyword a builder accepts → the default it renders with. 

2200 

2201 With ``choices=True`` each name maps to ``{"default": …, "choices": …}``, 

2202 where ``choices`` is the tuple of values an enumerated option takes 

2203 (``heatmap_norm``: ``("Linear", "Log")``) and ``None`` for a free one. An 

2204 enumerated option takes any spelling of a choice — case, spaces, ``-`` and 

2205 ``_`` are ignored, so the CLI's ``"log"`` and ``"mark-border"`` work — and 

2206 raises ``ValueError`` listing them for anything else. 

2207 

2208 ``kind="static"`` covers [`plot_scanpath`][scanpath_studio.api.plot_scanpath], 

2209 ``kind="animation"`` [`animate_scanpath`][scanpath_studio.api.animate_scanpath] 

2210 (whose builder supports a subset), and ``kind="comparison"`` 

2211 [`compare_scanpaths`][scanpath_studio.api.compare_scanpaths]. The values are the 

2212 defaults a call actually renders with (the app's Scanpath design) — so a scripted caller can diff its intended 

2213 settings against what it would get:: 

2214 

2215 {k: v for k, v in sps.figure_options().items() if k.startswith("show_")} 

2216 """ 

2217 if kind == "static": 

2218 params = _STATIC_FIGURE_PARAMS 

2219 defaults = FigureSettings.defaults(params) | CANONICAL_FIGURE_DEFAULTS 

2220 elif kind == "animation": 

2221 params = _ANIMATION_FIGURE_PARAMS 

2222 defaults = FigureSettings.defaults(params) | _animation_defaults() 

2223 elif kind == "comparison": 

2224 # CMP-9: `compare_scanpaths` validates against this set, and its TypeError 

2225 # points the caller here — so it has to be answerable. 

2226 params = _COMPARISON_FIGURE_PARAMS 

2227 defaults = FigureSettings.defaults(params) | { 

2228 key: value 

2229 for key, value in CANONICAL_FIGURE_DEFAULTS.items() 

2230 if key in _COMPARISON_FIGURE_PARAMS 

2231 } 

2232 else: 

2233 raise ValueError( 

2234 f"Unknown kind {kind!r}; use 'static', 'animation' or 'comparison'." 

2235 ) 

2236 options = {} 

2237 for name in sorted(params): 

2238 if name in defaults: 

2239 # Some public defaults are ordered field lists. Return an independent 

2240 # value so callers can edit the option reference without changing the 

2241 # canonical defaults used by every later plot. 

2242 options[name] = deepcopy(defaults[name]) 

2243 else: # pragma: no cover - every option is a FigureSettings field 

2244 options[name] = None 

2245 if choices: 

2246 return { 

2247 name: {"default": default, "choices": FIGURE_OPTION_CHOICES.get(name)} 

2248 for name, default in options.items() 

2249 } 

2250 return options 

2251 

2252 

2253def _animation_defaults() -> dict: 

2254 """The canonical defaults the animation builder can actually take.""" 

2255 return { 

2256 key: value 

2257 for key, value in CANONICAL_FIGURE_DEFAULTS.items() 

2258 if key in _ANIMATION_FIGURE_PARAMS 

2259 } 

2260 

2261 

2262def _apply_drift_correction( 

2263 trial_words: pd.DataFrame, 

2264 trial_fixations: pd.DataFrame, 

2265 settings: dict, 

2266 method: str | None, 

2267 connectors: bool, 

2268 explicit: dict, 

2269) -> pd.DataFrame: 

2270 """Snap fixations to their assigned text line, in place of the raw y. 

2271 

2272 Mirrors what the app does on the static plot (``tabs.render_single_trial_tab``): 

2273 run ``alignment.correct``, color the corrected fixations by line, and 

2274 optionally draw original→corrected connectors. Returns the fixations to plot. 

2275 """ 

2276 if method is None or str(method).lower() == "off": 

2277 return trial_fixations 

2278 # PRE-21: raise rather than ignore. A share link degrades silently because a 

2279 # human can see the figure and the rail; a script cannot, so quietly 

2280 # returning uncorrected fixations under a stated `drift_correction=` would 

2281 # be a wrong result with no signal. Name the env var so it is one step to fix. 

2282 if not drift_correction_enabled(): 

2283 raise ValueError( 

2284 "drift_correction is not available in this release (vertical " 

2285 f"drift correction is not fully integrated yet). Set " 

2286 f"{EXPERIMENTAL_ENV_VAR}=1 to enable it, or pass drift_correction=None." 

2287 ) 

2288 from . import alignment as _alignment # local: pulls in scipy 

2289 

2290 name = str(method).lower() 

2291 if name not in _alignment.ALGORITHMS: 

2292 raise ValueError( 

2293 f"Unknown drift_correction {method!r}; choose one of " 

2294 f"{', '.join(_alignment.ALGORITHMS)} (or None to leave the fixations " 

2295 "uncorrected)." 

2296 ) 

2297 if trial_fixations.empty or trial_words.empty: 

2298 return trial_fixations 

2299 original_y = tuple(pd.to_numeric(trial_fixations["y"], errors="coerce")) 

2300 corrected, _ = _alignment.correct(trial_fixations, trial_words, method=name) 

2301 # Colouring by line is what makes the correction legible; an explicit 

2302 # `color_by_line=` still wins. 

2303 if "color_by_line" not in explicit: 

2304 settings["color_by_line"] = True 

2305 if connectors: 

2306 settings["show_connectors"] = True 

2307 settings["connector_y"] = original_y 

2308 return corrected 

2309 

2310 

2311def _check_canvas_size(canvas_size) -> None: 

2312 """#374: ``canvas_size="1920x1080"`` was read character by character into a 

2313 1 x 9 px canvas and drew an empty figure.""" 

2314 if canvas_size is None: 

2315 return 

2316 try: 

2317 pair = not isinstance(canvas_size, str) and len(tuple(canvas_size)) == 2 

2318 except TypeError: 

2319 pair = False 

2320 if not pair: 

2321 raise ValueError( 

2322 "canvas_size must be a (width, height) pair in pixels, e.g. " 

2323 f"(2560, 1440); got {canvas_size!r} (--canvas WxH on the CLI)." 

2324 ) 

2325 

2326 

2327def plot_scanpath( 

2328 words: pd.DataFrame | None = None, 

2329 fixations: pd.DataFrame | None = None, 

2330 participant: str | None = None, 

2331 trial: str | None = None, 

2332 *, 

2333 screen: str | None = None, 

2334 canvas_size: tuple[int, int] | None = None, 

2335 base_font_size: int = 16, 

2336 font_family: str = FONT_FAMILY, 

2337 raw_gaze: pd.DataFrame | None = None, 

2338 drift_correction: str | None = None, 

2339 drift_connectors: bool = False, 

2340 fix_index_range: tuple[int, int] | None = None, 

2341 illustration: bool = False, 

2342 illustration_label: str = "auto", 

2343 title: str = "", 

2344 caption: str = "", 

2345 column_names: dict | None = None, 

2346 **figure_overrides, 

2347) -> go.Figure: 

2348 """Build one trial's scanpath figure (by default the app's Scanpath design). 

2349 

2350 ``words`` / ``fixations`` are normalized frames from 

2351 [`load_scanpath_data`][scanpath_studio.api.load_scanpath_data]. ``participant`` / 

2352 ``trial`` may be omitted when the frames hold exactly one trial. ``canvas_size`` 

2353 is the monitor size in px; by default it is estimated from the data extents — pass 

2354 the real monitor resolution (e.g. ``(2560, 1440)`` for OneStop) to keep coordinates 

2355 true to scale. For a multipart trial, ``screen`` selects one child screen; omitting 

2356 it selects the first recorded screen and never concatenates coordinate spaces. 

2357 ``raw_gaze`` is a frame from [`load_raw_gaze`][scanpath_studio.api.load_raw_gaze], 

2358 filtered to the selected trial and drawn as recorded. It can be the only table: 

2359 for a dataset recorded as raw gaze alone pass ``None`` for ``words`` and 

2360 ``fixations`` (``plot_scanpath(raw_gaze=samples, trial=…)``) — the trial is 

2361 looked up in the samples, the canvas is estimated from their extent, and the 

2362 figure is the samples alone. Nothing is derived from them: no fixations are 

2363 detected, so the fixation, saccade and heatmap layers stay empty. 

2364 

2365 ``drift_correction`` / ``drift_connectors`` are experimental: without 

2366 ``SCANPATH_EXPERIMENTAL=1`` any ``drift_correction`` other than ``None`` / 

2367 ``"off"`` raises ``ValueError``. 

2368 

2369 ``fix_index_range=(start, end)`` draws only fixations ``start`` 

2370 through ``end`` (1-based, both inclusive) of the trial — the headless form of 

2371 the app's fixation-index window. 

2372 

2373 ``title`` / ``caption`` stamp a title/caption band onto the figure 

2374 without shrinking the plot area, like the app's *Title & labels* — literal 

2375 text here, not the app's ``{trial_id}``-style pattern, since the caller 

2376 already knows which trial this is. 

2377 

2378 ``illustration=True`` applies the Illustration preset (snapped fixations, 

2379 arced saccades, uniform colors, no heatmap or word boxes); keywords you pass 

2380 still win. ``illustration_label`` is ``"auto"`` (label the figure when it no 

2381 longer shows the data as recorded), ``"show"`` or ``"hide"``. ``palette=`` 

2382 (``"default"``, ``"print"`` or ``"high-contrast"``, or the app's names) sets 

2383 a group of colors at once; a color you pass explicitly wins. 

2384 

2385 Remaining keywords override the app's defaults and are forwarded to 

2386 `plots.make_scanpath_figure` (e.g. ``show_heatmap=True``, 

2387 ``color_by="pass_index"``, ``x_field="order_in_trial"``); an unknown keyword raises 

2388 a ``TypeError`` naming the closest valid options, and 

2389 [`figure_options`][scanpath_studio.api.figure_options] lists them all with their 

2390 defaults (``choices=True`` adds the values each enumerated option takes; a 

2391 value is matched ignoring case, spaces, ``-`` and ``_``, and any other value 

2392 raises a ``ValueError``). A ``color_by`` / ``highlight_column`` naming a column the trial's table 

2393 doesn't have raises a ``ValueError`` naming the closest ones, rather than drawing 

2394 without it. 

2395 

2396 Frames under the dataset's own column names (what ``load_scanpath_data`` 

2397 returns by default) are read through the map they carry, and an option naming 

2398 a column takes either name. ``column_names`` is that map for frames loaded with 

2399 ``names="canonical"`` (``data.column_names``): the options then take the 

2400 dataset's names too, and the figure's text uses them. 

2401 """ 

2402 if illustration: 

2403 figure_overrides = { 

2404 "show_words": False, 

2405 "show_word_labels": True, 

2406 "show_fixations": True, 

2407 "show_order": False, 

2408 "show_saccades": True, 

2409 "show_saccade_arrows": False, 

2410 "show_heatmap": False, 

2411 "color_by": UNIFORM_COLOR_FIELD, 

2412 "saccade_color_mode": "Uniform", 

2413 "saccade_render_mode": "Arc", 

2414 "fixation_snap_to_word": True, 

2415 "fixation_opacity": 1.0, 

2416 **figure_overrides, 

2417 } 

2418 _reject_unknown_options( 

2419 figure_overrides, _STATIC_FIGURE_PARAMS | {"palette"}, "plot_scanpath" 

2420 ) 

2421 _check_canvas_size(canvas_size) 

2422 words, word_names = _named_in(words, "words", optional=True) 

2423 fixations, fix_names = _named_in(fixations, "fixations", optional=True) 

2424 gaze_names = None 

2425 if raw_gaze is not None: 

2426 raw_gaze, gaze_names = _named_in(raw_gaze, "raw_gaze") 

2427 names = _call_names( 

2428 column_names, fixations=fix_names, words=word_names, raw_gaze=gaze_names 

2429 ) 

2430 word_side = _table_names(column_names, "words", word_names) 

2431 figure_overrides = _canonical_options(figure_overrides, names, words=word_side) 

2432 trial_words, trial_fixations, pid, tid, selected_screen = _select_part( 

2433 words, fixations, participant, trial, screen, raw_gaze=raw_gaze 

2434 ) 

2435 if raw_gaze is not None: 

2436 raw_gaze = _data.filter_raw_gaze(raw_gaze, [pid], [tid]) 

2437 if selected_screen is not None and SCREEN_ID in raw_gaze.columns: 

2438 raw_gaze = extract_part(raw_gaze, pid, tid, selected_screen) 

2439 _check_column_options( 

2440 figure_overrides, words=trial_words, fixations=trial_fixations 

2441 ) 

2442 full_fix_range = None 

2443 if not trial_fixations.empty and "order_in_trial" in trial_fixations.columns: 

2444 order = pd.to_numeric( 

2445 trial_fixations["order_in_trial"], errors="coerce" 

2446 ).dropna() 

2447 if not order.empty: 

2448 full_fix_range = (int(order.min()), int(order.max())) 

2449 if canvas_size is None: 

2450 canvas_size = screen_canvas_size(trial_words) 

2451 if canvas_size is None: 

2452 canvas_size = screen_canvas_size(trial_fixations) 

2453 if canvas_size is None: 

2454 canvas_size = _recorded_screen(words, fixations) 

2455 if canvas_size is None: 

2456 # VIZ-45: a trial with no fixations is sized from its samples, as the 

2457 # app sizes a raw-gaze-only dataset's canvas. 

2458 canvas_size = _data.compute_canvas_size( 

2459 trial_words, 

2460 trial_fixations 

2461 if not trial_fixations.empty or raw_gaze is None 

2462 else raw_gaze, 

2463 ) 

2464 # Window first, correct second — the app's order (tabs._slice_fix_range runs 

2465 # before alignment.correct), so a windowed correction sees only the kept 

2466 # fixations. 

2467 trial_fixations = _apply_fix_index_range(trial_fixations, fix_index_range, pid, tid) 

2468 settings = _figure_kwargs(figure_overrides) 

2469 label_mode = str(illustration_label).capitalize() 

2470 if label_mode not in {"Auto", "Show", "Hide"}: 

2471 raise ValueError("illustration_label must be 'auto', 'show', or 'hide'.") 

2472 if "illustration_reasons" not in figure_overrides: 

2473 from .illustration import illustration_reasons, resolve_label_reasons 

2474 

2475 reasons = illustration_reasons( 

2476 settings, 

2477 fix_index_range=fix_index_range, 

2478 full_fixation_range=full_fix_range, 

2479 ) 

2480 settings["illustration_reasons"] = resolve_label_reasons(label_mode, reasons) 

2481 # Spatial fields are explicit kwargs of make_scanpath_figure, so they can't 

2482 # ride along in **settings without a "multiple values" TypeError. 

2483 x_field = settings.pop("x_field", "x") 

2484 y_field = settings.pop("y_field", "y") 

2485 trial_fixations = _apply_drift_correction( 

2486 trial_words, 

2487 trial_fixations, 

2488 settings, 

2489 drift_correction, 

2490 drift_connectors, 

2491 figure_overrides, 

2492 ) 

2493 if raw_gaze is not None: 

2494 settings.setdefault("show_raw_gaze", True) 

2495 render_settings = FigureSettings.from_mapping( 

2496 settings, 

2497 canvas_width=int(canvas_size[0]), 

2498 canvas_height=int(canvas_size[1]), 

2499 base_font_size=int(base_font_size), 

2500 font_family=font_family, 

2501 x_field=x_field, 

2502 y_field=y_field, 

2503 column_labels=_column_labels( 

2504 names, trial_words, trial_fixations, words=word_side 

2505 ), 

2506 ) 

2507 fig = make_scanpath_figure( 

2508 trial_words, 

2509 trial_fixations, 

2510 settings=render_settings, 

2511 raw_gaze=raw_gaze, 

2512 ) 

2513 annotate_figure(fig, title=title, caption=caption) 

2514 return fig 

2515 

2516 

2517def animate_scanpath( 

2518 words: pd.DataFrame | None = None, 

2519 fixations: pd.DataFrame | None = None, 

2520 participant: str | None = None, 

2521 trial: str | None = None, 

2522 *, 

2523 screen: str | None = None, 

2524 screen_b: str | None = None, 

2525 canvas_size: tuple[int, int] | None = None, 

2526 base_font_size: int = 16, 

2527 font_family: str = FONT_FAMILY, 

2528 playback_speed: float = 1.0, 

2529 autoplay: bool = True, 

2530 fix_index_range: tuple[int, int] | None = None, 

2531 fix_index_range_b: tuple[int, int] | None = None, 

2532 illustration_label: str = "auto", 

2533 title: str = "", 

2534 caption: str = "", 

2535 column_names: dict | None = None, 

2536 trial_b: tuple[str, str] | None = None, 

2537 dataset_b: str | None = None, 

2538 setup: SetupSnapshot | None = None, 

2539 setup_b: SetupSnapshot | None = None, 

2540 **animation_overrides, 

2541) -> go.Figure: 

2542 """Build the animated scanpath replay for one trial. 

2543 

2544 Same trial selection, canvas and column-name semantics as 

2545 [`plot_scanpath`][scanpath_studio.api.plot_scanpath] (``column_names`` included), 

2546 and ``screen`` selection for multipart trials. The replay takes the reading time divided by 

2547 ``playback_speed``: save it as interactive HTML with 

2548 [`save_figure`][scanpath_studio.api.save_figure], whose page keeps that clock 

2549 itself, or rasterize it to GIF/MP4 with `animation_export.export_animation`, which 

2550 lasts as long. (`fig.show()` plays it on Plotly's own frame queue, which runs 

2551 slow.) ``fix_index_range=(start, end)`` replays only that window of the trial's 

2552 fixations (1-based, inclusive), like 

2553 [`plot_scanpath`][scanpath_studio.api.plot_scanpath]. 

2554 

2555 With ``autoplay`` (default ``True``) the saved interactive HTML auto-starts 

2556 the replay on load *at ``playback_speed``* — 

2557 [`save_figure`][scanpath_studio.api.save_figure] honors the marker the builder 

2558 stamps on the figure. Pass ``autoplay=False`` to save a figure that opens paused 

2559 (press ▶ Play to run it). Autoplay only affects the interactive HTML; a GIF/MP4 

2560 always plays from its first frame. 

2561 

2562 When ``playback_speed`` is not ``1``, the automatic Illustration label says 

2563 the replay timing was changed. ``illustration_label`` accepts ``"auto"``, 

2564 ``"show"``, or ``"hide"`` like [`plot_scanpath`][scanpath_studio.api.plot_scanpath]. 

2565 

2566 In a co-animation ``fix_index_range`` windows A only (the app's 

2567 rule — A's slider never cuts B), ``fix_index_range_b`` windows B, and 

2568 ``fixation_flags_b`` gives B flags of its own (``None``: A's 

2569 ``fixation_flags``, or the ``fixation_flags`` of ``style_b`` when it 

2570 names some). 

2571 

2572 ``style_a`` / ``style_b`` style the two scanpaths of a co-animation as they 

2573 style [`compare_scanpaths`][scanpath_studio.api.compare_scanpaths]' — the 

2574 same keys (``fix_color``, ``marker_size_range``, ``opacity``, ``hollow``, 

2575 ``saccade_color``, ``saccade_style``, ``saccade_width``), resolved the same 

2576 way, so the replay and the static comparison draw each trial alike. The 

2577 replay has no saccade-class filter, so a style naming ``saccade_classes`` 

2578 raises ``ValueError``. A lone replay ignores both. 

2579 

2580 ``trial_b=(participant, trial)`` co-animates a second trial on the same 

2581 clock, like the app's Animate + Compare. It is looked up in ``words_b`` / 

2582 ``fixations_b`` when given, else in ``words`` / ``fixations`` — the way 

2583 [`compare_scanpaths`][scanpath_studio.api.compare_scanpaths] takes it. 

2584 Without ``trial_b``, ``words_b`` / ``fixations_b`` must hold one trial; B 

2585 frames holding several raise ``ValueError`` rather than drawing them all. A 

2586 multipart B is drawn at ``screen_b`` — looked up in B's own trial — or at 

2587 its first recorded screen without it, as A is with ``screen``. 

2588 

2589 **Two datasets.** Both readings are drawn in A's coordinates, so a 

2590 co-animation is an overlay, and a trial from another dataset has to share 

2591 A's screen. Name that dataset with ``dataset_b`` (or give its ``setup_b``) 

2592 and the pair is checked the way `compare_scanpaths` checks an overlay: two 

2593 different canvases raise ``IncomparableScreensError``, a ``ValueError``, 

2594 rather than draw. ``setup`` / ``setup_b`` are 

2595 `experimental_setup.SetupSnapshot` values; a side without one is read off 

2596 its data — the extent of that one trial, which rarely spans the whole 

2597 screen, so state both when you know them — and ``canvas_size`` covers A 

2598 when you only have a resolution. ``dataset_b`` also prefixes B's 

2599 participant ids with the dataset's name, as `compare_scanpaths` does, so a 

2600 hover says whose participant it is. ``words_b`` / ``fixations_b`` passed without 

2601 either are taken to be from A's dataset, as `render` passes them for 

2602 ``--compare-with`` alone, and are not checked: two readings of one corpus 

2603 can span different extents, and inferring a canvas from each would refuse 

2604 pairs that shared a screen. 

2605 

2606 The animation builder accepts a subset of the static figure's options 

2607 (``show_words``, ``show_word_labels``, ``show_saccades``, ``show_order``, styling, 

2608 and second-scanpath overlays) — see ``figure_options("animation")``; an unsupported 

2609 key raises a ``ValueError`` naming the valid ones. The shared options default to the same values as 

2610 [`plot_scanpath`][scanpath_studio.api.plot_scanpath] (`CANONICAL_FIGURE_DEFAULTS`), 

2611 so the replay matches the static figure. ``palette=`` works here too; the 

2612 colors it implies that the animation doesn't support are dropped rather than 

2613 raising, since the caller named a look, not those individual keys. 

2614 

2615 ``title`` / ``caption`` — same as 

2616 [`plot_scanpath`][scanpath_studio.api.plot_scanpath]. 

2617 

2618 The replay is made of fixations, so a trial without any — one recorded as 

2619 raw gaze alone, or a words-only one — raises ``ValueError`` rather than 

2620 returning an empty replay, and ``raw_gaze=`` is refused: the replay draws no 

2621 raw-gaze layer, and nothing detects fixations from samples. Draw samples with 

2622 [`plot_scanpath`][scanpath_studio.api.plot_scanpath]`(raw_gaze=…)`. 

2623 """ 

2624 _check_canvas_size(canvas_size) 

2625 if "raw_gaze" in animation_overrides: 

2626 raise ValueError( 

2627 "animate_scanpath replays fixations and has no raw-gaze layer, and " 

2628 "Scanpath Studio does not detect fixations from gaze samples. Draw the " 

2629 "samples with plot_scanpath(..., raw_gaze=...) instead." 

2630 ) 

2631 valid = set(_ANIMATION_FIGURE_PARAMS) 

2632 explicit = set(animation_overrides) - {"palette"} 

2633 animation_overrides = _expand_palette(animation_overrides) 

2634 # Only the keys the caller named are held to the "is this supported?" rule; 

2635 # a palette's extras (heatmap colorscale, highlight text colour, …) that the 

2636 # animation has no parameter for are simply dropped. 

2637 animation_overrides = { 

2638 k: v for k, v in animation_overrides.items() if k in valid or k in explicit 

2639 } 

2640 unknown = explicit - valid 

2641 if unknown: 

2642 raise ValueError( 

2643 f"animate_scanpath() does not support: {', '.join(sorted(unknown))}. " 

2644 "figure_options('animation') lists the options it takes." 

2645 ) 

2646 for side in ("style_a", "style_b"): 

2647 style = animation_overrides.get(side) 

2648 if isinstance(style, dict) and style.get("saccade_classes") is not None: 

2649 raise ValueError( 

2650 f"{side}['saccade_classes'] filters a comparison figure's " 

2651 "saccades; the co-animation has no saccade-class filter. Drop " 

2652 "it, or draw the pair with compare_scanpaths." 

2653 ) 

2654 named = {k: v for k, v in animation_overrides.items() if k in explicit} 

2655 # Same defaults as the static figure for every option both builders share, so 

2656 # `plot_scanpath` and `animate_scanpath` don't render the same trial 

2657 # differently (the app feeds both from one settings dict). 

2658 animation_overrides = {**_animation_defaults(), **animation_overrides} 

2659 words, word_names = _named_in(words, "words", optional=True) 

2660 fixations, fix_names = _named_in(fixations, "fixations", optional=True) 

2661 names = _call_names(column_names, fixations=fix_names, words=word_names) 

2662 word_side = _table_names(column_names, "words", word_names) 

2663 animation_overrides = _canonical_options( 

2664 animation_overrides, names, words=word_side 

2665 ) 

2666 named = _canonical_options(named, names, words=word_side) 

2667 trial_words, trial_fixations, pid, tid, _selected_screen = _select_part( 

2668 words, fixations, participant, trial, screen 

2669 ) 

2670 if trial_fixations.empty: 

2671 raise ValueError( 

2672 f"participant={pid!r}, trial={tid!r} has no fixations to replay — the " 

2673 "replay is built from fixations. A trial recorded as raw gaze alone " 

2674 "can be drawn with plot_scanpath(..., raw_gaze=...); its samples are " 

2675 "not turned into fixations." 

2676 ) 

2677 _check_column_options(named, words=trial_words, fixations=trial_fixations) 

2678 full_fix_range = None 

2679 if not trial_fixations.empty and "order_in_trial" in trial_fixations.columns: 

2680 full_order = pd.to_numeric( 

2681 trial_fixations["order_in_trial"], errors="coerce" 

2682 ).dropna() 

2683 if not full_order.empty: 

2684 full_fix_range = (int(full_order.min()), int(full_order.max())) 

2685 trial_fixations = _apply_fix_index_range(trial_fixations, fix_index_range, pid, tid) 

2686 # A's screen: a stated `setup`, else `canvas_size`, else read off the data — 

2687 # the order `compare_scanpaths` resolves it in, and what CMP-21's gate reads. 

2688 setup_a = _compare_setup( 

2689 setup, canvas_size, trial_words, trial_fixations, side="setup" 

2690 ) 

2691 passed_b = ( 

2692 animation_overrides.pop("words_b", None), 

2693 animation_overrides.pop("fixations_b", None), 

2694 ) 

2695 second_dataset = dataset_b is not None or setup_b is not None 

2696 if second_dataset and all(frame is None for frame in passed_b): 

2697 raise ValueError( 

2698 "dataset_b / setup_b describe scanpath B's own dataset, but neither " 

2699 "words_b nor fixations_b was passed. Pass B's frames too, or leave " 

2700 "both out to draw trial_b from these frames." 

2701 ) 

2702 if screen_b is not None and trial_b is None and all(f is None for f in passed_b): 

2703 raise ValueError( 

2704 "screen_b= picks scanpath B's screen, but there is no scanpath B. " 

2705 "Pass trial_b=(participant, trial) too." 

2706 ) 

2707 words_b, fixations_b = _second_reading( 

2708 words, fixations, *passed_b, trial_b, screen_b=screen_b 

2709 ) 

2710 if second_dataset and fixations_b is not None and not fixations_b.empty: 

2711 _refuse_co_animation_across_screens( 

2712 setup_a, 

2713 setup_b, 

2714 words_b, 

2715 fixations_b, 

2716 a_inferred=setup is None and canvas_size is None, 

2717 ) 

2718 elif fixations_b is not None and not fixations_b.empty: 

2719 # One dataset, two screen sizes it knows of: refused as the app and 

2720 # `compare_scanpaths`' overlay refuse them. 

2721 same_b = _same_dataset_setup_b( 

2722 setup_a, 

2723 setup_b, 

2724 a_known=setup is not None or canvas_size is not None, 

2725 words_a=trial_words, 

2726 fixations_a=trial_fixations, 

2727 words_b=words_b, 

2728 fixations_b=fixations_b, 

2729 ) 

2730 if same_b is not None: 

2731 _refuse_co_animation_across_screens( 

2732 setup_a, same_b, words_b, fixations_b, a_inferred=False 

2733 ) 

2734 full_fix_range_b = None 

2735 if ( 

2736 fixations_b is not None 

2737 and not fixations_b.empty 

2738 and "order_in_trial" in fixations_b.columns 

2739 ): 

2740 order_b = pd.to_numeric(fixations_b["order_in_trial"], errors="coerce").dropna() 

2741 if not order_b.empty: 

2742 full_fix_range_b = (int(order_b.min()), int(order_b.max())) 

2743 if fix_index_range_b is not None and fixations_b is not None: 

2744 pid_b, tid_b = (str(v) for v in (trial_b or ("B", "B"))) 

2745 fixations_b = _apply_fix_index_range( 

2746 fixations_b, fix_index_range_b, pid_b, tid_b 

2747 ) 

2748 if dataset_b is not None: 

2749 # As `compare_scanpaths` and the app do: B's readers carry their 

2750 # dataset's name, so a hover says whose reader it is. 

2751 from .utils import qualify_for_compare 

2752 

2753 words_b, fixations_b = ( 

2754 None if frame is None else qualify_for_compare(frame, dataset_b) 

2755 for frame in (words_b, fixations_b) 

2756 ) 

2757 label_mode = str(illustration_label).capitalize() 

2758 if label_mode not in {"Auto", "Show", "Hide"}: 

2759 raise ValueError("illustration_label must be 'auto', 'show', or 'hide'.") 

2760 if "illustration_reasons" not in animation_overrides: 

2761 from .illustration import illustration_reasons, resolve_label_reasons 

2762 

2763 reasons = illustration_reasons( 

2764 {**animation_overrides, "playback_speed": playback_speed}, 

2765 fix_index_range=fix_index_range, 

2766 full_fixation_range=full_fix_range, 

2767 # CMP-24: B's own flags and window, when it co-animates. 

2768 fixation_flags_b=animation_overrides.get("fixation_flags_b") 

2769 if fixations_b is not None 

2770 else None, 

2771 fix_index_range_b=fix_index_range_b, 

2772 full_fixation_range_b=full_fix_range_b, 

2773 ) 

2774 animation_overrides["illustration_reasons"] = resolve_label_reasons( 

2775 label_mode, reasons 

2776 ) 

2777 render_settings = FigureSettings.from_mapping( 

2778 animation_overrides, 

2779 canvas_width=int(setup_a.canvas_width), 

2780 canvas_height=int(setup_a.canvas_height), 

2781 base_font_size=int(base_font_size), 

2782 font_family=font_family, 

2783 playback_speed=playback_speed, 

2784 autoplay=autoplay, 

2785 column_labels=_column_labels( 

2786 names, trial_words, trial_fixations, words=word_side 

2787 ), 

2788 ) 

2789 fig = make_scanpath_animation( 

2790 trial_words, 

2791 trial_fixations, 

2792 settings=render_settings, 

2793 fixations_b=fixations_b, 

2794 words_b=words_b, 

2795 ) 

2796 add_illustration_label( 

2797 fig, 

2798 animation_overrides.get("illustration_reasons"), 

2799 text=render_settings.illustration_text, 

2800 ) 

2801 annotate_figure(fig, title=title, caption=caption) 

2802 return fig 

2803 

2804 

2805def _second_reading( 

2806 words: pd.DataFrame, 

2807 fixations: pd.DataFrame, 

2808 words_b: pd.DataFrame | None, 

2809 fixations_b: pd.DataFrame | None, 

2810 trial_b: tuple[str, str] | None, 

2811 *, 

2812 screen_b: str | None = None, 

2813) -> tuple[pd.DataFrame | None, pd.DataFrame | None]: 

2814 """Scanpath B's frames for a co-animation, cut to one reading. 

2815 

2816 The animation builder draws every row it is handed, so frames passed the 

2817 way `compare_scanpaths` takes them — B's whole corpus — drew every fixation 

2818 in it. ``trial_b`` picks the reading, in B's own frames when given and A's 

2819 otherwise, as `compare_scanpaths` does; without it B's frames must hold one 

2820 trial, since guessing among several would draw somebody else's reading. 

2821 A multipart B keeps one screen, never all of them — each is its own 

2822 coordinate space: ``screen_b``, else its first, as A without ``screen=`` 

2823 and the app's B navigator start. 

2824 """ 

2825 if trial_b is None: 

2826 source = fixations_b if fixations_b is not None else words_b 

2827 if source is None or source.empty: 

2828 return words_b, fixations_b 

2829 label = "fixations_b" if fixations_b is not None else "words_b" 

2830 pairs = _require_normalized(source, label)[ 

2831 ["participant_id", "trial_id"] 

2832 ].drop_duplicates() 

2833 if len(pairs) > 1: 

2834 raise ValueError( 

2835 f"{label} holds {len(pairs)} trials, so the second scanpath is " 

2836 "ambiguous. Pass trial_b=(participant, trial) to pick one — " 

2837 "compare_scanpaths takes it the same way." 

2838 ) 

2839 pid_b, tid_b = (str(value) for value in pairs.iloc[0]) 

2840 else: 

2841 pid_b, tid_b = str(trial_b[0]), str(trial_b[1]) 

2842 words_b = words if words_b is None else words_b 

2843 fixations_b = fixations if fixations_b is None else fixations_b 

2844 

2845 def one_reading(frame: pd.DataFrame | None, label: str) -> pd.DataFrame | None: 

2846 # `extract_part` masks afresh, as `_select_trial` slices A. Not 

2847 # `utils.extract_trial`: its position cache is keyed by the frame's 

2848 # identity, so an in-place edit between two calls handed back somebody 

2849 # else's rows. 

2850 if frame is None or frame.empty: 

2851 return frame 

2852 return extract_part(_require_normalized(frame, label), pid_b, tid_b) 

2853 

2854 trial_words_b = one_reading(words_b, "words_b") 

2855 trial_fix_b = one_reading(fixations_b, "fixations_b") 

2856 if trial_b is not None and (trial_fix_b is None or trial_fix_b.empty): 

2857 raise ValueError( 

2858 f"No fixations for the second scanpath participant={pid_b!r}, " 

2859 f"trial={tid_b!r}. list_trials() shows what the frames contain." 

2860 ) 

2861 catalog = part_catalog(trial_words_b, trial_fix_b) 

2862 if catalog.empty: 

2863 if screen_b is not None: 

2864 raise ValueError( 

2865 "screen_b= was supplied for a single-screen trial " 

2866 f"(participant={pid_b!r}, trial={tid_b!r})." 

2867 ) 

2868 else: 

2869 available = catalog[SCREEN_ID].astype(str).tolist() 

2870 if screen_b is not None and str(screen_b) not in available: 

2871 raise ValueError( 

2872 f"Unknown screen_b={str(screen_b)!r} for participant={pid_b!r}, " 

2873 f"trial={tid_b!r}. " 

2874 f"Available: {', '.join(repr(value) for value in available)}." 

2875 ) 

2876 screen_b = str(screen_b) if screen_b is not None else available[0] 

2877 trial_words_b, trial_fix_b = ( 

2878 extract_part(frame, pid_b, tid_b, screen_b) 

2879 if frame is not None and SCREEN_ID in frame.columns 

2880 else frame 

2881 for frame in (trial_words_b, trial_fix_b) 

2882 ) 

2883 return trial_words_b, trial_fix_b 

2884 

2885 

2886def _inferred_screen_hint(*, a_inferred: bool, b_inferred: bool) -> str: 

2887 """How to state a screen that a refusal only read off the data. 

2888 

2889 `setups_comparable` says the readings were *recorded* on different screens, 

2890 but a screen nobody stated is the extent of that trial's data, which rarely 

2891 spans the whole display — so the refusal names the parameter that states it. 

2892 """ 

2893 if a_inferred and b_inferred: 

2894 return ( 

2895 " Neither screen was stated, so both were read off the data, which " 

2896 "rarely spans the whole screen; if they were shown on one, pass it as " 

2897 "setup= and setup_b=." 

2898 ) 

2899 if a_inferred: 

2900 return ( 

2901 " A's screen was read off its data, which rarely spans the whole " 

2902 "screen; if both were shown on one, pass A's as setup= or canvas_size=." 

2903 ) 

2904 if b_inferred: 

2905 return ( 

2906 " B's screen was read off its data, which rarely spans the whole " 

2907 "screen; if both were shown on one, pass B's as setup_b=." 

2908 ) 

2909 return "" 

2910 

2911 

2912def _same_dataset_setup_b( 

2913 setup_a: SetupSnapshot, 

2914 setup_b: SetupSnapshot | None, 

2915 *, 

2916 a_known: bool, 

2917 words_a: pd.DataFrame | None, 

2918 fixations_a: pd.DataFrame | None, 

2919 words_b: pd.DataFrame | None, 

2920 fixations_b: pd.DataFrame | None, 

2921) -> SetupSnapshot | None: 

2922 """B's screen when it is known to differ from A's, within one dataset. 

2923 

2924 One dataset can hold screens of different sizes, so a same-dataset pair is 

2925 gated too — but only on screens either side actually *knows*: a stated 

2926 setup, or the selected screen's own canvas columns. Two data extents say 

2927 nothing (two readings of one screen rarely span the same area). Returns 

2928 B's snapshot when both are known and the canvases differ — the pair an 

2929 overlay or co-animation must refuse — else ``None``. ``a_known`` is whether 

2930 the caller stated A's screen. 

2931 """ 

2932 own_b = screen_canvas_size(words_b) or screen_canvas_size(fixations_b) 

2933 a_known = ( 

2934 a_known 

2935 or screen_canvas_size(words_a) is not None 

2936 or screen_canvas_size(fixations_a) is not None 

2937 ) 

2938 if setup_b is not None: 

2939 resolved = setup_b 

2940 elif own_b is not None: 

2941 resolved = replace( 

2942 setup_a, canvas_width=int(own_b[0]), canvas_height=int(own_b[1]) 

2943 ) 

2944 else: 

2945 return None 

2946 if not a_known or resolved.canvas == setup_a.canvas: 

2947 return None 

2948 return resolved 

2949 

2950 

2951def _refuse_co_animation_across_screens( 

2952 setup_a: SetupSnapshot, 

2953 setup_b: SetupSnapshot | None, 

2954 words_b: pd.DataFrame | None, 

2955 fixations_b: pd.DataFrame, 

2956 *, 

2957 a_inferred: bool, 

2958) -> None: 

2959 """Refuse a co-animation of two datasets shown on different screens. 

2960 

2961 A co-animation draws both readings on one clock in A's coordinates, which 

2962 makes it an overlay, so it is held to `compare_scanpaths`'s overlay gate: the 

2963 same `setups_comparable` predicate and the same error, with B's screen read 

2964 off its data when the caller did not state it. 

2965 """ 

2966 from .experimental_setup import IncomparableScreensError, setups_comparable 

2967 

2968 resolved_b = _compare_setup( 

2969 setup_b, 

2970 None, 

2971 words_b if words_b is not None else pd.DataFrame(), 

2972 fixations_b, 

2973 side="setup_b", 

2974 ) 

2975 comparable, note = setups_comparable(setup_a, resolved_b) 

2976 if not comparable: 

2977 hint = _inferred_screen_hint(a_inferred=a_inferred, b_inferred=setup_b is None) 

2978 raise IncomparableScreensError( 

2979 f"{note} A co-animation replays both trials on one clock in one " 

2980 "coordinate space, so none was drawn; compare them with " 

2981 "compare_scanpaths(layout='side_by_side') (or 'stacked'), each drawn " 

2982 f"to its own screen.{hint}", 

2983 reason=note, 

2984 ) 

2985 if note: 

2986 # As in `compare_scanpaths`: matching canvases, but at least one screen 

2987 # was never recorded — drawn, with the caveat where a script can see it. 

2988 logging.getLogger(__name__).warning("animate_scanpath: %s", note) 

2989 

2990 

2991def render_parent_trial( 

2992 words: pd.DataFrame, 

2993 fixations: pd.DataFrame, 

2994 participant: str | None = None, 

2995 trial: str | None = None, 

2996 *, 

2997 animate: bool = False, 

2998 transition_mode: str = "instant", 

2999 screens: Sequence[str] | None = None, 

3000 **options, 

3001) -> dict[str, go.Figure]: 

3002 """Render every screen of one logical trial without stitching coordinates. 

3003 

3004 ``screens`` renders only those screen ids (one id may be given as a 

3005 string), in the trial's own order, as the app's Export → *Screens* does; 

3006 an id the trial does not have raises ``ValueError``. ``None`` renders them 

3007 all. ``screen_index`` in each figure's meta stays the screen's place in the 

3008 trial. 

3009 

3010 The ordered mapping is keyed by ``screen_id``. Each value is the same figure 

3011 returned by [`plot_scanpath`][scanpath_studio.api.plot_scanpath] or 

3012 [`animate_scanpath`][scanpath_studio.api.animate_scanpath]; callers can save them 

3013 into deterministic per-screen files. ``transition_mode`` is ``"instant"`` or 

3014 ``"recorded"``. For animated output, each figure's 

3015 ``layout.meta['transition_after_ms']`` records the delay before the next screen 

3016 (zero for instant mode, or the observed parent-clock gap). No visual saccade is ever 

3017 drawn across the boundary. 

3018 """ 

3019 if transition_mode not in {"instant", "recorded"}: 

3020 raise ValueError("transition_mode must be 'instant' or 'recorded'.") 

3021 raw_gaze = options.get("raw_gaze") 

3022 pid, tid = _resolve_trial( 

3023 _cn.to_canonical_frame(words), 

3024 _cn.to_canonical_frame(fixations), 

3025 participant, 

3026 trial, 

3027 raw_gaze=_cn.to_canonical_frame(raw_gaze), 

3028 ) 

3029 # Read here under the internal names; the renderers take the frames as 

3030 # given and name their figures as they were named (DATA-66). 

3031 catalog = _cn.to_canonical_frame( 

3032 list_parts(words, fixations, pid, tid, raw_gaze=raw_gaze) 

3033 ) 

3034 canonical_fixations = _cn.to_canonical_frame(fixations) 

3035 if catalog.empty: 

3036 if screens is not None: 

3037 raise ValueError( 

3038 f"Trial {tid!r} of participant {pid!r} has no screens to choose from." 

3039 ) 

3040 renderer = animate_scanpath if animate else plot_scanpath 

3041 return {"screen-1": renderer(words, fixations, pid, tid, **options)} 

3042 

3043 screen_ids = catalog[SCREEN_ID].astype(str).tolist() 

3044 if isinstance(screens, str): 

3045 screens = [screens] 

3046 if screens is not None: 

3047 missing = [str(s) for s in screens if str(s) not in screen_ids] 

3048 if missing: 

3049 raise ValueError( 

3050 f"Trial {tid!r} of participant {pid!r} has no screen " 

3051 f"{', '.join(map(repr, missing))}; its screens are " 

3052 f"{', '.join(screen_ids)}." 

3053 ) 

3054 chosen = None if screens is None else {str(s) for s in screens} 

3055 # The screens drawn, in the trial's order; a recorded transition runs to 

3056 # the next one *drawn*, and the last drawn has none. 

3057 drawn = [s for s in screen_ids if chosen is None or s in chosen] 

3058 rendered: dict[str, go.Figure] = {} 

3059 for position, screen_id in enumerate(drawn): 

3060 renderer = animate_scanpath if animate else plot_scanpath 

3061 fig = renderer(words, fixations, pid, tid, screen=screen_id, **options) 

3062 delay = 0.0 

3063 if animate and transition_mode == "recorded" and position < len(drawn) - 1: 

3064 current = extract_part(canonical_fixations, pid, tid, screen_id) 

3065 following = extract_part(canonical_fixations, pid, tid, drawn[position + 1]) 

3066 if not current.empty and not following.empty: 

3067 current_end = ( 

3068 pd.to_numeric(current["timestamp_ms"], errors="coerce") 

3069 + pd.to_numeric(current["duration_ms"], errors="coerce").fillna(0) 

3070 ).max() 

3071 next_start = pd.to_numeric( 

3072 following["timestamp_ms"], errors="coerce" 

3073 ).min() 

3074 if pd.notna(current_end) and pd.notna(next_start): 

3075 delay = max(0.0, float(next_start - current_end)) 

3076 existing_meta = fig.layout.meta if isinstance(fig.layout.meta, dict) else {} 

3077 fig.update_layout( 

3078 meta={ 

3079 **existing_meta, 

3080 "participant_id": pid, 

3081 "trial_id": tid, 

3082 "screen_id": screen_id, 

3083 "screen_index": screen_ids.index(screen_id) + 1, 

3084 "transition_mode": transition_mode, 

3085 "transition_after_ms": delay, 

3086 } 

3087 ) 

3088 rendered[screen_id] = fig 

3089 return rendered 

3090 

3091 

3092#: Layout names `compare_scanpaths` accepts, mapped to the builder's spelling. 

3093#: Hyphens are accepted so `cli.render --compare-layout side-by-side`, the share 

3094#: link's `cmp_layout`, and this function all name the layout the same way. 

3095_COMPARE_LAYOUTS = { 

3096 "overlay": "overlay", 

3097 "side_by_side": "side_by_side", 

3098 "side-by-side": "side_by_side", 

3099 "stacked": "stacked", 

3100} 

3101 

3102 

3103def _compare_setup( 

3104 setup: SetupSnapshot | None, 

3105 canvas_size: tuple[int, int] | None, 

3106 words: pd.DataFrame, 

3107 fixations: pd.DataFrame, 

3108 *, 

3109 side: str, 

3110) -> SetupSnapshot: 

3111 """One side's `SetupSnapshot`, from an explicit one, a canvas, or the data. 

3112 

3113 The provenance is the point, because the overlay gate reads it: a canvas the 

3114 caller *stated* is ``MEASURED``, one inferred from the data extents is 

3115 ``ESTIMATED``. Both count as knowing the screen; neither claims a physical 

3116 display, which a comparison never uses. 

3117 """ 

3118 if setup is not None: 

3119 if not isinstance(setup, SetupSnapshot): 

3120 raise TypeError( 

3121 f"{side} must be an experimental_setup.SetupSnapshot, got " 

3122 f"{type(setup).__name__}." 

3123 ) 

3124 return setup 

3125 provenance = Provenance.MEASURED 

3126 if canvas_size is None: 

3127 canvas_size = _recorded_screen(words, fixations) 

3128 if canvas_size is None: 

3129 provenance = Provenance.ESTIMATED 

3130 canvas_size = screen_canvas_size(words) or screen_canvas_size(fixations) 

3131 if canvas_size is None: 

3132 canvas_size = _data.compute_canvas_size(words, fixations) 

3133 return SetupSnapshot( 

3134 canvas_width=int(canvas_size[0]), 

3135 canvas_height=int(canvas_size[1]), 

3136 screen_provenance=provenance, 

3137 ) 

3138 

3139 

3140def compare_scanpaths( 

3141 words: pd.DataFrame, 

3142 fixations: pd.DataFrame, 

3143 trial_a: tuple[str, str], 

3144 trial_b: tuple[str, str], 

3145 *, 

3146 screen: str | None = None, 

3147 screen_b: str | None = None, 

3148 words_b: pd.DataFrame | None = None, 

3149 fixations_b: pd.DataFrame | None = None, 

3150 dataset_b: str = "Dataset B", 

3151 raw_gaze: pd.DataFrame | None = None, 

3152 raw_gaze_b: pd.DataFrame | None = None, 

3153 layout: str = "overlay", 

3154 compare_stimulus: str = "both", 

3155 setup: SetupSnapshot | None = None, 

3156 setup_b: SetupSnapshot | None = None, 

3157 canvas_size: tuple[int, int] | None = None, 

3158 labels: tuple[str, str] | None = None, 

3159 style_a: dict | None = None, 

3160 style_b: dict | None = None, 

3161 base_font_size: int = 16, 

3162 font_family: str = FONT_FAMILY, 

3163 fix_index_range: tuple[int, int] | None = None, 

3164 fix_index_range_b: tuple[int, int] | None = None, 

3165 drift_correction: str | None = None, 

3166 title: str = "", 

3167 caption: str = "", 

3168 column_names: dict | None = None, 

3169 **figure_overrides, 

3170) -> go.Figure: 

3171 """Build a two-scanpath comparison figure. 

3172 

3173 The headless form of the app's **Compare** mode. ``trial_a`` / ``trial_b`` 

3174 are ``(participant, trial)`` pairs; ``layout`` is ``"overlay"``, 

3175 ``"side_by_side"`` (``"side-by-side"`` also accepted) or ``"stacked"``. 

3176 

3177 **Multipart trials.** Each scanpath is one screen, never a whole multipart 

3178 trial: every screen is its own coordinate space, so pooling them would draw 

3179 saccades across page boundaries. ``screen`` picks A's screen and 

3180 ``screen_b`` B's, independently — B's is looked up in B's own frames, so it 

3181 may be a later page or another dataset's. Either one left out is that 

3182 trial's first recorded screen, as in 

3183 [`plot_scanpath`][scanpath_studio.api.plot_scanpath]; 

3184 ``list_parts()`` lists them. A screen named for a single-screen trial, or 

3185 one the trial does not have, raises ``ValueError``. 

3186 

3187 **Two datasets.** Pass ``words_b`` / ``fixations_b`` to draw B from a 

3188 *different* dataset. Two datasets can hold the same ``(participant_id, 

3189 trial_id)`` and the builder slices by exactly that pair, so B's participant 

3190 ids are namespaced with ``dataset_b`` inside the throwaway merged frames — 

3191 without it one trial would silently render as two. The frames you pass in 

3192 are never modified, and nothing in the returned figure's data depends on the 

3193 namespace beyond the trace labels. 

3194 

3195 **The overlay gate.** Across datasets an overlay needs both canvases to be 

3196 the same size; otherwise this raises ``ValueError`` (the app falls back to 

3197 side by side). One dataset can hold screens of different sizes too, so a 

3198 same-dataset pair is refused the same way when the two selected screens 

3199 carry different canvases (``canvas_width`` / ``canvas_height`` columns) or 

3200 ``setup_b`` states another screen. Pass ``layout="side_by_side"`` or 

3201 ``"stacked"`` to compare readings from different screens; each panel is 

3202 then drawn to its own. Nothing is rescaled. 

3203 

3204 ``setup`` / ``setup_b`` are `experimental_setup.SetupSnapshot` 

3205 values — what the gate reads. ``canvas_size`` covers A when you only have a 

3206 resolution; omit both and the canvas is read off the data. 

3207 

3208 **Stimulus images.** ``background_image`` is A's page. A split layout draws 

3209 B's panel over ``background_image_b`` (with ``background_image_size_b`` / 

3210 ``background_image_origin_b``) and over nothing without it — never A's, 

3211 since sharing a dataset says nothing about sharing a page. 

3212 

3213 ``compare_stimulus`` picks whose word boxes and text an **overlay** draws — 

3214 ``"both"`` (default), ``"a"`` or ``"b"``. Two datasets' AOIs coincide only 

3215 when the text is identical. Split layouts ignore it; each panel owns its own 

3216 stimulus. 

3217 

3218 **Per-scanpath style.** ``style_a`` / ``style_b`` restyle one scanpath: 

3219 ``fix_color``, ``marker_size_range``, ``opacity``, ``hollow``, 

3220 ``saccade_color``, ``saccade_style``, ``saccade_width``, ``box_color`` — 

3221 the outline of that reading's word boxes, its ``fix_color`` when left out — 

3222 ``box_fill_color``, their fill, ``word_box_fill_color`` when left out — and 

3223 ``raw_gaze_color``, that reading's raw-gaze samples, its ``fix_color`` when 

3224 left out. These three are this figure's only: the co-animation draws one set 

3225 of boxes, in ``word_box_color`` / ``word_box_fill_color``, and no raw gaze, 

3226 and ignores them. ``heatmap_colorscale`` gives that reading's word-box 

3227 heatmap its own color scale (``heatmap_colorscale`` when left out) on the 

3228 range both share; when A's and B's differ, each gets its own color bar. 

3229 

3230 **Filters, per scanpath.** ``fixation_flags`` and 

3231 ``saccade_classes`` filter both scanpaths, as they filter 

3232 [`plot_scanpath`][scanpath_studio.api.plot_scanpath]'s one; the same two keys 

3233 in ``style_a`` / ``style_b`` give that scanpath its own, overriding them — 

3234 e.g. ``style_b={"fixation_flags": {"short": {"mode": "Discard", 

3235 "threshold_ms": 80}}, "saccade_classes": ["regression"]}``. The app's 

3236 Compare mode draws A under the plot controls' filters and B under its own. 

3237 ``fix_index_range`` windows both scanpaths; ``fix_index_range_b`` gives B a 

3238 window of its own (the app's B slider). 

3239 

3240 **Raw gaze.** ``raw_gaze`` is a frame from 

3241 [`load_raw_gaze`][scanpath_studio.api.load_raw_gaze]; each reading's samples 

3242 are drawn under its scanpath, in that scanpath's color (``raw_gaze_marker_size`` 

3243 / ``raw_gaze_opacity`` style them). It serves both readings of a 

3244 same-dataset comparison; across datasets it is A's, and ``raw_gaze_b`` is 

3245 B's. Passing either turns the layer on; ``show_raw_gaze=False`` keeps it off. 

3246 

3247 Remaining keywords are forwarded to `plots.make_comparison_figure` 

3248 (e.g. ``show_words=False``, ``color_by="duration_ms"``); an unknown one 

3249 raises ``TypeError`` naming the closest valid options; 

3250 ``figure_options("comparison")`` lists the accepted keywords. Column names 

3251 follow [`plot_scanpath`][scanpath_studio.api.plot_scanpath]'s rule: A's 

3252 names (or ``column_names``) name the options and the figure's text, and 

3253 either dataset's frames may come under their own names. 

3254 """ 

3255 _check_canvas_size(canvas_size) 

3256 from .experimental_setup import IncomparableScreensError, setups_comparable 

3257 from .utils import ( 

3258 align_compare_columns, 

3259 qualify_for_compare, 

3260 self_compare_participant, 

3261 separate_self_compare, 

3262 ) 

3263 

3264 resolved_layout = _COMPARE_LAYOUTS.get(str(layout).strip().lower()) 

3265 if resolved_layout is None: 

3266 raise ValueError( 

3267 f"Unknown compare layout {layout!r}; choose one of " 

3268 f"{', '.join(sorted(set(_COMPARE_LAYOUTS.values())))}." 

3269 ) 

3270 _reject_unknown_options( 

3271 figure_overrides, 

3272 _COMPARISON_FIGURE_PARAMS | {"palette"}, 

3273 "compare_scanpaths", 

3274 ) 

3275 

3276 cross_dataset = words_b is not None or fixations_b is not None 

3277 # DATA-66: A's names name the figure's text and its options; every frame, 

3278 # A's or B's, is processed under the internal names. 

3279 carried = { 

3280 table: found[1] 

3281 for table, frame in (("fixations", fixations), ("words", words)) 

3282 if (found := _cn.frame_names(frame)) is not None 

3283 } 

3284 names = _call_names(column_names, **carried) 

3285 word_side = _table_names(column_names, "words", carried.get("words")) 

3286 figure_overrides = _canonical_options(figure_overrides, names, words=word_side) 

3287 words, fixations, words_b, fixations_b = ( 

3288 _cn.to_canonical_frame(frame) 

3289 for frame in (words, fixations, words_b, fixations_b) 

3290 ) 

3291 words_b = words if words_b is None else words_b 

3292 fixations_b = fixations if fixations_b is None else fixations_b 

3293 if raw_gaze is not None: 

3294 raw_gaze = _require_normalized(raw_gaze, "raw_gaze") 

3295 if raw_gaze_b is not None: 

3296 raw_gaze_b = _require_normalized(raw_gaze_b, "raw_gaze_b") 

3297 if raw_gaze_b is None and not cross_dataset: 

3298 raw_gaze_b = raw_gaze 

3299 

3300 for side, pair in (("trial_a", trial_a), ("trial_b", trial_b)): 

3301 if isinstance(pair, str) or len(tuple(pair)) != 2: 

3302 raise ValueError( 

3303 f"{side} must be a (participant, trial) pair, e.g. " 

3304 f"('l37_1129', 'l37_1129_2_1_1_Ele_r0'); got {pair!r}. " 

3305 "list_trials(words, fixations) lists the pairs." 

3306 ) 

3307 # Ids written before composite ids escaped a `_` inside a part still name 

3308 # their reading when that is unambiguous (`data.respell_reading`). 

3309 pid_a, tid_a = _data.respell_reading(*trial_a, _data.trial_keys(fixations)) 

3310 pid_b, tid_b = _data.respell_reading(*trial_b, _data.trial_keys(fixations_b)) 

3311 # One screen per side, each resolved in its own frames — the same contract 

3312 # as `plot_scanpath`'s `screen`. Extracting whole parent trials pooled every 

3313 # page of a multipart reading into one scanpath, saccades across pages and all. 

3314 trial_words_a, trial_fix_a, pid_a, tid_a, _screen_a = _select_part( 

3315 words, fixations, str(pid_a), str(tid_a), screen 

3316 ) 

3317 trial_words_b, trial_fix_b, pid_b, tid_b, _screen_b = _select_part( 

3318 words_b, 

3319 fixations_b, 

3320 str(pid_b), 

3321 str(tid_b), 

3322 screen_b, 

3323 screen_param="screen_b", 

3324 ) 

3325 trial_raw_a = _compare_raw_gaze(raw_gaze, pid_a, tid_a, trial_fix_a) 

3326 trial_raw_b = _compare_raw_gaze(raw_gaze_b, pid_b, tid_b, trial_fix_b) 

3327 for frame, (pid, tid) in ((trial_fix_a, trial_a), (trial_fix_b, trial_b)): 

3328 if frame.empty: 

3329 raise ValueError( 

3330 f"No fixations for participant={pid!r}, trial={tid!r}. " 

3331 f"list_trials() shows what the frames contain." 

3332 ) 

3333 # Either reading may carry the column (two corpora need not share them), so 

3334 # it is looked for across both. 

3335 _check_column_options( 

3336 figure_overrides, 

3337 words=pd.concat([trial_words_a, trial_words_b], ignore_index=True), 

3338 fixations=pd.concat([trial_fix_a, trial_fix_b], ignore_index=True), 

3339 ) 

3340 

3341 setup_a = _compare_setup( 

3342 setup, canvas_size, trial_words_a, trial_fix_a, side="setup" 

3343 ) 

3344 resolved_setup_b = _compare_setup( 

3345 setup_b, None, trial_words_b, trial_fix_b, side="setup_b" 

3346 ) 

3347 gate = cross_dataset 

3348 if not cross_dataset: 

3349 same_b = _same_dataset_setup_b( 

3350 setup_a, 

3351 setup_b, 

3352 a_known=setup is not None or canvas_size is not None, 

3353 words_a=trial_words_a, 

3354 fixations_a=trial_fix_a, 

3355 words_b=trial_words_b, 

3356 fixations_b=trial_fix_b, 

3357 ) 

3358 gate = same_b is not None 

3359 if same_b is not None: 

3360 resolved_setup_b = same_b 

3361 elif setup_b is None: 

3362 # Not refused: B's split panel is drawn to its own screen's canvas, 

3363 # else A's known one, else (neither known) its own data's extent. 

3364 own_b = screen_canvas_size(trial_words_b) or screen_canvas_size(trial_fix_b) 

3365 if own_b is None and ( 

3366 setup is not None 

3367 or canvas_size is not None 

3368 or screen_canvas_size(trial_words_a) is not None 

3369 or screen_canvas_size(trial_fix_a) is not None 

3370 ): 

3371 resolved_setup_b = setup_a 

3372 elif own_b is not None: 

3373 resolved_setup_b = replace( 

3374 setup_a, canvas_width=int(own_b[0]), canvas_height=int(own_b[1]) 

3375 ) 

3376 if resolved_layout == "overlay" and gate: 

3377 comparable, note = setups_comparable(setup_a, resolved_setup_b) 

3378 if not comparable: 

3379 # BUG-85: the reason says why; this says what happened here and how 

3380 # to ask for the split in Python. `render` rewords it in its flags. 

3381 hint = ( 

3382 _inferred_screen_hint( 

3383 a_inferred=setup is None and canvas_size is None, 

3384 b_inferred=setup_b is None, 

3385 ) 

3386 if cross_dataset 

3387 else "" 

3388 ) 

3389 raise IncomparableScreensError( 

3390 f"{note} So no overlay was drawn; pass layout='side_by_side' (or " 

3391 f"'stacked') to compare them in separate panels, each drawn to " 

3392 f"its own screen.{hint}", 

3393 reason=note, 

3394 ) 

3395 if note: 

3396 # The canvases match but at least one corpus never recorded a screen, 

3397 # so the overlay is drawn with a caveat rather than refused. A script 

3398 # has no caption to read it in, so it goes to the logger — loud enough 

3399 # to appear in a pipeline's output, quiet enough not to be an error. 

3400 logging.getLogger(__name__).warning("compare_scanpaths: %s", note) 

3401 

3402 figure_pid_b = pid_b 

3403 if cross_dataset: 

3404 trial_words_b = qualify_for_compare(trial_words_b, dataset_b) 

3405 trial_fix_b = qualify_for_compare(trial_fix_b, dataset_b) 

3406 trial_raw_b = qualify_for_compare(trial_raw_b, dataset_b) 

3407 figure_pid_b = ( 

3408 str(trial_fix_b["participant_id"].iloc[0]) 

3409 if not trial_fix_b.empty 

3410 else pid_b 

3411 ) 

3412 trial_fix_a = _apply_fix_index_range(trial_fix_a, fix_index_range, pid_a, tid_a) 

3413 trial_fix_b = _apply_fix_index_range( 

3414 trial_fix_b, 

3415 fix_index_range if fix_index_range_b is None else fix_index_range_b, 

3416 pid_b, 

3417 tid_b, 

3418 ) 

3419 if drift_correction: 

3420 # PRE-21: same contract as plot_scanpath — raise, don't silently skip. 

3421 if not drift_correction_enabled(): 

3422 raise ValueError( 

3423 "drift_correction is not available in this release. Set " 

3424 f"{EXPERIMENTAL_ENV_VAR}=1 to enable it, or pass " 

3425 "drift_correction=None." 

3426 ) 

3427 from .alignment import correct 

3428 

3429 trial_fix_a, _ = correct(trial_fix_a, trial_words_a, drift_correction) 

3430 trial_fix_b, _ = correct(trial_fix_b, trial_words_b, drift_correction) 

3431 

3432 if not cross_dataset and (pid_a, tid_a) == (pid_b, tid_b): 

3433 # CMP-22: a trial compared with itself — rename B's copy apart, or the 

3434 # figure's (participant, trial) slice hands each side both copies. 

3435 # The renamed id is for slicing only, so B's default legend name is 

3436 # resolved here from the real one. 

3437 if not labels: 

3438 labels = tuple( 

3439 _resolve_trial_display_name(pid_a, tid_a, trial_words_a, None, idx) 

3440 for idx in (0, 1) 

3441 ) 

3442 trial_words_b = separate_self_compare(trial_words_b, pid_b) 

3443 trial_fix_b = separate_self_compare(trial_fix_b, pid_b) 

3444 trial_raw_b = separate_self_compare(trial_raw_b, pid_b) 

3445 figure_pid_b = self_compare_participant(pid_b) 

3446 merged_words, merged_words_b, _ = align_compare_columns( 

3447 trial_words_a, trial_words_b 

3448 ) 

3449 merged_fix, merged_fix_b, _ = align_compare_columns(trial_fix_a, trial_fix_b) 

3450 merged_raw = None 

3451 if not (trial_raw_a.empty and trial_raw_b.empty): 

3452 merged_raw = pd.concat(align_compare_columns(trial_raw_a, trial_raw_b)[:2]) 

3453 settings = _figure_kwargs(figure_overrides) 

3454 settings.pop("illustration_reasons", None) 

3455 if raw_gaze is not None or raw_gaze_b is not None: 

3456 settings.setdefault("show_raw_gaze", True) 

3457 render_settings = FigureSettings.from_mapping( 

3458 {k: v for k, v in settings.items() if k in _COMPARISON_FIGURE_PARAMS}, 

3459 canvas_width=int(setup_a.canvas_width), 

3460 canvas_height=int(setup_a.canvas_height), 

3461 base_font_size=int(base_font_size), 

3462 font_family=font_family, 

3463 layout=resolved_layout, 

3464 compare_stimulus=normalize_option_value("compare_stimulus", compare_stimulus), 

3465 trial_labels=tuple(labels) if labels else None, 

3466 style_a=style_a, 

3467 style_b=style_b, 

3468 column_labels=_column_labels( 

3469 names, trial_words_a, trial_fix_a, words=word_side 

3470 ), 

3471 # Only the split layouts read this; an overlay that got here has two 

3472 # equal canvases anyway, so it is the same value either way. 

3473 canvas_b=resolved_setup_b.canvas, 

3474 ) 

3475 fig = make_comparison_figure( 

3476 pd.concat([merged_words, merged_words_b], ignore_index=True), 

3477 pd.concat([merged_fix, merged_fix_b], ignore_index=True), 

3478 (pid_a, tid_a), 

3479 (figure_pid_b, tid_b), 

3480 settings=render_settings, 

3481 raw_gaze=merged_raw, 

3482 ) 

3483 annotate_figure(fig, title=title, caption=caption) 

3484 return fig 

3485 

3486 

3487def _compare_raw_gaze( 

3488 raw_gaze: pd.DataFrame | None, pid: str, tid: str, trial_fix: pd.DataFrame 

3489) -> pd.DataFrame: 

3490 """One comparison reading's samples — its trial's, and its screen's when the 

3491 reading is one screen of a multipart trial.""" 

3492 if raw_gaze is None or raw_gaze.empty: 

3493 return pd.DataFrame() 

3494 samples = _data.filter_raw_gaze(raw_gaze, [pid], [tid]) 

3495 if SCREEN_ID in samples.columns and SCREEN_ID in trial_fix.columns: 

3496 screens = trial_fix[SCREEN_ID].dropna().unique() 

3497 if len(screens) == 1: 

3498 samples = extract_part(samples, pid, tid, screens[0]) 

3499 return samples 

3500 

3501 

3502def save_figure( 

3503 fig: go.Figure, 

3504 path: str | Path, 

3505 *, 

3506 scale: float = 2, 

3507 width: int | None = None, 

3508 height: int | None = None, 

3509 width_mm: float | None = None, 

3510 width_in: float | None = None, 

3511 dpi: int | None = None, 

3512) -> Path: 

3513 """Save a figure by extension: ``.html`` (interactive, needs no browser) or 

3514 ``.png``/``.svg``/``.pdf`` (static via Kaleido — needs Chrome, Chromium or 

3515 Edge; run ``plotly_get_chrome -y`` once if none is installed). ``width`` / 

3516 ``height`` set the image size in px (overriding the figure's own size); 

3517 both ignored for ``.html``. Returns the written path. 

3518 

3519 ``width_mm`` or ``width_in`` with ``dpi`` (default 300) sizes a PNG for 

3520 print, as the app's Export → *Current figure* does: 180 mm at 600 dpi is 

3521 a 4,252 px wide PNG, its height following the figure's aspect, with the 

3522 dpi written into the file. They replace ``scale``.""" 

3523 path = Path(path) 

3524 suffix = path.suffix.lower() 

3525 if suffix not in (".html", ".png", ".svg", ".pdf"): 

3526 raise ValueError( 

3527 f"save_figure writes .html, .png, .svg or .pdf, not {suffix or path.name!r}. " 

3528 "For a GIF or MP4 replay, use " 

3529 "scanpath_studio.animation_export.export_animation." 

3530 ) 

3531 if not path.parent.is_dir(): 

3532 raise FileNotFoundError( 

3533 f"Can't write {path}: the folder {path.parent} does not exist." 

3534 ) 

3535 if width_mm is not None or width_in is not None: 

3536 if width_mm is not None and width_in is not None: 

3537 raise ValueError("Pass width_mm or width_in, not both.") 

3538 if suffix != ".png": 

3539 raise ValueError( 

3540 "width_mm / width_in / dpi size a PNG; save as .png, or set " 

3541 "width / height / scale for other formats." 

3542 ) 

3543 dpi = int(dpi or _export.DEFAULT_PRINT_DPI) 

3544 unit, value = ("mm", width_mm) if width_mm is not None else ("in", width_in) 

3545 base = int(width or fig.layout.width or 700) 

3546 scale = _export.print_scale(base, float(value), unit, dpi) 

3547 elif dpi is not None: 

3548 raise ValueError( 

3549 "dpi is the resolution of a print width: pass width_mm or width_in too." 

3550 ) 

3551 if suffix == ".html": 

3552 # BUG-93: an animation replays on the wall-clock player, which also 

3553 # autoplays it at the configured speed when asked (VIZ-10). Plotly's own 

3554 # `auto_play` stays off — it ignores the frame duration. PERF-17: the 

3555 # frames are written packed and rebuilt by the page's own script, so a 

3556 # long replay writes a fraction of the bytes. Static figures write 

3557 # unchanged. 

3558 page = replay_page(fig) 

3559 if page is not None: 

3560 figure_dict, script = page 

3561 pio.write_html( 

3562 figure_dict, 

3563 str(path), 

3564 validate=False, 

3565 auto_play=False, 

3566 post_script=script, 

3567 config={**PLOTLY_CONFIG}, 

3568 ) 

3569 elif fig.frames: 

3570 fig.write_html(str(path), auto_play=False, config={**PLOTLY_CONFIG}) 

3571 else: 

3572 fig.write_html(str(path), config={**PLOTLY_CONFIG}) 

3573 return path 

3574 if suffix in (".png", ".svg", ".pdf"): 

3575 try: 

3576 fig.write_image(str(path), scale=scale, width=width, height=height) 

3577 if dpi is not None: 

3578 _export.set_png_dpi(path, dpi) 

3579 except OSError: 

3580 raise # filesystem problem — the original error says it best 

3581 except Exception as exc: # Kaleido raises various types 

3582 raise RuntimeError( 

3583 f"Static {suffix} export failed ({exc}). Kaleido needs Chrome, " 

3584 "Chromium or Edge — install one, or run `plotly_get_chrome -y` " 

3585 "once — or save as .html, which needs no browser." 

3586 ) from exc 

3587 return path 

3588 

3589 

3590def save_figure_layers( 

3591 fig: go.Figure, 

3592 directory: str | Path, 

3593 *, 

3594 fmt: str = "svg", 

3595 scale: int = 2, 

3596 width: int | None = None, 

3597 height: int | None = None, 

3598) -> dict: 

3599 """Split a scanpath figure into its layers and save one file per layer. 

3600 

3601 Writes ``<directory>/<layer>.<fmt>`` for each *visible* layer (word boxes / 

3602 fixations / saccades / heatmap / labels / stimulus image / frame) and returns 

3603 ``{layer: Path}``. Each layer is the full figure with only that layer's elements 

3604 and a transparent background, at the same size and axis ranges — so the files 

3605 register perfectly when stacked in Illustrator / Inkscape. ``fmt`` is any 

3606 [`save_figure`][scanpath_studio.api.save_figure] extension without the dot 

3607 (``svg`` / ``pdf`` are vector and best for editing; ``png`` / ``html`` also 

3608 work). ``scale`` / ``width`` / ``height`` are forwarded to 

3609 [`save_figure`][scanpath_studio.api.save_figure].""" 

3610 directory = Path(directory) 

3611 # ENG-54: a failed render (most often Kaleido with no Chrome) used to leave 

3612 # an empty `<output>_layers/` behind, which reads as "exported, but lost". 

3613 # Whatever this call created is removed again if nothing was written to it. 

3614 created = [path for path in (directory, *directory.parents) if not path.exists()] 

3615 directory.mkdir(parents=True, exist_ok=True) 

3616 written: dict = {} 

3617 try: 

3618 for layer, layer_fig in split_scanpath_layers(fig).items(): 

3619 path = directory / f"{layer}.{fmt.lstrip('.')}" 

3620 written[layer] = save_figure( 

3621 layer_fig, path, scale=scale, width=width, height=height 

3622 ) 

3623 except Exception: 

3624 for path in created: # deepest first 

3625 if path.is_dir() and not any(path.iterdir()): 

3626 path.rmdir() 

3627 raise 

3628 return written 

3629 

3630 

3631def figure_code( 

3632 *, 

3633 kind: str = "static", 

3634 source: str = "demo", 

3635 source_options: dict | None = None, 

3636 participant: str = "", 

3637 trial: str = "", 

3638 screen: str | None = None, 

3639 compare: tuple[str, str] | None = None, 

3640 compare_screen: str | None = None, 

3641 compare_layout: str = "overlay", 

3642 compare_stimulus: str = "both", 

3643 compare_dataset: str = "", 

3644 compare_canvas: tuple[int, int] | None = None, 

3645 compare_labels: tuple[str, str] | None = None, 

3646 canvas_size: tuple[int, int] | None = None, 

3647 base_font_size: int = 16, 

3648 font_family: str = FONT_FAMILY, 

3649 title: str = "", 

3650 caption: str = "", 

3651 fix_index_range: tuple[int, int] | None = None, 

3652 illustration_label: str = "auto", 

3653 drift_correction: str | None = None, 

3654 drift_connectors: bool = False, 

3655 playback_speed: float = 1.0, 

3656 autoplay: bool = True, 

3657 flavor: str = "python", 

3658 explicit: bool = False, 

3659 output: str | None = None, 

3660 **figure_overrides, 

3661) -> str: 

3662 """The API or CLI code that reproduces a figure. 

3663 

3664 The headless twin of the app's 🔗 Share → *Reproduce this figure in code* block: give 

3665 it the same arguments you would give 

3666 [`plot_scanpath`][scanpath_studio.api.plot_scanpath] (``kind="static"``), 

3667 [`animate_scanpath`][scanpath_studio.api.animate_scanpath] (``"animation"``) or 

3668 [`compare_scanpaths`][scanpath_studio.api.compare_scanpaths] (``"comparison"``) and 

3669 it returns the snippet that rebuilds that figure, rather than the figure:: 

3670 

3671 print(sps.figure_code(participant="l7_1090", trial="l7_1090_2_1_1_Ele_r0", 

3672 show_heatmap=True, flavor="cli")) 

3673 

3674 ``source`` names how the data is loaded — ``"demo"``, ``"synthetic"``, ``"files"``, 

3675 ``"potec"``, ``"onestop"``, ``"author"``, or 

3676 ``"unknown"`` for data a snippet can't name — with ``source_options`` carrying that 

3677 loader's arguments (``{"root": …}``, ``{"words": [...], "fixations": [...]}``, and 

3678 so on). With ``show_raw_gaze=True`` the raw-gaze table is read too: the demo's own, 

3679 or the path(s) given as ``source_options["raw_gaze"]`` (plus an optional 

3680 ``"raw_gaze_schema"``) — [`load_raw_gaze`][scanpath_studio.api.load_raw_gaze] in the 

3681 Python form, ``--raw-gaze`` in the CLI one. ``source="raw_gaze"`` is a dataset 

3682 recorded as raw gaze alone: the samples at ``source_options["raw_gaze"]`` are the 

3683 data, and ``plot_scanpath`` is handed ``None`` for the words and fixations. 

3684 

3685 ``screen`` / ``compare_screen`` are A's and B's screens of a multipart trial 

3686 (``screen=`` / ``screen_b=``, ``--screen`` / ``--compare-screen``). 

3687 

3688 ``compare_dataset`` names the dataset scanpath B was loaded from when it is a 

3689 *second* one. B's participant id belongs to that dataset rather than 

3690 the one the snippet loads, so both forms then load B's own tables and name 

3691 B in them — ``words_b=`` / ``fixations_b=`` / ``dataset_b=``, and 

3692 ``--compare-words`` / ``--compare-fixations`` beside ``--compare-with`` — 

3693 from the placeholder paths ``B_WORDS`` / ``B_FIXATIONS``, which you point 

3694 at its files. ``compare_canvas`` is B's screen, ``(width, height)``, when 

3695 you know it: written as ``setup_b=`` and ``--compare-canvas``, which a 

3696 co-animation across datasets needs. 

3697 

3698 ``compare_labels`` is the pair you would pass 

3699 [`compare_scanpaths`][scanpath_studio.api.compare_scanpaths] as ``labels=`` — the 

3700 two trace labels, when they are not the composed defaults. Both forms 

3701 carry them: ``labels=`` in the Python snippet, ``--label-a`` / ``--label-b`` in the 

3702 CLI one. 

3703 

3704 With ``participant`` / ``trial`` left empty the snippet renders the first 

3705 available trial, as ``render`` does. ``canvas_size`` defaults to the screen 

3706 ``render`` assumes for the source (the demo's 2560×1440, PoTeC's 1680×1050, 

3707 …), so both flavors draw the same figure; ``output`` defaults to 

3708 ``scanpath.html`` for an animation — ``render --animate`` writes only HTML — 

3709 and to a PNG otherwise. 

3710 

3711 Only the options that differ from 

3712 [`figure_options`][scanpath_studio.api.figure_options] are written, so the snippet 

3713 stays readable; ``explicit=True`` emits every option at its current value. 

3714 ``flavor`` is ``"python"``, ``"cli"``, or ``"both"`` (the two separated by a blank 

3715 line). Anything neither form can reproduce — a raw-gaze table with no path to 

3716 name, an uploaded stimulus image, B's rows from a second corpus — follows as 

3717 ``# Note:`` comments, matching the ⚠️ captions the app shows and the ``Note:`` lines 

3718 `render --print-code` writes to stderr. See `code_snippet.ReproductionCode` for the 

3719 structured form. 

3720 """ 

3721 from . import code_snippet as _snippet 

3722 

3723 if flavor not in ("python", "cli", "both"): 

3724 raise ValueError(f"flavor must be 'python', 'cli' or 'both', got {flavor!r}.") 

3725 _reject_unknown_options( 

3726 figure_overrides, 

3727 set(figure_options(kind)) | {"palette"}, 

3728 "figure_code", 

3729 ) 

3730 if canvas_size is None: 

3731 # EXP-14: `render` snaps these sources to their recorded screen while 

3732 # `plot_scanpath` estimates one from the data, so leaving the canvas 

3733 # unnamed made the two flavours of one recipe disagree. 

3734 canvas_size = _snippet.source_canvas(source) 

3735 state = _snippet.FigureState( 

3736 kind=kind, 

3737 settings={**figure_options(kind), **_expand_palette(figure_overrides)}, 

3738 participant=participant, 

3739 trial=trial, 

3740 screen=screen, 

3741 canvas=canvas_size, 

3742 base_font_size=base_font_size, 

3743 font_family=font_family, 

3744 title=title, 

3745 caption=caption, 

3746 fix_index_range=fix_index_range, 

3747 illustration_label=illustration_label, 

3748 drift_correction=drift_correction, 

3749 drift_connectors=drift_connectors, 

3750 playback_speed=playback_speed, 

3751 autoplay=autoplay, 

3752 compare=( 

3753 _snippet.CompareTarget( 

3754 participant=str(compare[0]), 

3755 trial=str(compare[1]), 

3756 screen=None if compare_screen is None else str(compare_screen), 

3757 layout=compare_layout, 

3758 compare_stimulus=compare_stimulus, 

3759 dataset=str(compare_dataset), 

3760 canvas=( 

3761 (int(compare_canvas[0]), int(compare_canvas[1])) 

3762 if compare_canvas and compare_dataset 

3763 else None 

3764 ), 

3765 labels=( 

3766 (str(compare_labels[0]), str(compare_labels[1])) 

3767 if compare_labels 

3768 else None 

3769 ), 

3770 ) 

3771 if compare is not None 

3772 else None 

3773 ), 

3774 ) 

3775 code = _snippet.reproduction_code( 

3776 _snippet.SnippetSource( 

3777 kind=source, label=source, options=dict(source_options or {}) 

3778 ), 

3779 state, 

3780 explicit=explicit, 

3781 output=output or _snippet.DEFAULT_OUTPUT.get(kind, "scanpath.png"), 

3782 ) 

3783 cli = code.cli 

3784 if code.cli_unsupported: 

3785 cli += "\n# No `render` flag for: " + ", ".join(code.cli_unsupported) 

3786 # The caveats apply to *both* snippets, so on "both" they are appended once 

3787 # at the end rather than to each half — two identical blocks would read as 

3788 # two different warnings. 

3789 notes = "".join(f"\n# Note: {note}" for note in code.caveats) 

3790 if flavor == "python": 

3791 return code.python + notes 

3792 if flavor == "cli": 

3793 return cli + notes 

3794 return f"{code.python}\n\n{cli}{notes}" 

3795 

3796 

3797def cache_status() -> dict: 

3798 """Describe the on-device recovery cache a local app run keeps. 

3799 

3800 The app stores completed uploaded datasets, column mappings, view settings, 

3801 saved designs, metadata tables and annotations under the user's cache directory so a refresh or restart resumes 

3802 where it left off — on localhost/desktop only, never on a hosted deployment. This 

3803 reports that store without launching the app: ``enabled``, ``directory``, 

3804 ``datasets`` (name + per-frame row counts), ``rows``, ``annotations``, 

3805 ``designs``, ``metadata``, ``settings``, ``bytes``, ``saved_at``, plus ``exists`` / ``readable`` for a 

3806 missing or unreadable manifest, ``damaged`` (name + reason) for a stored 

3807 dataset whose entry or files are broken — the app restores the others and 

3808 keeps that one in the cache rather than dropping it — and 

3809 ``damaged_metadata``, the reason the stored metadata tables would not 

3810 restore (``""`` when they would). Delete it with 

3811 [`clear_cache`][scanpath_studio.api.clear_cache]; the same information is in the 

3812 app's 🗂️ Data Management → *Saved on this computer* section and in 

3813 ``scanpath-studio cache``.""" 

3814 from .persistence import cache_status as _cache_status 

3815 

3816 return _cache_status(url="http://localhost") 

3817 

3818 

3819def clear_cache() -> dict: 

3820 """Delete the on-device recovery cache and return its status afterwards. 

3821 

3822 Removes only the files this app wrote (``manifest.json`` and the dataset 

3823 Parquet files); anything else in the folder is left alone. A *running* local 

3824 app writes its session back out at the end of its next change — start it 

3825 with ``scanpath-studio run --no-persist`` or ``SCANPATH_STUDIO_PERSIST=0`` to 

3826 stop that.""" 

3827 from .persistence import clear_local_state 

3828 

3829 clear_local_state() 

3830 return cache_status() 

3831 

3832 

3833def version_info() -> BuildInfo: 

3834 """Which build of Scanpath Studio this is — no network access. 

3835 

3836 ``version`` is what ``scanpath_studio.__version__`` holds: the release itself 

3837 (``"0.35.0"``), or between releases a PEP 440 version that sorts after it — 

3838 ``"0.35.0.post3+g8f18219"`` is three commits after v0.35.0, at commit 

3839 ``8f18219``, and it ends ``.dirty`` with uncommitted changes. ``release`` is 

3840 the release it descends from (``scanpath_studio.__release__``), ``distance`` 

3841 the commits since (``None`` when unknown), ``commit``, ``dirty``, and 

3842 ``source`` — how it was worked out: ``"checkout"`` (``git describe``), 

3843 ``"stamp"`` (a desktop bundle's build stamp), ``"vcs"`` (a 

3844 ``pip install git+…``) or ``"release"``. ``describe()`` says it in a 

3845 sentence. The same is in Help → About and ``scanpath-studio version``.""" 

3846 from .build_info import build_info 

3847 

3848 return build_info() 

3849 

3850 

3851def check_for_updates(timeout: float = 5.0) -> UpdateCheck: 

3852 """Ask GitHub whether a newer release than this build is out. 

3853 

3854 The one call here that uses the network, and only when made: it reads the 

3855 latest release from ``api.github.com`` (drafts and pre-releases excluded) 

3856 and compares it with [`version_info`][scanpath_studio.api.version_info]. It 

3857 never raises. ``status`` is ``"up_to_date"``, ``"update_available"``, 

3858 ``"ahead"`` (a development build past the latest release) or ``"error"`` 

3859 (offline, no answer within ``timeout`` seconds, rate-limited, …), and 

3860 ``message`` says it in a sentence. With an update available, ``command`` is 

3861 the shell command that updates this install (``pip install -U 

3862 scanpath-studio``, ``uv tool upgrade scanpath-studio``, ``git pull``, …), 

3863 ``latest.url`` the release notes, and in the desktop app ``download`` the 

3864 archive for this computer. The same check is Help → About → *Check for 

3865 updates* and ``scanpath-studio version --check``.""" 

3866 from .updates import check_for_updates as _check 

3867 

3868 return _check(timeout)