Coverage for scanpath_studio/plots.py: 97%

3079 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""Plotly figure builders for scanpath visualization.""" 

2 

3from __future__ import annotations 

4 

5import base64 

6import copy 

7import html 

8import itertools 

9import math 

10import re 

11import struct 

12from collections.abc import Callable, Iterable, Iterator, Mapping, Sequence 

13from contextlib import contextmanager 

14from contextvars import ContextVar 

15from dataclasses import MISSING, dataclass, fields, replace 

16from pathlib import Path 

17from typing import Any 

18 

19import numpy as np 

20import pandas as pd 

21import plotly.graph_objects as go 

22 

23from . import progress 

24from .constants import ( 

25 APP_THEME, 

26 CANVAS_PAD_FRACTION, 

27 CANVAS_PAD_MIN_PX, 

28 COMPARE_FIXATION_OPACITY, 

29 COMPARISON_PALETTE, 

30 CURRENT_FIX_COLOR, 

31 CURRENT_FIX_OUTLINE, 

32 DEFAULT_FIXATION_COLOR, 

33 DEFAULT_FIXATION_COLORSCALE, 

34 DEFAULT_FIXATION_SYMBOL, 

35 DEFAULT_HEATMAP_COLORSCALE, 

36 DEFAULT_LINE_SPACING, 

37 DEFAULT_MARKER_DURATION_RANGE, 

38 DEFAULT_MARKER_SIZE_RANGE, 

39 DEFAULT_MARKER_SIZE_SCALE, 

40 DEFAULT_SACCADE_WIDTH, 

41 FIX_MARKER_OUTLINE, 

42 FIXATION_GLYPH_SIZE_SCALE, 

43 FIXATION_GLYPH_SYMBOLS, 

44 FIXATION_SYMBOLS, 

45 FONT_FAMILY, 

46 HIGHLIGHTED_TEXT_COLOR, 

47 HOLLOW_OUTLINE_WIDTH, 

48 LEGEND_ARRANGEMENTS, 

49 LEGEND_KINDS, 

50 LEGEND_POSITIONS, 

51 MARKER_SIZE_SCALES, 

52 OUT_OF_TEXT_COLOR, 

53 SACCADE_CLASS_COLORS, 

54 SACCADE_CLASS_LABELS, 

55 SACCADE_CLASS_ORDER, 

56 SACCADE_COLOR, 

57 SACCADE_COLOR_MODES, 

58 SACCADE_DASH_OPTIONS, 

59 SACCADE_DIRECTION_CLASSES, 

60 SACCADE_DIRECTION_FOLD, 

61 SACCADE_DIRECTION_LABELS, 

62 SAMPLE_INDEX, 

63 TRENDLINE_COLOR, 

64 UNIFORM_COLOR_FIELD, 

65 WORD_BOX_COLOR, 

66 WORD_BOX_FILL_COLOR, 

67 WORD_BOX_FILL_OPACITY, 

68 WORD_BOX_LINE_OPACITY, 

69 WORD_LABEL_COLOR, 

70 compare_palette_color, 

71) 

72from .illustration import MANUAL_LABEL_REASON 

73from .multipart import SCREEN_ID 

74 

75COLORBAR_LEN_FRACTION = 0.33 

76 

77 

78@dataclass(frozen=True) 

79class FigureSettings: 

80 """Immutable rendering settings shared by every scanpath figure builder. 

81 

82 The app, headless API, CLI, and exporters used to forward parallel keyword 

83 lists into three builders with 45–69 parameters each. This object is the 

84 single rendering contract instead. Builder-specific fields live together 

85 deliberately: switching between static, animated, and comparison views must 

86 preserve the common visual choices without another translation layer. 

87 

88 ``canvas_width``, ``canvas_height``, and ``base_font_size`` are the only 

89 context-dependent required values. Everything else has the builder's 

90 behavior-preserving default and may be changed with :meth:`with_overrides`. 

91 """ 

92 

93 canvas_width: int 

94 canvas_height: int 

95 base_font_size: int 

96 font_family: str = FONT_FAMILY 

97 x_field: str = "x" 

98 y_field: str = "y" 

99 show_words: bool = True 

100 #: The word boxes' outline colour at ``word_box_line_opacity``, and their 

101 #: fill — a colour drawn at ``word_box_fill_opacity``. A comparison outlines 

102 #: each reading's boxes in its scanpath's ``box_color`` style (its fixation 

103 #: colour by default) instead, so ``word_box_color`` is static/replay only; 

104 #: the line opacity applies to every outline. 

105 word_box_color: str = WORD_BOX_COLOR 

106 word_box_line_opacity: float = WORD_BOX_LINE_OPACITY 

107 word_box_fill_color: str = WORD_BOX_FILL_COLOR 

108 word_box_fill_opacity: float = WORD_BOX_FILL_OPACITY 

109 show_word_labels: bool = True 

110 show_fixations: bool = True 

111 show_order: bool = True 

112 show_saccades: bool = True 

113 show_heatmap: bool = False 

114 color_by: str | None = UNIFORM_COLOR_FIELD 

115 heatmap_metric: str | None = None 

116 show_saccade_arrows: bool = False 

117 heatmap_style: str = "Word boxes" 

118 heatmap_norm: str = "Linear" 

119 #: The Interpolated heatmap's Gaussian σ in px; ``None`` picks it from the 

120 #: data (`interpolated_sigma_px`). 

121 heatmap_sigma_px: float | None = None 

122 marker_size_range: tuple[int, int] = DEFAULT_MARKER_SIZE_RANGE 

123 #: How duration maps onto ``marker_size_range`` — one of 

124 #: ``constants.MARKER_SIZE_SCALES``. The fixed scales ("sqrt", "linear", 

125 #: "log") map ``marker_duration_range`` (ms) onto it for every figure, so 

126 #: equal durations draw at equal sizes across trials, comparison sides, 

127 #: replays and exports; "relative" spans each figure's own durations. 

128 marker_size_scale: str = DEFAULT_MARKER_SIZE_SCALE 

129 marker_duration_range: tuple[float, float] = DEFAULT_MARKER_DURATION_RANGE 

130 #: A few reference circles labelled in ms, drawn under a fixed scale. 

131 duration_size_legend: bool = True 

132 #: Where each legend sits, as ``{kind: {"position", "arrangement", "size"}}`` 

133 #: for the kinds in ``LEGEND_KINDS``. A kind left out, or set to "auto" 

134 #: throughout, is drawn where it always was; whether it is drawn at all is 

135 #: still its own layer's switch. See :func:`normalize_legend_layout`. 

136 legend_layout: dict | None = None 

137 order_font_size: int | None = 10 

138 order_font_color: str = "#111111" 

139 #: Each colour scale's bar has its own switch and style: the fixations' 

140 #: (a numeric ``color_by``) and the heatmap's. 

141 show_fixation_colorbar: bool = True 

142 fixation_colorbar_orientation: str = "Vertical" 

143 fixation_colorbar_tickangle: int = 0 

144 fixation_colorbar_tickfont_size: int = 12 

145 show_heatmap_colorbar: bool = True 

146 heatmap_colorbar_orientation: str = "Vertical" 

147 heatmap_colorbar_tickangle: int = 0 

148 heatmap_colorbar_tickfont_size: int = 12 

149 fixation_color_range: tuple[float, float] | None = None 

150 heatmap_range: tuple[float, float] | None = None 

151 fixation_colorscale: str = DEFAULT_FIXATION_COLORSCALE 

152 heatmap_colorscale: str = DEFAULT_HEATMAP_COLORSCALE 

153 show_raw_gaze: bool = False 

154 raw_gaze_color: str = "#888888" 

155 raw_gaze_marker_size: float = 4.0 

156 raw_gaze_opacity: float = 0.6 

157 critical_span_style: str = "Mark text" 

158 highlight_column: str | None = "is_in_aspan" 

159 saccade_color: str = SACCADE_COLOR 

160 saccade_style: str = "solid" 

161 saccade_width: float = DEFAULT_SACCADE_WIDTH 

162 saccade_color_mode: str = "Uniform" 

163 saccade_class_colors: dict | None = None 

164 saccade_type_legend: bool = True 

165 saccade_classes: Iterable[str] | None = None 

166 saccade_render_mode: str = "Straight" 

167 fixation_snap_to_word: bool = False 

168 hollow_fixations: bool = False 

169 fixation_opacity: float = 1.0 

170 fixation_color: str | None = DEFAULT_FIXATION_COLOR 

171 fixation_symbol: str = DEFAULT_FIXATION_SYMBOL 

172 text_color: str = WORD_LABEL_COLOR 

173 highlight_text_color: str = HIGHLIGHTED_TEXT_COLOR 

174 background_color: str | None = None 

175 color_by_line: bool = False 

176 fixation_flags: dict | None = None 

177 #: CMP-24: scanpath B's own flags in a co-animation (Animate + Compare); 

178 #: ``None`` gives B the same ``fixation_flags`` as A. 

179 fixation_flags_b: dict | None = None 

180 span_border_color: str = "#000000" 

181 line_spacing: float = DEFAULT_LINE_SPACING 

182 scale_text_to_boxes: bool = True 

183 background_image: str | None = None 

184 background_image_size: tuple[float, float] | None = None 

185 background_image_origin: tuple[float, float] | None = None 

186 background_image_opacity: float = 1.0 

187 fit_to_monitor: bool = False 

188 show_coordinate_grid: bool = False 

189 coordinate_grid_spacing: float | None = None 

190 word_heatmap_col: str | None = None 

191 word_heatmap_title: str | None = None 

192 word_hover_measure: str | None = "total_fixation_duration_ms" 

193 word_hover_fields: Sequence[str] | None = None 

194 fixation_hover_fields: Sequence[str] | None = None 

195 show_connectors: bool = False 

196 connector_y: Sequence[float] | None = None 

197 illustration_reasons: Sequence[str] | None = None 

198 #: The Illustration label's text; empty writes "Illustration · <reasons>". 

199 illustration_text: str = "" 

200 playback_speed: float = 1.0 

201 label_a: str = "Scanpath A" 

202 label_b: str = "Scanpath B" 

203 show_legend: bool = False 

204 autoplay: bool = True 

205 anim_grid_step_ms: float | None = None 

206 anim_max_frames: int | None = None 

207 trial_labels: tuple[str, str] | None = None 

208 layout: str = "overlay" 

209 style_a: dict | None = None 

210 style_b: dict | None = None 

211 # CMP-8 §4 — scanpath B's own screen, honoured *only* by 

212 # `_make_split_comparison_figure` (side-by-side / stacked). `None` means "the 

213 # same screen as A", which is every same-dataset comparison and so leaves 

214 # every existing figure byte-identical. Overlay never needs these: CMP-11 

215 # lets a cross-dataset pair overlay only when both screens are equal, so 

216 # there is no second canvas for it to reconcile. 

217 canvas_b: tuple[int, int] | None = None 

218 background_image_b: str | None = None 

219 background_image_size_b: tuple[float, float] | None = None 

220 background_image_origin_b: tuple[float, float] | None = None 

221 # CMP-11 — which reading supplies the stimulus layer (word boxes + labels) 

222 # on an OVERLAY: "both" (the default, and byte-identical to pre-CMP-11), 

223 # "a", or "b". Two datasets' AOIs coincide only when the text is identical, 

224 # so an overlay across corpora can otherwise stack two offset sets of 

225 # rectangles. Split layouts ignore it — each panel owns its own stimulus, 

226 # and hiding one panel's boxes would just leave a blank half. 

227 compare_stimulus: str = "both" 

228 # DATA-66 — the dataset's own names for the columns the figure's text names 

229 # (hover rows, colour-bar and legend titles, non-spatial axis titles), 

230 # canonical column → label. Built by the caller (`tabs`); not a figure 

231 # option, so no option set, link or `render` flag carries it. 

232 column_labels: dict | None = None 

233 

234 @classmethod 

235 def from_mapping( 

236 cls, 

237 settings: FigureSettings | Mapping[str, Any] | None = None, 

238 /, 

239 **overrides: Any, 

240 ) -> FigureSettings: 

241 """Build settings from another instance or a plain option mapping.""" 

242 valid = {field.name for field in fields(cls)} 

243 unknown = sorted(set(overrides) - valid) 

244 if isinstance(settings, cls): 

245 if unknown: 

246 raise TypeError(f"Unknown figure settings: {', '.join(unknown)}") 

247 return replace(settings, **overrides) if overrides else settings 

248 values = dict(settings or {}) 

249 unknown = sorted((set(values) | set(overrides)) - valid) 

250 if unknown: 

251 raise TypeError(f"Unknown figure settings: {', '.join(unknown)}") 

252 values.update(overrides) 

253 return cls(**values) 

254 

255 def with_overrides(self, **overrides: Any) -> FigureSettings: 

256 """Return a copy with the named settings replaced.""" 

257 return self.from_mapping(self, **overrides) 

258 

259 def for_builder(self, names: Iterable[str]) -> dict[str, Any]: 

260 """Return just the fields consumed by one concrete renderer.""" 

261 return {name: getattr(self, name) for name in names} 

262 

263 @classmethod 

264 def defaults(cls, names: Iterable[str]) -> dict[str, Any]: 

265 """Return dataclass defaults for the requested non-context fields.""" 

266 defaults: dict[str, Any] = {} 

267 by_name = {field.name: field for field in fields(cls)} 

268 for name in names: 

269 field = by_name.get(name) 

270 if field is None: 

271 continue 

272 if field.default is not MISSING: 

273 defaults[name] = field.default 

274 return defaults 

275 

276 

277#: The figure options that take one of a fixed set of values → those values, 

278#: spelt as the builders compare them. `normalize_option_values` reads any 

279#: spelling of one (case, spaces, ``-`` / ``_`` and ``/`` ignored, so the CLI's 

280#: ``word-boxes`` and ``mark-border`` work) and refuses anything else, so a 

281#: script never gets the default drawn in place of a value it misspelt. 

282FIGURE_OPTION_CHOICES: dict[str, tuple[str, ...]] = { 

283 "heatmap_style": ("Word boxes", "Interpolated"), 

284 "heatmap_norm": ("Linear", "Log"), 

285 "critical_span_style": ("Mark text", "Mark border", "None"), 

286 "saccade_color_mode": tuple(SACCADE_COLOR_MODES), 

287 "saccade_render_mode": ("Straight", "Arc"), 

288 "saccade_style": tuple(SACCADE_DASH_OPTIONS.values()), 

289 "marker_size_scale": tuple(MARKER_SIZE_SCALES), 

290 "fixation_symbol": tuple(FIXATION_SYMBOLS), 

291 "fixation_colorbar_orientation": ("Vertical", "Horizontal"), 

292 "heatmap_colorbar_orientation": ("Vertical", "Horizontal"), 

293 "compare_stimulus": ("both", "a", "b"), 

294} 

295 

296 

297def _choice_key(value: object) -> str: 

298 return "".join(ch for ch in str(value).casefold() if ch.isalnum()) 

299 

300 

301#: Other spellings a choice is known by: the CLI's flag names and the app's 

302#: labels where they differ from the value (``Dashed`` is ``dash``). 

303_CHOICE_ALIASES: dict[str, dict[str, str]] = { 

304 "saccade_color_mode": { 

305 "type": "By type", 

306 "direction": "Forward / regression", 

307 "bydirection": "Forward / regression", 

308 }, 

309 "saccade_render_mode": {"arcs": "Arc"}, 

310 "saccade_style": { 

311 _choice_key(label): value for label, value in SACCADE_DASH_OPTIONS.items() 

312 }, 

313} 

314 

315 

316def normalize_option_value(name: str, value: object) -> object: 

317 """``value`` for the enumerated figure option ``name``, spelt as the 

318 builders compare it; any other option's value is returned as it is. 

319 

320 Raises ``ValueError`` listing the choices for a value that is none of 

321 them. ``critical_span_style=None`` is the app's "None" (no marking).""" 

322 choices = FIGURE_OPTION_CHOICES.get(name) 

323 if choices is None: 

324 return value 

325 if value is None and "None" in choices: 

326 return "None" 

327 key = _choice_key(value) 

328 for choice in choices: 

329 if _choice_key(choice) == key: 

330 return choice 

331 alias = _CHOICE_ALIASES.get(name, {}).get(key) 

332 if alias is not None: 

333 return alias 

334 raise ValueError(f"Unknown {name} {value!r}; choose one of {', '.join(choices)}.") 

335 

336 

337#: The palettes' short names (#374): no spaces or brackets, so a shell needs 

338#: no quotes — `render --palette print`. The app's own names work too. 

339PALETTE_SLUGS = { 

340 "default": "Default (colourblind-safe)", 

341 "print": "Print / greyscale", 

342 "high-contrast": "High contrast", 

343} 

344_PALETTE_ALIASES = { 

345 "colourblindsafe": "Default (colourblind-safe)", 

346 "colorblindsafe": "Default (colourblind-safe)", 

347 # #374: the names the app shows (US spelling, `PALETTE_LABELS`). 

348 "defaultcolorblindsafe": "Default (colourblind-safe)", 

349 "greyscale": "Print / greyscale", 

350 "grayscale": "Print / greyscale", 

351 "printgrayscale": "Print / greyscale", 

352} 

353 

354 

355def normalize_palette(value: object) -> str: 

356 """The palette ``value`` names — its app name, short name (``print``) or 

357 any spelling of either — else ``ValueError`` listing them.""" 

358 from .constants import PALETTES, palette_label 

359 

360 key = _choice_key(value) 

361 for name in PALETTES: 

362 if _choice_key(name) == key: 

363 return name 

364 for slug, name in PALETTE_SLUGS.items(): 

365 if _choice_key(slug) == key: 

366 return name 

367 if key in _PALETTE_ALIASES: 

368 return _PALETTE_ALIASES[key] 

369 raise ValueError( 

370 f"Unknown palette {value!r}; choose one of {', '.join(PALETTE_SLUGS)} " 

371 f"({', '.join(palette_label(name) for name in PALETTES)})." 

372 ) 

373 

374 

375def palette_slug(name: str) -> str: 

376 """A palette's short name, for a command line.""" 

377 return next((slug for slug, full in PALETTE_SLUGS.items() if full == name), name) 

378 

379 

380def normalize_option_values(options: Mapping[str, Any]) -> dict[str, Any]: 

381 """``options`` with every enumerated value read by 

382 `normalize_option_value` (inside ``style_a`` / ``style_b`` too).""" 

383 out = {name: normalize_option_value(name, value) for name, value in options.items()} 

384 for side in ("style_a", "style_b"): 

385 style = out.get(side) 

386 if isinstance(style, Mapping): 

387 out[side] = { 

388 key: normalize_option_value(key, value) for key, value in style.items() 

389 } 

390 return out 

391 

392 

393def _sample_colorscale_colors( 

394 values, colorscale: str, cmin: float | None, cmax: float | None 

395) -> object: 

396 """Map numeric values to concrete CSS colours via a named Plotly colorscale. 

397 

398 Used for hollow markers: Plotly can render a colorscale on a marker *fill* 

399 but not on its outline, so the gradient is sampled to literal colours that 

400 can sit on ``marker.line.color``. Falls back to a single outline colour if 

401 sampling is unavailable. 

402 """ 

403 try: 

404 from plotly.colors import sample_colorscale 

405 except Exception: 

406 return FIX_MARKER_OUTLINE 

407 vals = pd.to_numeric(pd.Series(list(values)), errors="coerce") 

408 lo = float(cmin) if cmin is not None else float(vals.min()) 

409 hi = float(cmax) if cmax is not None else float(vals.max()) 

410 if not np.isfinite(lo) or not np.isfinite(hi) or hi <= lo: 

411 norm = [0.5] * len(vals) 

412 else: 

413 norm = ((vals.clip(lo, hi) - lo) / (hi - lo)).fillna(0.0).tolist() 

414 try: 

415 return sample_colorscale(colorscale, norm) 

416 except Exception: 

417 return FIX_MARKER_OUTLINE 

418 

419 

420def _make_hollow(marker: dict) -> dict: 

421 """Return a copy of a fixation marker dict rendered as outline-only. 

422 

423 The fill colour is moved onto the outline (so the colour is preserved) and 

424 the fill itself is made transparent. Numeric colorscale colours are sampled 

425 to concrete CSS colours because Plotly can't map a colorscale onto an 

426 outline. The colorbar is dropped in hollow mode (it needs the coloured fill). 

427 """ 

428 m = dict(marker) 

429 color = m.get("color") 

430 colorscale = m.get("colorscale") 

431 if colorscale is not None and color is not None and not isinstance(color, str): 

432 outline_color = _sample_colorscale_colors( 

433 color, colorscale, m.get("cmin"), m.get("cmax") 

434 ) 

435 else: 

436 outline_color = color if color is not None else FIX_MARKER_OUTLINE 

437 line = dict(m.get("line") or {}) 

438 line["color"] = outline_color 

439 line["width"] = HOLLOW_OUTLINE_WIDTH 

440 m["line"] = line 

441 m["color"] = "rgba(0,0,0,0)" 

442 m["colorscale"] = None 

443 m["showscale"] = False 

444 m["colorbar"] = None 

445 return m 

446 

447 

448#: The numeric columns a figure places, sizes or times things by. 

449_PLOTTED_NUMBERS = ("x", "y", "width", "height", "duration_ms", "timestamp_ms") 

450#: A word whose box has an infinite edge has no box to draw. 

451_WORD_BOX_COLUMNS = ("x", "y", "width", "height") 

452 

453 

454def _finite_for_plotting( 

455 frame: pd.DataFrame | None, 

456 extra: Iterable[str] = (), 

457 *, 

458 drop_on: Iterable[str] = (), 

459) -> pd.DataFrame | None: 

460 """``frame`` with ±inf in its plotted numbers read as missing. 

461 

462 Data checks reports such rows and keeps them in every table; the figure 

463 cannot place, size or time them, so it draws them as it draws a missing 

464 value — no marker and no saccade to or from them, the smallest marker for a 

465 duration, a replay timed by durations. A row infinite in a ``drop_on`` 

466 column is left out (a word box). Copies only when there is one.""" 

467 if frame is None or frame.empty: 

468 return frame 

469 numbers = { 

470 c: pd.to_numeric(frame[c], errors="coerce").astype(float) 

471 for c in dict.fromkeys((*_PLOTTED_NUMBERS, *extra)) 

472 if c in frame.columns 

473 } 

474 infinite = {c: np.isinf(v) for c, v in numbers.items()} 

475 infinite = {c: m for c, m in infinite.items() if m.any()} 

476 if not infinite: 

477 return frame 

478 out = frame.copy() 

479 drop = pd.Series(False, index=frame.index) 

480 for column, mask in infinite.items(): 

481 out[column] = numbers[column].mask(mask) 

482 if column in drop_on: 

483 drop |= mask 

484 return out[~drop] if drop.any() else out 

485 

486 

487def _compute_axis_ranges( 

488 canvas_width: int, 

489 canvas_height: int, 

490 *frames_with_xy: tuple[pd.DataFrame | None, str, str], 

491 word_frames: Iterable[pd.DataFrame] = (), 

492 fit_to_monitor: bool = False, 

493) -> tuple[list, list, float | None, float | None, float | None, float | None]: 

494 """Compute padded x/y ranges from any number of (frame, x_col, y_col) tuples. 

495 

496 word_frames contribute box-extent bounds: x, x+width and y, y+height. 

497 Falls back to (0..canvas_width, canvas_height..0) when there's no data. 

498 Returns: x_range, y_range (y inverted), and the unpadded mins/maxs. 

499 

500 With ``fit_to_monitor`` the range always spans the full virtual monitor 

501 (0..canvas_width, canvas_height..0) regardless of where the data sits, so the 

502 whole presentation screen is shown and the scanpath appears at its true 

503 on-monitor position rather than the view cropping to the data extent. The 

504 returned data mins/maxs still describe the actual data (they size the 

505 interpolated heatmap grid), so only the visible window changes. 

506 """ 

507 x_candidates: list = [] 

508 y_candidates: list = [] 

509 

510 for df, x_col, y_col in frames_with_xy: 

511 if df is None or df.empty: 

512 continue 

513 if x_col in df.columns: 

514 x_candidates.extend([df[x_col].min(), df[x_col].max()]) 

515 if y_col in df.columns: 

516 y_candidates.extend([df[y_col].min(), df[y_col].max()]) 

517 

518 for df in word_frames: 

519 if df is None or df.empty: 

520 continue 

521 x_candidates.extend([df["x"].min(), (df["x"] + df["width"]).max()]) 

522 y_candidates.extend([df["y"].min(), (df["y"] + df["height"]).max()]) 

523 

524 x_range = [0, canvas_width] 

525 y_range = [canvas_height, 0] 

526 if not x_candidates or not y_candidates: 

527 return x_range, y_range, None, None, None, None 

528 

529 x_min = float(np.nanmin(x_candidates)) 

530 x_max = float(np.nanmax(x_candidates)) 

531 y_min = float(np.nanmin(y_candidates)) 

532 y_max = float(np.nanmax(y_candidates)) 

533 

534 if fit_to_monitor: 

535 # Show the whole monitor; the scanpath keeps its true on-screen position. 

536 # Real data mins/maxs are still returned (heatmap-grid extent). 

537 return [0, canvas_width], [canvas_height, 0], x_min, x_max, y_min, y_max 

538 

539 x_span = max(x_max - x_min, 1.0) 

540 y_span = max(y_max - y_min, 1.0) 

541 pad_x = max(CANVAS_PAD_MIN_PX, CANVAS_PAD_FRACTION * x_span) 

542 pad_y = max(CANVAS_PAD_MIN_PX, CANVAS_PAD_FRACTION * y_span) 

543 x_range = [x_min - pad_x, x_max + pad_x] 

544 y_range = [y_max + pad_y, y_min - pad_y] 

545 return x_range, y_range, x_min, x_max, y_min, y_max 

546 

547 

548@dataclass(frozen=True) 

549class CoordinateGridTicks: 

550 """Deterministic, zero-anchored screen-coordinate tick contract.""" 

551 

552 major_spacing: float 

553 minor_spacing: float 

554 x_values: tuple[float, ...] 

555 x_labels: tuple[str, ...] 

556 y_values: tuple[float, ...] 

557 y_labels: tuple[str, ...] 

558 

559 

560def _nice_grid_spacing(raw: float) -> float: 

561 """Round a positive interval up to the 1/2/5×10ⁿ sequence.""" 

562 exponent = math.floor(math.log10(max(raw, 1e-12))) 

563 unit = 10.0**exponent 

564 normalized = raw / unit 

565 for candidate in (1.0, 2.0, 5.0, 10.0): 

566 if normalized <= candidate: 

567 return candidate * unit 

568 return 10.0 * unit 

569 

570 

571def _anchored_grid_values(lo: float, hi: float, spacing: float) -> tuple[float, ...]: 

572 """Ticks clipped to ``lo..hi`` and anchored at global screen coordinate 0.""" 

573 lo, hi = min(lo, hi), max(lo, hi) 

574 first = math.ceil((lo - spacing * 1e-9) / spacing) 

575 last = math.floor((hi + spacing * 1e-9) / spacing) 

576 count = max(0, last - first + 1) 

577 if count > 10_000: 

578 raise ValueError( 

579 "Coordinate-grid spacing creates more than 10,000 ticks; choose a larger interval." 

580 ) 

581 return tuple(round(index * spacing, 10) for index in range(first, last + 1)) 

582 

583 

584def _grid_label(value: float) -> str: 

585 return f"{value:g}" 

586 

587 

588def coordinate_grid_ticks( 

589 x_range: Sequence[float], 

590 y_range: Sequence[float], 

591 *, 

592 spacing: float | None = None, 

593 rendered_width: int = 900, 

594 rendered_height: int = 650, 

595 monitor_bounds: tuple[float, float, float, float] | None = None, 

596) -> CoordinateGridTicks: 

597 """Return screen-X/Y grid ticks without changing either visible range. 

598 

599 ``spacing=None`` chooses one shared 1/2/5×10ⁿ major interval for both axes. 

600 Manual intervals stay exact. Labels are thinned when the rendered display 

601 could not fit them, while the underlying tick/grid positions remain stable. 

602 ``monitor_bounds`` is accepted to make the coordinate frame explicit; ticks 

603 are intentionally clipped to the *visible* ranges and always anchored at 

604 screen zero, so cropped/full-monitor transitions cannot shift the grid. 

605 """ 

606 if len(x_range) != 2 or len(y_range) != 2: 

607 raise ValueError("Coordinate-grid ranges must each contain two values.") 

608 if monitor_bounds is not None and len(monitor_bounds) != 4: 

609 raise ValueError("monitor_bounds must be (x_min, x_max, y_min, y_max).") 

610 x_span = abs(float(x_range[1]) - float(x_range[0])) 

611 y_span = abs(float(y_range[1]) - float(y_range[0])) 

612 if spacing is None: 

613 target_x = x_span / max(float(rendered_width) / 90.0, 1.0) 

614 target_y = y_span / max(float(rendered_height) / 70.0, 1.0) 

615 major = _nice_grid_spacing(max(target_x, target_y, 1.0)) 

616 else: 

617 major = float(spacing) 

618 if not math.isfinite(major) or major <= 0: 

619 raise ValueError( 

620 "Coordinate-grid spacing must be a positive finite number." 

621 ) 

622 x_values = _anchored_grid_values(float(x_range[0]), float(x_range[1]), major) 

623 y_values = _anchored_grid_values(float(y_range[0]), float(y_range[1]), major) 

624 

625 def _labels( 

626 values: tuple[float, ...], pixels: int, minimum_px: int 

627 ) -> tuple[str, ...]: 

628 capacity = max(int(pixels) // minimum_px, 1) 

629 stride = max(1, math.ceil(len(values) / capacity)) 

630 return tuple( 

631 _grid_label(value) if index % stride == 0 else "" 

632 for index, value in enumerate(values) 

633 ) 

634 

635 return CoordinateGridTicks( 

636 major_spacing=major, 

637 minor_spacing=major / 5.0, 

638 x_values=x_values, 

639 x_labels=_labels(x_values, rendered_width, 68), 

640 y_values=y_values, 

641 y_labels=_labels(y_values, rendered_height, 42), 

642 ) 

643 

644 

645_GRID_LEFT_RESERVE_PX = 52 

646_GRID_BOTTOM_RESERVE_PX = 36 

647_GRID_TICK_FONT_PX = 18 # remains legible after true-scale responsive downscaling 

648 

649 

650def _coordinate_grid_axis_options( 

651 ticks: CoordinateGridTicks, *, axis: str 

652) -> dict[str, Any]: 

653 """Plotly axis options for one restrained major/minor coordinate grid.""" 

654 values = ticks.x_values if axis == "x" else ticks.y_values 

655 labels = ticks.x_labels if axis == "x" else ticks.y_labels 

656 return dict( 

657 showticklabels=True, 

658 showgrid=True, 

659 tickmode="array", 

660 tickvals=list(values), 

661 ticktext=list(labels), 

662 ticks="outside", 

663 ticklen=4, 

664 tickwidth=1, 

665 tickcolor="#667085", 

666 tickfont=dict(size=_GRID_TICK_FONT_PX, color="#475467"), 

667 gridcolor="rgba(71,84,103,0.22)", 

668 gridwidth=1, 

669 zeroline=True, 

670 zerolinecolor="rgba(16,24,40,0.42)", 

671 zerolinewidth=1.25, 

672 minor=dict( 

673 showgrid=True, 

674 dtick=ticks.minor_spacing, 

675 gridcolor="rgba(71,84,103,0.09)", 

676 gridwidth=0.5, 

677 ticks="", 

678 ), 

679 ) 

680 

681 

682def _apply_coordinate_grid_axes( 

683 xaxis: dict, 

684 yaxis: dict, 

685 *, 

686 show: bool, 

687 spacing: float | None, 

688 x_range: Sequence[float], 

689 y_range: Sequence[float], 

690 rendered_width: int, 

691 rendered_height: int, 

692) -> None: 

693 """Mutate spatial axis dicts only when the optional grid is enabled.""" 

694 if not show: 

695 return 

696 ticks = coordinate_grid_ticks( 

697 x_range, 

698 y_range, 

699 spacing=spacing, 

700 rendered_width=rendered_width, 

701 rendered_height=rendered_height, 

702 ) 

703 xaxis.update(_coordinate_grid_axis_options(ticks, axis="x")) 

704 yaxis.update(_coordinate_grid_axis_options(ticks, axis="y")) 

705 

706 

707# Cap the *fixed* render size so the true-to-scale plot (rendered at exactly 

708# these pixels via tabs._render_true_scale_chart) fits a typical research display 

709# without horizontal scrolling. Aspect ratio is preserved when shrinking — both 

710# dims scale together, so boxes/text/fixations keep one true scale. A wider 

711# monitor just leaves margin (the plot is "narrower than the column", never 

712# stretched); a narrower window scrolls rather than distorting. 

713_DISPLAY_MAX_HEIGHT = 690 

714_DISPLAY_MAX_WIDTH = 960 

715 

716 

717def _fit_display_size( 

718 canvas_width: int, 

719 canvas_height: int, 

720 x_range: list, 

721 y_range: list, 

722 spatial_axes: bool, 

723) -> tuple[int, int]: 

724 """Return (width, height) for `fig.update_layout` so the plot fits onscreen. 

725 

726 With `scaleanchor="x", scaleratio=1` the plot domain shrinks to the data 

727 aspect ratio, leaving large blank vertical strips when the figure box is 

728 the full monitor height. We match the figure box to the actual plot 

729 domain — and additionally clamp both dims so the whole plot fits in one 

730 viewport without scrolling. Falls back to (canvas_w, canvas_h) when axes 

731 aren't spatial or the data range is degenerate. 

732 """ 

733 if not spatial_axes: 

734 return canvas_width, canvas_height 

735 x_span = x_range[1] - x_range[0] 

736 y_span = y_range[0] - y_range[1] # y_range is inverted [y_max, y_min] 

737 if x_span <= 0 or y_span <= 0: 

738 return canvas_width, canvas_height 

739 aspect = x_span / y_span 

740 w, h = canvas_width, round(canvas_width / aspect) 

741 # Shrink (preserving aspect) until both dims fit the viewport caps. 

742 if h > _DISPLAY_MAX_HEIGHT: 

743 h = _DISPLAY_MAX_HEIGHT 

744 w = round(h * aspect) 

745 if w > _DISPLAY_MAX_WIDTH: 

746 w = _DISPLAY_MAX_WIDTH 

747 h = round(w / aspect) 

748 return max(w, 100), max(h, 100) 

749 

750 

751#: The two colour bars' settings (fixations', heatmap's) and their defaults — 

752#: the one list the app's settings dicts and saved config copy them by. 

753COLORBAR_DEFAULTS: dict = { 

754 f.name: f.default 

755 for f in fields(FigureSettings) 

756 if f.name.endswith("_colorbar") or "_colorbar_" in f.name 

757} 

758 

759# Extra figure size (px) reserved OUTSIDE the equal-aspect plot region for a 

760# right-side colorbar or a top legend. Without this, Plotly's automargin shrinks 

761# the scaleanchor'd plot domain to fit them — and because the word labels are 

762# sized for the full fitted_w x fitted_h plot region, a shrunken plot leaves the 

763# text overflowing the boxes (the "colorbar / discrete colour legend shrinks the 

764# plot and breaks the aspect ratio" bug). Mirroring the _CONTROLS_MARGIN_PX trick 

765# the animation uses for its transport controls, we instead grow the figure by 

766# the reserve and pin it as an explicit margin, so the plot region stays exactly 

767# fitted_w x fitted_h whether or not a colorbar/legend is shown. 

768_COLORBAR_RESERVE_PX = 160 

769_LEGEND_RESERVE_PX = 60 

770# Top reserve for the overlay-comparison figure's title + A/B legend (same idea 

771# as _LEGEND_RESERVE_PX, but the title needs a touch more room). 

772_OVERLAY_TOP_PX = 64 

773#: UX-172: the Compare A/B legend is the only thing naming the two readings, so 

774#: it reads larger than the figure's body text (overlay, split and co-animation). 

775_COMPARE_LEGEND_FONT_SCALE = 1.3 

776 

777 

778def _compare_legend_font(base_font_size, font_family=None) -> dict: 

779 """The A/B legend's font in every Compare figure (UX-172).""" 

780 font = {"size": round(float(base_font_size or 16) * _COMPARE_LEGEND_FONT_SCALE)} 

781 if font_family: 

782 font["family"] = font_family 

783 return font 

784 

785 

786# A horizontal colorbar sits below the plot, so it reserves bottom (not right). 

787_COLORBAR_BOTTOM_PX = 96 

788 

789 

790def _decoration_margins( 

791 fitted_w: int, 

792 fitted_h: int, 

793 *, 

794 legend: bool, 

795 colorbar_right: bool = False, 

796 colorbar_below: bool = False, 

797 bottom: int = 0, 

798 coordinate_grid: bool = False, 

799) -> dict: 

800 """Grow a spatial figure so a right/bottom colorbar + top legend sit in 

801 reserved margin instead of stealing space from the equal-aspect plot region. 

802 

803 Returns ``{"width", "height", "margin"}`` for ``fig.update_layout``: the plot 

804 region stays ``fitted_w x fitted_h`` (so the true-to-scale word labels keep 

805 matching the boxes); ``bottom`` reserves additional space below the plot for 

806 transport controls (the animation figure); ``colorbar_right`` / 

807 ``colorbar_below`` reserve room for a vertical / horizontal colour bar — 

808 both, when the fixations' and the heatmap's bars point different ways. 

809 """ 

810 right = _COLORBAR_RESERVE_PX if colorbar_right else 0 

811 cb_bottom = _COLORBAR_BOTTOM_PX if colorbar_below else 0 

812 top = _LEGEND_RESERVE_PX if legend else 0 

813 left = _GRID_LEFT_RESERVE_PX if coordinate_grid else 0 

814 grid_bottom = _GRID_BOTTOM_RESERVE_PX if coordinate_grid else 0 

815 return { 

816 "width": fitted_w + left + right, 

817 "height": fitted_h + top + bottom + cb_bottom + grid_bottom, 

818 "margin": dict(l=left, r=right, t=top, b=bottom + cb_bottom + grid_bottom), 

819 } 

820 

821 

822def _colorbar_reserves(*bars: tuple[bool, str]) -> dict: 

823 """``colorbar_right`` / ``colorbar_below`` for `_decoration_margins`, from 

824 each bar's ``(drawn, orientation)``.""" 

825 return { 

826 "colorbar_right": any(on and o != "Horizontal" for on, o in bars), 

827 "colorbar_below": any(on and o == "Horizontal" for on, o in bars), 

828 } 

829 

830 

831def _colorbar_dict( 

832 title: str, 

833 *, 

834 orientation: str = "Vertical", 

835 tickangle: int = 0, 

836 tickfont_size: int = 12, 

837) -> dict: 

838 """A styled Plotly colorbar dict (vertical right / horizontal below), with 

839 rotatable, sizable tick labels and a slim bar.""" 

840 horizontal = orientation == "Horizontal" 

841 cb = dict( 

842 title=dict( 

843 text=title, 

844 side="top" if horizontal else "right", 

845 font=dict(size=max(10, int(tickfont_size) + 1)), 

846 ), 

847 thickness=14, 

848 tickangle=int(tickangle), 

849 tickfont=dict(size=int(tickfont_size)), 

850 outlinewidth=0, 

851 ) 

852 if horizontal: 

853 cb.update( 

854 orientation="h", 

855 x=0.5, 

856 xanchor="center", 

857 y=-0.04, 

858 yanchor="top", 

859 lenmode="fraction", 

860 len=0.6, 

861 ) 

862 else: 

863 cb.update( 

864 x=1.02, 

865 xanchor="left", 

866 y=0.5, 

867 yanchor="middle", 

868 lenmode="fraction", 

869 len=COLORBAR_LEN_FRACTION, 

870 ) 

871 return cb 

872 

873 

874def _colorbar_owners(fig: go.Figure) -> list: 

875 """Every object drawing a colour bar on ``fig``, in trace order: a trace 

876 with its own scale (``go.Heatmap``) or a trace's marker.""" 

877 owners = [] 

878 for trace in fig.data: 

879 if getattr(trace, "showscale", None): 

880 owners.append(trace) 

881 continue 

882 marker = getattr(trace, "marker", None) 

883 if marker is not None and getattr(marker, "showscale", None): 

884 owners.append(marker) 

885 return owners 

886 

887 

888def _arrange_colorbars(fig: go.Figure) -> None: 

889 """Give each of several colour bars its own place (round-7 review, finding 16). 

890 

891 `_colorbar_dict` puts every bar of one orientation at one spot, so a 

892 heatmap's scale and the fixations' were drawn over each other. With two or 

893 more pointing the same way, vertical bars stand side by side to the right 

894 of the plot and horizontal ones stack below it, the figure growing by the 

895 room they take — the plot region, and so the true-to-scale text, keep 

896 their size. A bar alone in its orientation keeps its geometry exactly (a 

897 vertical and a horizontal bar each already have their reserved margin). 

898 Each bar is its own mapping (variable, units, palette, range): none is 

899 merged into another, so every scale stays readable. 

900 """ 

901 owners = _colorbar_owners(fig) 

902 below = [o for o in owners if o.colorbar.orientation == "h"] 

903 right = [o for o in owners if o.colorbar.orientation != "h"] 

904 layout = fig.layout 

905 if len(below) > 1: 

906 first = below[0].colorbar 

907 tick_px = float(first.tickfont.size or 12) 

908 rotated = abs(float(first.tickangle or 0)) > 30 

909 # Title above the bar, the bar, its tick labels below. 

910 row_px = 56.0 + 2.0 * tick_px + (2.0 * tick_px if rotated else 0.0) 

911 height = float(layout.height or 450) 

912 top, bottom = float(layout.margin.t or 0), float(layout.margin.b or 0) 

913 plot_h = max(height - top - bottom, 1.0) 

914 base_y = float(first.y if first.y is not None else -0.04) 

915 for i, owner in enumerate(below): 

916 owner.colorbar.y = base_y - i * row_px / plot_h 

917 grow = (len(below) - 1) * row_px 

918 fig.update_layout(height=height + grow, margin=dict(b=bottom + grow)) 

919 if len(right) > 1: 

920 first = right[0].colorbar 

921 tick_px = float(first.tickfont.size or 12) 

922 # The bar, its tick labels, then its title read sideways. 

923 step_px = 70.0 + 3.0 * tick_px 

924 # Every spatial builder sizes its figure; Plotly's own default otherwise. 

925 width = float(layout.width or 700) 

926 left, margin_r = float(layout.margin.l or 0), float(layout.margin.r or 0) 

927 plot_w = max(width - left - margin_r, 1.0) 

928 base_x = float(first.x if first.x is not None else 1.02) 

929 for i, owner in enumerate(right): 

930 owner.colorbar.x = base_x + i * step_px / plot_w 

931 new_right = ( 

932 max(margin_r, float(_COLORBAR_RESERVE_PX)) + (len(right) - 1) * step_px 

933 ) 

934 fig.update_layout( 

935 width=width + (new_right - margin_r), margin=dict(r=new_right) 

936 ) 

937 

938 

939# Text in Plotly is sized in screen pixels with no native "data unit" mode, so 

940# to keep word labels true-to-scale we convert a real (monitor-pixel) font size 

941# into the figure's screen pixels using the same scale the boxes/fixations use. 

942_MIN_LABEL_PX = 1.0 

943 

944# Advance-width / em of a monospaced glyph, used to back the box *width* cap that 

945# stops long words from colliding when the on-screen font is a touch wider than 

946# the one the experiment was rendered in. Latin monospace (DejaVu Sans Mono, 

947# Courier, …) ≈ 0.6 em; in a full-width CJK monospace (Noto Sans Mono CJK) the CJK 

948# glyphs are a full square (1.0 em) while Latin glyphs are half-width (0.5 em). 

949# Reading stimuli are monospaced (OneStop, MultiplEYE), so summing per-character 

950# advances recovers a per-word em (≈ the font size) even for mixed CJK+Latin runs 

951# (a Chinese paragraph with an English URL), which a single global aspect can't. 

952_MONO_ASPECT = 0.6 

953_CJK_LATIN_ASPECT = 0.5 

954_FULLWIDTH_ASPECT = 1.0 

955_WIDTH_FIT_MARGIN = 0.92 # leave a sliver of horizontal padding inside each box 

956 

957 

958def _is_fullwidth(ch: str) -> bool: 

959 """Whether ``ch`` is an East-Asian wide / full-width glyph (≈ 1 em advance).""" 

960 o = ord(ch) 

961 return ( 

962 0x1100 <= o <= 0x115F # Hangul Jamo 

963 or 0x2E80 <= o <= 0x303E # CJK radicals / Kangxi / CJK symbols & punct 

964 or 0x3041 <= o <= 0x33FF # Hiragana, Katakana, CJK symbols 

965 or 0x3400 <= o <= 0x4DBF # CJK Unified Ext A 

966 or 0x4E00 <= o <= 0x9FFF # CJK Unified 

967 or 0xA000 <= o <= 0xA4CF # Yi 

968 or 0xAC00 <= o <= 0xD7A3 # Hangul syllables 

969 or 0xF900 <= o <= 0xFAFF # CJK compatibility 

970 or 0xFF00 <= o <= 0xFF60 # full-width forms 

971 or 0xFFE0 <= o <= 0xFFE6 # full-width signs 

972 ) 

973 

974 

975def _latin_advance(words: pd.DataFrame) -> float: 

976 """Per-em advance of *Latin* glyphs for this corpus' font. 

977 

978 In a full-width CJK monospace (Noto Sans Mono CJK — MultiplEYE) Latin glyphs 

979 are half-width (0.5 em); in a plain Latin monospace (Courier/DejaVu — OneStop) 

980 they're ≈ 0.6 em. Detected from whether the labels are CJK-heavy, so a Chinese 

981 corpus' embedded English isn't measured with the wrong cell width. 

982 """ 

983 if "text" not in words.columns: 

984 return _MONO_ASPECT 

985 # dropna first: with the Arrow `str` dtype, `.astype(str)` leaves a NaN as a 

986 # float (it doesn't stringify it), which would break the join + char scan. 

987 text = "".join(words["text"].dropna().astype(str).tolist()) 

988 if not text: 

989 return _MONO_ASPECT 

990 wide = sum(_is_fullwidth(ch) for ch in text) 

991 return _CJK_LATIN_ASPECT if wide >= 0.3 * len(text) else _MONO_ASPECT 

992 

993 

994def _line_pitch(words: pd.DataFrame) -> float | None: 

995 """Median line-to-line distance (data px) of the word boxes. 

996 

997 The true-to-scale font budget is a fraction of the *line pitch* (the gap 

998 between consecutive baselines), not the box height: some corpora (MultiplEYE) 

999 draw AOI boxes tight around the glyph (height ≈ font), while the line slot is 

1000 much taller. OneStop's boxes tile the lines (height == pitch), so this returns 

1001 the same value there and leaves OneStop sizing unchanged. Falls back to the 

1002 median box height when there's only one line or geometry is missing. 

1003 """ 

1004 if words.empty or "y" not in words.columns or "height" not in words.columns: 

1005 return None 

1006 from .measures import cluster_word_lines 

1007 

1008 heights = pd.to_numeric(words["height"], errors="coerce") 

1009 y_center = pd.to_numeric(words["y"], errors="coerce") + heights.fillna(0) / 2.0 

1010 centers = y_center.groupby(cluster_word_lines(words)).median().sort_values() 

1011 if len(centers) >= 2: 

1012 pitch = float(centers.diff().dropna().median()) 

1013 if pitch > 0: 

1014 return pitch 

1015 box_h = float(heights.median()) if heights.notna().any() else None 

1016 return box_h if box_h and box_h > 0 else None 

1017 

1018 

1019def _width_fit_font(words: pd.DataFrame) -> float | None: 

1020 """Largest font (data px) at which every word still fits its box width. 

1021 

1022 Word boxes hug the rendered text, so a word's box width equals the sum of its 

1023 glyph advances. Each glyph advances 1 em (full-width CJK) or ``_latin_advance`` 

1024 em (Latin), so ``box_width / Σ advances`` recovers the em ≈ the font size — per 

1025 word, which is correct even for a CJK word boxed beside a half-width Latin URL 

1026 (a single global aspect would size the line from the narrowest run). The 

1027 tightest words bind, so we take a low quantile (robust to one odd box). For an 

1028 all-Latin corpus this reduces exactly to the old ``(box_width / n) / aspect``. 

1029 Returns None when there's no text/width to measure. 

1030 """ 

1031 if "width" not in words.columns or "text" not in words.columns: 

1032 return None 

1033 latin_adv = _latin_advance(words) 

1034 widths = pd.to_numeric(words["width"], errors="coerce") 

1035 ems = [] 

1036 for text, width in zip(words["text"], widths): 

1037 # Skip a NaN label (Arrow `str` keeps it a float, not "nan") or NaN width — 

1038 # matches the old vectorized path, where both dropped out before the quantile. 

1039 if pd.isna(text) or not np.isfinite(width): 

1040 continue 

1041 units = sum( 

1042 _FULLWIDTH_ASPECT if _is_fullwidth(c) else latin_adv for c in str(text) 

1043 ) 

1044 if units > 0: 

1045 ems.append(width / units) 

1046 if not ems: 

1047 return None 

1048 tight = float(pd.Series(ems).quantile(0.05)) 

1049 return tight * _WIDTH_FIT_MARGIN if tight > 0 else None 

1050 

1051 

1052# A monospace word box is its glyphs plus the same padding on every word — half 

1053# the gap to each neighbour (a space, plus any extra word spacing). Line-start 

1054# words carry only the right half, and a fixation cross's box none, so a box 

1055# agrees when it carries the full padding, half of it or none, and most of a 

1056# trial's boxes must agree before the font is read off them. 

1057_PADDED_MIN_AGREEMENT = 0.8 

1058_PADDED_MIN_WORDS = 5 

1059_PADDED_TOL = 0.02 # of one cell, or 1.5 data px, whichever is larger 

1060 

1061 

1062def _padded_monospace_font(words: pd.DataFrame) -> float | None: 

1063 """The font (data px) of a monospace layout whose boxes are padded alike: 

1064 the character cell (:func:`_padded_monospace_layout`) over the font's 

1065 advance. ``None`` when the layout is not one.""" 

1066 layout = _padded_monospace_layout(words) 

1067 return None if layout is None else layout[0] / _latin_advance(words) 

1068 

1069 

1070def _padded_monospace_layout(words: pd.DataFrame) -> tuple[float, float] | None: 

1071 """``(cell, padding)`` in data px of a monospace layout padded alike. 

1072 

1073 Each box is ``n_chars`` cells plus a padding shared by every word, so the 

1074 cell is the slope of box width against word length and the font is that 

1075 cell over the font's advance — exact, with no margin to guess: the padding 

1076 *is* the margin. The slope is read from the *regular* boxes only, leaving 

1077 out each line's first box (it carries only the right half of the padding) 

1078 and any box starting where another does (a fixation cross over the first 

1079 word), since on a short screen those few would tip a median. ``None`` when 

1080 the boxes do not agree (proportional fonts, too few words or lengths, 

1081 full-width text), so the caller falls back to :func:`_width_fit_font`. 

1082 """ 

1083 if not {"x", "width", "text"} <= set(words.columns): 

1084 return None 

1085 frame = words.dropna(subset=["text"]) 

1086 text = frame["text"].astype(str) 

1087 if any(_is_fullwidth(ch) for ch in "".join(text.tolist())): 

1088 return None 

1089 chars = text.str.len().to_numpy(dtype=float) 

1090 x = pd.to_numeric(frame["x"], errors="coerce").to_numpy(dtype=float) 

1091 width = pd.to_numeric(frame["width"], errors="coerce").to_numpy(dtype=float) 

1092 ok = (chars > 0) & np.isfinite(width) & (width > 0) & np.isfinite(x) 

1093 if ok.sum() < _PADDED_MIN_WORDS: 

1094 return None 

1095 regular = ok.copy() 

1096 if {"y", "height"} <= set(frame.columns): 

1097 from .measures import cluster_word_lines 

1098 

1099 lines = np.asarray(cluster_word_lines(frame)) 

1100 for line in pd.unique(lines[ok]): 

1101 on_line = np.flatnonzero(ok & (lines == line)) 

1102 regular[on_line[x[on_line] <= x[on_line].min()]] = False 

1103 shared_start = pd.Series(x).duplicated(keep=False).to_numpy() 

1104 regular &= ~shared_start 

1105 lengths = np.unique(chars[regular]) 

1106 if regular.sum() < _PADDED_MIN_WORDS - 1 or len(lengths) < 2: 

1107 return None 

1108 typical = np.array([np.median(width[regular & (chars == n)]) for n in lengths]) 

1109 i, j = np.triu_indices(len(lengths), k=1) 

1110 cell = float(np.median((typical[j] - typical[i]) / (lengths[j] - lengths[i]))) 

1111 if not np.isfinite(cell) or cell <= 0: 

1112 return None 

1113 pad = float(np.median(width[regular] - chars[regular] * cell)) 

1114 if pad < -0.5 * cell: 

1115 return None 

1116 tol = max(1.5, _PADDED_TOL * cell) 

1117 glyphs = chars[ok] * cell 

1118 agree = ( 

1119 (np.abs(width[ok] - glyphs - pad) <= tol) 

1120 | (np.abs(width[ok] - glyphs - pad / 2) <= tol) 

1121 | (np.abs(width[ok] - glyphs) <= tol) 

1122 ) 

1123 if agree.mean() < _PADDED_MIN_AGREEMENT: 

1124 return None 

1125 return cell, pad 

1126 

1127 

1128def _word_label_x(words: pd.DataFrame) -> np.ndarray: 

1129 """Where each word label is centred: its box's middle (BUG-97) — except a 

1130 line's first word in a padded monospace layout. 

1131 

1132 There the box carries only the right half of the padding (the line starts 

1133 at the text), so the word sat flush with the box's left edge in the 

1134 experiment; centring it in the box shifted it right by a quarter of the gap. 

1135 Such a word is centred on its own letters instead, starting at ``x``. 

1136 Right-to-left words and every other layout keep the box's middle. 

1137 """ 

1138 from .measures import cluster_word_lines, word_box_bounds 

1139 

1140 x0, _, x1, _ = word_box_bounds(words) 

1141 label_x = (x0 + x1) / 2.0 

1142 layout = _padded_monospace_layout(words) if len(words) else None 

1143 if layout is None or not {"y", "height"} <= set(words.columns): 

1144 return label_x 

1145 cell, pad = layout 

1146 chars = words["text"].astype(str).str.len().to_numpy(dtype=float) 

1147 width = x1 - x0 

1148 half_padded = np.abs(width - chars * cell - pad / 2) <= max(1.5, _PADDED_TOL * cell) 

1149 rtl = words.get("right_to_left") 

1150 ltr = ( 

1151 np.ones(len(words), dtype=bool) 

1152 if rtl is None 

1153 else ~rtl.fillna(False).astype(bool).to_numpy() 

1154 ) 

1155 lines = np.asarray(cluster_word_lines(words)) 

1156 first = np.zeros(len(words), dtype=bool) 

1157 for line in pd.unique(lines): 

1158 on_line = np.flatnonzero(lines == line) 

1159 first[on_line[x0[on_line] <= np.nanmin(x0[on_line])]] = True 

1160 flush = first & half_padded & ltr & (chars > 0) 

1161 label_x = label_x.copy() 

1162 label_x[flush] = x0[flush] + chars[flush] * cell / 2.0 

1163 return label_x 

1164 

1165 

1166def _display_scale(x_range: list, y_range: list, fitted_w: int, fitted_h: int) -> float: 

1167 """Screen px per data unit for a fixed-size, equal-aspect spatial plot. 

1168 

1169 With ``scaleratio=1`` the x and y mappings are identical; we take the min so 

1170 rounding can never make text/markers sized through this overflow the boxes. 

1171 Returns 1.0 for degenerate ranges. 

1172 """ 

1173 x_span = x_range[1] - x_range[0] 

1174 y_span = y_range[0] - y_range[1] # y_range is inverted [y_max, y_min] 

1175 if x_span <= 0 or y_span <= 0: 

1176 return 1.0 

1177 return min(fitted_w / x_span, fitted_h / y_span) 

1178 

1179 

1180def _word_label_font_px( 

1181 words: pd.DataFrame, 

1182 *, 

1183 scale: float, 

1184 line_spacing: float, 

1185 manual_font_px: float, 

1186 scale_text_to_boxes: bool, 

1187) -> float: 

1188 """Word-label font size in *screen* px so the text stays true-to-scale. 

1189 

1190 The experiment's font lives in monitor pixels. To keep the rendered glyphs 

1191 the same physical fraction of the (data-space) word boxes at any display 

1192 size, the font is expressed in data px and multiplied by ``scale`` (screen 

1193 px per data unit): 

1194 

1195 - ``scale_text_to_boxes`` (default): one line of text fills 

1196 ``1 / line_spacing`` of the **line pitch** (the median line-to-line distance 

1197 from the data, see :func:`_line_pitch`) — *not* the raw box height, which is 

1198 only equal to the pitch when the boxes tile the lines. For OneStop the boxes 

1199 tile the lines (pitch == height) and ``line_spacing == 3`` (one blank line 

1200 above + below), so the budget is height / 3 as before; for corpora whose AOI 

1201 boxes hug the glyph (MultiplEYE), the pitch is the right, larger budget. The 

1202 size is *also* capped so the longest words still fit their box width (see 

1203 :func:`_width_fit_font`), which keeps the font from colliding; the smaller of 

1204 the two wins. 

1205 **Except** when the boxes hold their word plus the same padding, half the 

1206 gap to each neighbour (:func:`_padded_monospace_font`): then the font is 

1207 read off the boxes exactly, 

1208 and neither the line-spacing guess nor the fit's safety margin applies — 

1209 each shrank such text by its own few percent. 

1210 - otherwise / no usable boxes: ``manual_font_px`` is treated as the real 

1211 monitor font size and scaled the same way. 

1212 """ 

1213 font_data_px = float(manual_font_px) 

1214 exact = _padded_monospace_font(words) if scale_text_to_boxes else None 

1215 if exact: 

1216 font_data_px = exact 

1217 elif scale_text_to_boxes and not words.empty and "height" in words.columns: 

1218 pitch = _line_pitch(words) 

1219 height_fit = pitch / line_spacing if (pitch and line_spacing > 0) else None 

1220 width_fit = _width_fit_font(words) 

1221 candidates = [c for c in (height_fit, width_fit) if c and c > 0] 

1222 if candidates: 

1223 font_data_px = min(candidates) 

1224 return max(font_data_px * scale, _MIN_LABEL_PX) 

1225 

1226 

1227_QUALITATIVE_PALETTE = [ 

1228 "#1f77b4", 

1229 "#ff7f0e", 

1230 "#2ca02c", 

1231 "#d62728", 

1232 "#9467bd", 

1233 "#8c564b", 

1234 "#e377c2", 

1235 "#7f7f7f", 

1236 "#bcbd22", 

1237 "#17becf", 

1238] 

1239 

1240 

1241def _resolve_marker_colors( 

1242 color_data: pd.Series | None, 

1243 is_numeric_color: bool, 

1244 uniform_color: str = DEFAULT_FIXATION_COLOR, 

1245) -> tuple[object, list]: 

1246 """Return (marker_color, category_legend) for the fixation scatter trace. 

1247 

1248 - Numeric color_data is passed straight through (Plotly maps it via colorscale). 

1249 - Categorical color_data is mapped to a discrete palette so the picker has 

1250 visible effect; the returned legend is a list of (category, hex) pairs the 

1251 caller can render as legend-only scatter traces. 

1252 - No color_data (VIZ-17's uniform default, or a `color_by` column that isn't 

1253 in the frame) paints every marker ``uniform_color``. 

1254 """ 

1255 if color_data is None: 

1256 return uniform_color, [] 

1257 if is_numeric_color: 

1258 return color_data, [] 

1259 series = color_data.fillna("(missing)").astype(str) 

1260 unique_vals = list(pd.unique(series)) 

1261 cat_to_color = { 

1262 val: _QUALITATIVE_PALETTE[i % len(_QUALITATIVE_PALETTE)] 

1263 for i, val in enumerate(unique_vals) 

1264 } 

1265 marker_color = [cat_to_color[val] for val in series] 

1266 legend = [(val, cat_to_color[val]) for val in unique_vals] 

1267 return marker_color, legend 

1268 

1269 

1270def _fixation_category_labels( 

1271 fixations: pd.DataFrame, 

1272 words: pd.DataFrame | None, 

1273 color_by: str | None, 

1274 color_by_line: bool = False, 

1275) -> pd.Series | None: 

1276 """The discrete label each fixation is coloured by, or ``None`` when the 

1277 colouring is not discrete (uniform, numeric, or a column the frame lacks). 

1278 

1279 The same labels the static figure draws: ``"Line N"`` / ``"Out of bounds"`` 

1280 for colour-by-line, against ``words``' own geometry; a categorical column's 

1281 values as strings, ``"(missing)"`` for a gap. Aligned to ``fixations``. 

1282 """ 

1283 if fixations.empty: 

1284 return None 

1285 if color_by_line or color_by == "line": 

1286 if words is None or words.empty: 

1287 return None 

1288 from .measures import assign_fixation_lines 

1289 

1290 line_ids = assign_fixation_lines(fixations, words) 

1291 return line_ids.map( 

1292 lambda v: f"Line {int(v) + 1}" if pd.notna(v) else "Out of bounds" 

1293 ) 

1294 if ( 

1295 not color_by 

1296 or color_by == UNIFORM_COLOR_FIELD 

1297 or color_by not in fixations.columns 

1298 or pd.api.types.is_numeric_dtype(fixations[color_by]) 

1299 ): 

1300 return None 

1301 return fixations[color_by].fillna("(missing)").astype(str) 

1302 

1303 

1304def _shared_category_colors( 

1305 labels: Sequence[pd.Series | None], 

1306 avoid: Iterable[str] = (), 

1307) -> tuple[list[list[str] | None], list[tuple[str, str]]]: 

1308 """One category→colour mapping across several scanpaths (Compare, the dual 

1309 replay), so a category wears the same colour on A and on B. 

1310 

1311 Categories are numbered in order of first appearance, A's before B's. 

1312 ``avoid`` names the scanpaths' own colours, which outline their markers: a 

1313 category never fills in one, or that scanpath's outline would vanish into 

1314 its fill. Returns each scanpath's per-row colours (``None`` where it had no 

1315 labels) and the shared legend. 

1316 """ 

1317 present = [series for series in labels if series is not None] 

1318 if not present: 

1319 return [None] * len(labels), [] 

1320 taken = {str(color).lower() for color in avoid if color} 

1321 palette = [c for c in _QUALITATIVE_PALETTE if c.lower() not in taken] 

1322 palette = palette or list(_QUALITATIVE_PALETTE) 

1323 order = list(pd.unique(pd.concat(present, ignore_index=True))) 

1324 cat_to_color = {val: palette[i % len(palette)] for i, val in enumerate(order)} 

1325 colors = [ 

1326 None if series is None else [cat_to_color[val] for val in series] 

1327 for series in labels 

1328 ] 

1329 return colors, [(val, cat_to_color[val]) for val in order] 

1330 

1331 

1332def _add_category_legend( 

1333 fig: go.Figure, legend: Sequence[tuple[str, str]], color_label: str 

1334) -> None: 

1335 """Legend-only entries naming each colour category (``label: value``), with 

1336 a "… +N more" entry past the qualitative palette's length.""" 

1337 limit = len(_QUALITATIVE_PALETTE) 

1338 for category, color in list(legend)[:limit]: 

1339 fig.add_trace( 

1340 go.Scatter( 

1341 x=[None], 

1342 y=[None], 

1343 mode="markers", 

1344 marker=dict( 

1345 size=10, 

1346 color=color, 

1347 line=dict(color=FIX_MARKER_OUTLINE, width=0.5), 

1348 ), 

1349 name=category 

1350 if color_label == "line" 

1351 else f"{_column_name(color_label)}: {category}", 

1352 showlegend=True, 

1353 hoverinfo="skip", 

1354 meta=_COLORS_LEGEND_META, 

1355 ) 

1356 ) 

1357 if len(legend) > limit: 

1358 fig.add_trace( 

1359 go.Scatter( 

1360 x=[None], 

1361 y=[None], 

1362 mode="markers", 

1363 marker=dict(size=10, color="#cccccc"), 

1364 name=f"… +{len(legend) - limit} more", 

1365 showlegend=True, 

1366 hoverinfo="skip", 

1367 meta=_COLORS_LEGEND_META, 

1368 ) 

1369 ) 

1370 

1371 

1372#: How much larger (font px) the outline layer behind a glyph marker is: a text 

1373#: glyph takes no stroke, so its outline is a second, larger glyph drawn under it. 

1374_GLYPH_OUTLINE_PX = 4.0 

1375#: The outline-only form of each glyph shape, for hollow markers. 

1376_HOLLOW_GLYPHS = {"♥": "♡"} 

1377 

1378 

1379def _glyph_scatter_traces( 

1380 x, y, marker: dict, glyph: str, **top: Any 

1381) -> list[go.Scatter]: 

1382 """Draw a fixation ``marker`` dict as text glyphs (VIZ-15's ♥), bottom first. 

1383 

1384 Plotly's marker-symbol enum has no heart, so every builder draws one as 

1385 text, from the marker dict it would otherwise have used, keeping what that 

1386 dict says: duration→size (an array ``textfont.size``, scaled by 

1387 ``FIXATION_GLYPH_SIZE_SCALE``), the colour (a colorscale is sampled to 

1388 literal colours — ``textfont.color`` takes none), the opacity, and a 

1389 non-default outline (Compare's A/B cue) as a larger glyph underneath in the 

1390 outline colour. A hollow marker becomes the outline glyph (♡) in its 

1391 outline colour. ``top`` goes onto the glyph layer (name, hover, legend); 

1392 the outline layer under it takes no hover and no legend entry. 

1393 

1394 The two halves are separate so a replay can state the layers once 

1395 (:func:`_glyph_layers`) and draw them at each frame's positions 

1396 (:func:`_glyph_layer_traces`) without re-sampling the colours. 

1397 """ 

1398 return _glyph_layer_traces(x, y, _glyph_layers(marker, glyph, len(x)), **top) 

1399 

1400 

1401def _glyph_layers(marker: dict, glyph: str, n: int) -> list[dict]: 

1402 """What :func:`_glyph_scatter_traces` draws, bottom first, minus positions. 

1403 

1404 One dict per layer — ``text`` / ``textfont`` / ``opacity`` — with the 

1405 colours sampled and the sizes scaled, for ``n`` fixations.""" 

1406 sizes = np.asarray(marker.get("size"), dtype=float) * FIXATION_GLYPH_SIZE_SCALE 

1407 sizes = np.broadcast_to(sizes, (n,)) if sizes.ndim == 0 else sizes 

1408 color = marker.get("color") 

1409 line = marker.get("line") or {} 

1410 layers: list[tuple[str, object, np.ndarray]] = [] 

1411 if isinstance(color, str) and color == "rgba(0,0,0,0)": 

1412 layers.append( 

1413 ( 

1414 _HOLLOW_GLYPHS.get(glyph, glyph), 

1415 line.get("color") or FIX_MARKER_OUTLINE, 

1416 sizes, 

1417 ) 

1418 ) 

1419 else: 

1420 if ( 

1421 marker.get("colorscale") is not None 

1422 and color is not None 

1423 and not isinstance(color, str) 

1424 ): 

1425 color = _sample_colorscale_colors( 

1426 color, marker["colorscale"], marker.get("cmin"), marker.get("cmax") 

1427 ) 

1428 elif color is not None and not isinstance(color, str): 

1429 color = list(color) 

1430 outline = line.get("color") 

1431 if isinstance(outline, str) and outline != FIX_MARKER_OUTLINE: 

1432 layers.append((glyph, outline, sizes + _GLYPH_OUTLINE_PX)) 

1433 layers.append((glyph, color, sizes)) 

1434 opacity = float(marker.get("opacity", 1.0)) 

1435 return [ 

1436 dict( 

1437 text=[char] * n, 

1438 textfont=dict(color=layer_color, size=list(layer_sizes)), 

1439 opacity=opacity, 

1440 ) 

1441 for char, layer_color, layer_sizes in layers 

1442 ] 

1443 

1444 

1445def _glyph_layer_traces( 

1446 x, y, layers: list[dict], *, make: Callable = go.Scatter, **top: Any 

1447) -> list: 

1448 """:func:`_glyph_layers`' layers drawn at ``x``/``y``; ``top`` on the last. 

1449 

1450 ``make=dict`` returns the traces unvalidated, for a replay frame (see the 

1451 frame loop in :func:`_render_scanpath_animation`).""" 

1452 traces = [] 

1453 for i, layer in enumerate(layers): 

1454 is_top = i == len(layers) - 1 

1455 traces.append( 

1456 make( 

1457 x=x, 

1458 y=y, 

1459 mode="text", 

1460 text=layer["text"], 

1461 textfont=layer["textfont"], 

1462 textposition="middle center", 

1463 opacity=layer["opacity"], 

1464 **( 

1465 top 

1466 if is_top 

1467 else dict( 

1468 hoverinfo="skip", 

1469 showlegend=False, 

1470 legendgroup=top.get("legendgroup"), 

1471 ) 

1472 ), 

1473 ) 

1474 ) 

1475 return traces 

1476 

1477 

1478def _glyph_colorbar_trace(marker: dict, values) -> go.Scatter | None: 

1479 """The colour bar a glyph marker's numeric colouring would have drawn. 

1480 

1481 A text glyph carries no colorscale, so the bar rides on an invisible 

1482 two-point marker trace pinned to the same range.""" 

1483 if not marker.get("showscale") or marker.get("colorscale") is None: 

1484 return None 

1485 numeric = pd.to_numeric(pd.Series(list(values)), errors="coerce") 

1486 lo = marker.get("cmin") 

1487 hi = marker.get("cmax") 

1488 lo = float(numeric.min()) if lo is None else float(lo) 

1489 hi = float(numeric.max()) if hi is None else float(hi) 

1490 if not (np.isfinite(lo) and np.isfinite(hi)): 

1491 return None 

1492 return go.Scatter( 

1493 x=[None, None], 

1494 y=[None, None], 

1495 mode="markers", 

1496 marker=dict( 

1497 color=[lo, hi], 

1498 colorscale=marker["colorscale"], 

1499 cmin=lo, 

1500 cmax=hi, 

1501 showscale=True, 

1502 colorbar=marker.get("colorbar"), 

1503 size=0, 

1504 ), 

1505 name="color scale", 

1506 showlegend=False, 

1507 hoverinfo="skip", 

1508 ) 

1509 

1510 

1511def _marker_symbol(symbol: str | None) -> str: 

1512 """A ``marker.symbol`` Plotly will accept. 

1513 

1514 The VIZ-15 glyph shapes (♥) aren't in Plotly's symbol enum — the static 

1515 figure draws them as text instead — so anywhere that *must* hand Plotly a 

1516 marker symbol falls back to the default rather than raising. 

1517 """ 

1518 if not symbol or symbol in FIXATION_GLYPH_SYMBOLS: 

1519 return DEFAULT_FIXATION_SYMBOL 

1520 return symbol 

1521 

1522 

1523def _duration_bounds(duration_range) -> tuple[float, float]: 

1524 """The fixed scale's (lo, hi) in ms, ordered and never empty.""" 

1525 lo, hi = sorted(float(v) for v in duration_range) 

1526 lo = max(lo, 1.0) # log needs a positive floor; no fixation is shorter 

1527 return lo, max(hi, lo + 1.0) 

1528 

1529 

1530def _scale_transform(scale: str): 

1531 try: 

1532 return {"sqrt": np.sqrt, "linear": lambda d: d, "log": np.log}[scale] 

1533 except KeyError: 

1534 raise ValueError( 

1535 f"Unknown marker_size_scale {scale!r}; choose one of " 

1536 f"{', '.join(MARKER_SIZE_SCALES)}." 

1537 ) from None 

1538 

1539 

1540def _compute_marker_sizes( 

1541 durations: pd.Series, 

1542 size_range: tuple[int, int] = DEFAULT_MARKER_SIZE_RANGE, 

1543 scale: str = DEFAULT_MARKER_SIZE_SCALE, 

1544 duration_range: tuple[float, float] = DEFAULT_MARKER_DURATION_RANGE, 

1545) -> np.ndarray: 

1546 """Map fixation durations to marker sizes (px diameter). 

1547 

1548 A fixed ``scale`` ("sqrt" / "linear" / "log") maps ``duration_range`` (ms) 

1549 onto ``size_range`` through that curve, the same for every figure: a 

1550 duration at or below the lower bound gets the smallest marker, at or above 

1551 the upper bound the largest, so a fixation's size never depends on the 

1552 other fixations drawn beside it. ``"relative"`` is the original scale — 

1553 linear between this set's own shortest and longest durations — which is 

1554 why it is only comparable within one figure. 

1555 """ 

1556 durations = pd.to_numeric(durations, errors="coerce").fillna(0) 

1557 min_size, max_size = size_range 

1558 if scale == "relative": 

1559 d_min, d_max = float(durations.min()), float(durations.max()) 

1560 if d_max - d_min > 0: 

1561 return np.interp(durations, (d_min, d_max), (min_size, max_size)) 

1562 return np.full(len(durations), (min_size + max_size) / 2) 

1563 transform = _scale_transform(scale) 

1564 lo, hi = _duration_bounds(duration_range) 

1565 clipped = np.clip(durations.to_numpy(dtype=float), lo, hi) 

1566 f_lo, f_hi = float(transform(lo)), float(transform(hi)) 

1567 frac = (transform(clipped) - f_lo) / (f_hi - f_lo) 

1568 return min_size + frac * (max_size - min_size) 

1569 

1570 

1571def _settings_size_scale(settings: FigureSettings) -> dict: 

1572 """The duration-scale keywords of :func:`_compute_marker_sizes`.""" 

1573 return { 

1574 "scale": settings.marker_size_scale, 

1575 "duration_range": settings.marker_duration_range, 

1576 } 

1577 

1578 

1579def _duration_key_references(duration_range) -> list[tuple[float, str]]: 

1580 """The durations a size key draws, each with its label. 

1581 

1582 The two bounds (labelled ``≤`` / ``≥``, because everything beyond them 

1583 clamps) plus the round values 100/200/400/800/1600 ms that fall strictly 

1584 between them — at most three of those, so the key stays compact.""" 

1585 lo, hi = _duration_bounds(duration_range) 

1586 inner = [d for d in (100, 200, 400, 800, 1600) if lo < d < hi][:3] 

1587 if not inner and hi - lo > 2: 

1588 inner = [round((lo + hi) / 2)] 

1589 refs = [(lo, f"≤{lo:g}")] 

1590 refs += [(float(d), f"{d:g}") for d in inner] 

1591 refs.append((hi, f"≥{hi:g} ms")) 

1592 return refs 

1593 

1594 

1595# Name of the size key's label annotations, so the layer split can find them. 

1596_SIZE_KEY_NAME = "duration_size_key" 

1597# The Illustration stamp shares the size key's bottom-right corner. 

1598_ILLUSTRATION_LABEL_NAME = "illustration_label" 

1599 

1600 

1601def _stack_bottom_right(fig: go.Figure) -> None: 

1602 """Lift the Illustration stamp above the duration size key when both are drawn. 

1603 

1604 Both sit in the plot's bottom-right corner, and either can be added first 

1605 (the public builders stamp the label inside ``make_scanpath_figure`` and add 

1606 the key after; the app's replay does the reverse), so each calls this once 

1607 it is on the figure. The key's height is read off its own circles — the 

1608 only pixel-sized circles in paper coordinates — so the stamp clears the 

1609 largest one at any size range.""" 

1610 key_top = [ 

1611 float(sh.y1) 

1612 for sh in fig.layout.shapes or () 

1613 if sh.type == "circle" 

1614 and sh.yref == "paper" 

1615 and sh.ysizemode == "pixel" 

1616 # Only a key inside the bottom-right corner shares the stamp's spot. 

1617 and sh.xanchor == 1 

1618 and sh.yanchor == 0 

1619 and float(sh.x1) <= 0 

1620 and float(sh.y0) >= 0 

1621 ] 

1622 if not key_top: 

1623 return 

1624 for ann in fig.layout.annotations or (): 

1625 if ann.name == _ILLUSTRATION_LABEL_NAME: 

1626 ann.yshift = max(key_top) + 4.0 

1627 

1628 

1629# --- Legend layout (where each legend sits) --------------------------------- 

1630# 

1631# Each legend *kind* can be moved to a spot of its own, sized, and laid out as a 

1632# stack or a row, on top of its own layer's show/hide switch. A kind left on 

1633# "auto" throughout is drawn exactly where it always was: every figure built 

1634# before this setting existed is unchanged. A kind moved anywhere gets its own 

1635# Plotly legend (``legend2`` …), and a spot outside the plot grows the figure on 

1636# that side, so the equal-aspect plot region — and the true-to-scale text with 

1637# it — keeps its size (the same rule as `_decoration_margins`). 

1638 

1639#: The size key's own spot when left on "auto": inside, bottom-right. 

1640_SIZE_KEY_AUTO_POSITION = "bottom-right" 

1641_COLORS_LEGEND_META = "legend:colors" 

1642_LEGEND_IDS = {"compare": "legend2", "saccades": "legend3", "colors": "legend4"} 

1643_LEGEND_GAP_PX = 8 

1644_OUTSIDE = ("above", "below", "left", "right") 

1645_CORNERS = ("top-left", "top-right", "bottom-left", "bottom-right") 

1646 

1647 

1648def normalize_legend_layout(layout: Mapping | None) -> dict: 

1649 """``{kind: {"position", "arrangement", "size"}}`` with every kind present. 

1650 

1651 Unknown kinds, positions and arrangements raise rather than being dropped, 

1652 so a typo in a script or a link can't quietly draw the default. ``size`` 

1653 is the legend's text size in px, or ``None`` for the figure's own. 

1654 """ 

1655 out = { 

1656 kind: {"position": "auto", "arrangement": "auto", "size": None} 

1657 for kind in LEGEND_KINDS 

1658 } 

1659 for kind, spec in dict(layout or {}).items(): 

1660 if kind not in out: 

1661 raise ValueError( 

1662 f"Unknown legend {kind!r}; expected one of {', '.join(LEGEND_KINDS)}." 

1663 ) 

1664 spec = dict(spec or {}) 

1665 unknown = set(spec) - {"position", "arrangement", "size"} 

1666 if unknown: 

1667 raise ValueError( 

1668 f"Unknown legend setting(s) {sorted(unknown)} for {kind!r}; " 

1669 "expected position, arrangement, size." 

1670 ) 

1671 position = str(spec.get("position") or "auto") 

1672 if position not in LEGEND_POSITIONS: 

1673 raise ValueError( 

1674 f"Legend position {position!r} for {kind!r}; expected one of " 

1675 f"{', '.join(LEGEND_POSITIONS)}." 

1676 ) 

1677 arrangement = str(spec.get("arrangement") or "auto") 

1678 if arrangement not in LEGEND_ARRANGEMENTS: 

1679 raise ValueError( 

1680 f"Legend arrangement {arrangement!r} for {kind!r}; expected one " 

1681 f"of {', '.join(LEGEND_ARRANGEMENTS)}." 

1682 ) 

1683 size = spec.get("size") 

1684 if size is not None: 

1685 size = int(size) 

1686 if size <= 0: 

1687 raise ValueError(f"Legend size for {kind!r} must be positive.") 

1688 out[kind] = {"position": position, "arrangement": arrangement, "size": size} 

1689 return out 

1690 

1691 

1692def parse_legend_spec(text: str) -> dict: 

1693 """``"right,stacked,14"`` → ``{"position", "arrangement"?, "size"?}``. 

1694 

1695 The spelling shared by ``render --legend KIND=…``, the ``legend_<kind>`` 

1696 link parameters and the code snippet: a spot first, then an arrangement 

1697 and a text size in either order, each recognised by its value. Raises 

1698 ``ValueError`` on anything else. 

1699 """ 

1700 head, *rest = [part.strip().lower() for part in str(text).split(",")] 

1701 spec: dict = {"position": head or "auto"} 

1702 for part in (p for p in rest if p): 

1703 if part in LEGEND_ARRANGEMENTS: 

1704 spec["arrangement"] = part 

1705 elif part.isdigit(): 

1706 spec["size"] = int(part) 

1707 else: 

1708 raise ValueError( 

1709 f"{part!r} is neither an arrangement " 

1710 f"({', '.join(LEGEND_ARRANGEMENTS[1:])}) nor a text size." 

1711 ) 

1712 normalize_legend_layout({"compare": spec}) # validates the values 

1713 return spec 

1714 

1715 

1716def legend_spec_text(spec: Mapping) -> str: 

1717 """The inverse of :func:`parse_legend_spec`, for a normalized entry.""" 

1718 parts = [str(spec["position"])] 

1719 if spec.get("arrangement", "auto") != "auto": 

1720 parts.append(str(spec["arrangement"])) 

1721 if spec.get("size") is not None: 

1722 parts.append(str(int(spec["size"]))) 

1723 return ",".join(parts) 

1724 

1725 

1726def _legend_is_moved(spec: Mapping) -> bool: 

1727 return ( 

1728 spec["position"] != "auto" 

1729 or spec["arrangement"] != "auto" 

1730 or spec["size"] is not None 

1731 ) 

1732 

1733 

1734def _trace_legend_kind(trace, comparing: bool) -> str: 

1735 if trace.legendgroup == "saccade_type": 

1736 return "saccades" 

1737 if trace.meta == _COLORS_LEGEND_META or not comparing: 

1738 return "colors" 

1739 return "compare" 

1740 

1741 

1742def _legend_extent(names: list, font_px: float, horizontal: bool) -> tuple: 

1743 """A legend's estimated ``(width, height)`` in px, for reserving room.""" 

1744 row = max(font_px * 1.3, 20.0) + 4.0 

1745 widths = [40.0 + 0.6 * font_px * len(str(n)) for n in names] or [0.0] 

1746 if horizontal: 

1747 return sum(widths) + 10.0, row + 10.0 

1748 return max(widths) + 10.0, row * len(names) + 10.0 

1749 

1750 

1751def _plot_px(fig: go.Figure) -> tuple: 

1752 """The plot region's ``(width, height)`` in px, and the margins.""" 

1753 m = fig.layout.margin 

1754 margin = {k: float(getattr(m, k) or 0) for k in ("l", "r", "t", "b")} 

1755 width = float(fig.layout.width or 0) - margin["l"] - margin["r"] 

1756 height = float(fig.layout.height or 0) - margin["t"] - margin["b"] 

1757 return max(width, 1.0), max(height, 1.0), margin 

1758 

1759 

1760def _grow(fig: go.Figure, side: str, px: float) -> None: 

1761 """Add ``px`` of margin on one side, growing the figure by as much, so the 

1762 plot region keeps its size.""" 

1763 if not px or not fig.layout.width or not fig.layout.height: 

1764 return 

1765 m = fig.layout.margin 

1766 setattr(m, side, float(getattr(m, side) or 0) + px) 

1767 if side in ("l", "r"): 

1768 fig.layout.width = float(fig.layout.width) + px 

1769 else: 

1770 fig.layout.height = float(fig.layout.height) + px 

1771 

1772 

1773def apply_legend_layout( 

1774 fig: go.Figure, layout: Mapping | None, *, comparing: bool = False 

1775) -> go.Figure: 

1776 """Move each legend kind the user placed into a Plotly legend of its own. 

1777 

1778 Kinds still on "auto" stay in the figure's default ``legend``; the size key 

1779 is placed by :func:`_add_duration_size_key`, not here. Legends sharing a 

1780 side are laid out one after another along it, and an outside side reserves 

1781 room for the widest (or tallest) of them. 

1782 """ 

1783 specs = normalize_legend_layout(layout) 

1784 moved = { 

1785 kind 

1786 for kind in ("compare", "saccades", "colors") 

1787 if _legend_is_moved(specs[kind]) 

1788 } 

1789 if not moved: 

1790 return fig 

1791 names: dict = {kind: [] for kind in moved} 

1792 titled: set = set() 

1793 for trace in fig.data: 

1794 kind = _trace_legend_kind(trace, comparing) 

1795 if kind not in moved: 

1796 continue 

1797 trace.legend = _LEGEND_IDS[kind] 

1798 if trace.showlegend is False: 

1799 continue 

1800 title = trace.legendgrouptitle.text if trace.legendgrouptitle else None 

1801 if title and trace.legendgroup not in titled: 

1802 titled.add(trace.legendgroup) 

1803 names[kind].append(title) 

1804 if trace.name: 

1805 names[kind].append(trace.name) 

1806 base_font = float((fig.layout.font and fig.layout.font.size) or 12) 

1807 plot_w, plot_h, margin = _plot_px(fig) 

1808 gap = _LEGEND_GAP_PX 

1809 along = {side: 0.0 for side in (*_OUTSIDE, *_CORNERS)} 

1810 reserve = {side: 0.0 for side in _OUTSIDE} 

1811 for kind in ("compare", "saccades", "colors"): 

1812 if kind not in moved or not names[kind]: 

1813 continue 

1814 spec = specs[kind] 

1815 position = "above" if spec["position"] == "auto" else spec["position"] 

1816 horizontal = ( 

1817 spec["arrangement"] == "side-by-side" 

1818 if spec["arrangement"] != "auto" 

1819 else position in ("above", "below") 

1820 ) 

1821 font_px = float(spec["size"] or base_font) 

1822 w, h = _legend_extent(names[kind], font_px, horizontal) 

1823 cfg: dict = { 

1824 "orientation": "h" if horizontal else "v", 

1825 "bgcolor": "rgba(255,255,255,0.75)", 

1826 } 

1827 if spec["size"]: 

1828 cfg["font"] = {"size": font_px} 

1829 if position == "above": 

1830 cfg.update( 

1831 xanchor="right", 

1832 x=1 - along["above"] / plot_w, 

1833 yanchor="bottom", 

1834 y=1 + gap / plot_h, 

1835 ) 

1836 along["above"] += w + gap 

1837 reserve["above"] = max(reserve["above"], h + gap) 

1838 elif position == "below": 

1839 cfg.update( 

1840 xanchor="left", 

1841 x=along["below"] / plot_w, 

1842 yanchor="top", 

1843 y=-(margin["b"] + gap) / plot_h, 

1844 ) 

1845 along["below"] += w + gap 

1846 reserve["below"] = max(reserve["below"], h + gap) 

1847 elif position == "left": 

1848 cfg.update( 

1849 xanchor="right", 

1850 x=-(margin["l"] + gap) / plot_w, 

1851 yanchor="top", 

1852 y=1 - along["left"] / plot_h, 

1853 ) 

1854 along["left"] += h + gap 

1855 reserve["left"] = max(reserve["left"], w + gap) 

1856 elif position == "right": 

1857 cfg.update( 

1858 xanchor="left", 

1859 x=1 + (margin["r"] + gap) / plot_w, 

1860 yanchor="top", 

1861 y=1 - along["right"] / plot_h, 

1862 ) 

1863 along["right"] += h + gap 

1864 reserve["right"] = max(reserve["right"], w + gap) 

1865 else: 

1866 top = position.startswith("top") 

1867 left = position.endswith("left") 

1868 inset = along[position] 

1869 cfg.update( 

1870 xanchor="left" if left else "right", 

1871 x=gap / plot_w if left else 1 - gap / plot_w, 

1872 yanchor="top" if top else "bottom", 

1873 y=1 - (gap + inset) / plot_h if top else (gap + inset) / plot_h, 

1874 ) 

1875 along[position] += h + gap 

1876 fig.update_layout({_LEGEND_IDS[kind]: cfg}) 

1877 # The default legend's strip above the plot: when every entry has moved out 

1878 # of it and the strip is exactly that reserve (the single-trial figures — 

1879 # a comparison's top margin also holds its title), it is handed back. 

1880 default_left = any( 

1881 t.legend in (None, "legend") and t.showlegend is not False and t.name 

1882 for t in fig.data 

1883 ) 

1884 if not default_left and margin["t"] == _LEGEND_RESERVE_PX and not comparing: 

1885 _grow(fig, "t", -_LEGEND_RESERVE_PX) 

1886 margin["t"] = 0.0 

1887 # Above: the default legend's own reserve may already cover it. 

1888 _grow(fig, "t", max(0.0, reserve["above"] - margin["t"])) 

1889 _grow(fig, "b", reserve["below"]) 

1890 _grow(fig, "l", reserve["left"]) 

1891 _grow(fig, "r", reserve["right"]) 

1892 return fig 

1893 

1894 

1895def _size_key_layout(layout: Mapping | None) -> dict: 

1896 """The size key's resolved spot, arrangement and label size.""" 

1897 spec = normalize_legend_layout(layout)["size_key"] 

1898 position = spec["position"] 

1899 return { 

1900 "position": _SIZE_KEY_AUTO_POSITION if position == "auto" else position, 

1901 "stacked": spec["arrangement"] == "stacked", 

1902 "size": spec["size"], 

1903 } 

1904 

1905 

1906def _add_duration_size_key( 

1907 fig: go.Figure, 

1908 size_range: tuple[int, int], 

1909 scale: str, 

1910 duration_range, 

1911 *, 

1912 font_family: str | None = None, 

1913 legend_layout: Mapping | None = None, 

1914) -> None: 

1915 """Draw the fixed duration scale's key: reference circles labelled in ms. 

1916 

1917 Pixel-sized shapes anchored to a corner of the plot (paper coordinates), so 

1918 each circle is exactly the diameter a fixation of that duration gets — at 

1919 any canvas size and through every export path, since they are layout 

1920 shapes rather than a trace. ``legend_layout``'s ``size_key`` entry picks the 

1921 spot (inside bottom-right by default), a row or a column, and the labels' 

1922 size; the circles themselves never scale, since their size *is* the key. 

1923 An outside spot grows the figure so the plot region keeps its size. 

1924 Nothing is drawn for the relative scale: its sizes mean something only 

1925 inside one figure.""" 

1926 if scale == "relative": 

1927 return 

1928 placement = _size_key_layout(legend_layout) 

1929 refs = _duration_key_references(duration_range) 

1930 sizes = [ 

1931 float(v) 

1932 for v in _compute_marker_sizes( 

1933 pd.Series([d for d, _ in refs]), size_range, scale, duration_range 

1934 ) 

1935 ] 

1936 label_font = float(placement["size"] or 10) 

1937 label_px = 1.6 * label_font 

1938 biggest = float(size_range[1]) 

1939 pad = 8.0 

1940 n = len(refs) 

1941 # The key's own box, in px, origin bottom-left and y up: where each circle's 

1942 # centre and each label sit inside it. 

1943 if placement["stacked"]: 

1944 row = max(biggest, label_px) + 6.0 

1945 label_w = 0.6 * label_font * max(len(label) for _, label in refs) 

1946 w = pad + biggest + 6.0 + label_w + pad 

1947 h = 2 * pad + n * row 

1948 centres = [(pad + biggest / 2, h - pad - (i + 0.5) * row) for i in range(n)] 

1949 labels = [(pad + biggest + 6.0, cy, "left") for _, cy in centres] 

1950 else: 

1951 slot = max(biggest, 3.0 * label_font) + 8.0 

1952 w = 2 * pad + n * slot 

1953 h = pad + label_px + biggest + pad 

1954 cy = pad + label_px + biggest / 2.0 

1955 centres = [(pad + (i + 0.5) * slot, cy) for i in range(n)] 

1956 labels = [(cx, pad + label_px / 2.0, "center") for cx, _ in centres] 

1957 left, bottom, ax, ay = _size_key_anchor(fig, placement["position"], w, h) 

1958 for (_, label), size, (cx, cy), (lx, ly, align) in zip( 

1959 refs, sizes, centres, labels 

1960 ): 

1961 r = size / 2.0 

1962 fig.add_shape( 

1963 type="circle", 

1964 xref="paper", 

1965 yref="paper", 

1966 xsizemode="pixel", 

1967 ysizemode="pixel", 

1968 xanchor=ax, 

1969 yanchor=ay, 

1970 x0=left + cx - r, 

1971 x1=left + cx + r, 

1972 y0=bottom + cy - r, 

1973 y1=bottom + cy + r, 

1974 line=dict(color="#555555", width=1), 

1975 fillcolor="rgba(120,120,120,0.35)", 

1976 layer="above", 

1977 # Rides the fixations layer of a separable export (VIZ-5). 

1978 name=_shape_layer_tag("fixations"), 

1979 ) 

1980 fig.add_annotation( 

1981 x=ax, 

1982 y=ay, 

1983 xref="paper", 

1984 yref="paper", 

1985 xshift=left + lx, 

1986 yshift=bottom + ly, 

1987 text=label, 

1988 showarrow=False, 

1989 xanchor=align, 

1990 yanchor="middle", 

1991 font=dict( 

1992 size=label_font, color="#444444", family=font_family or FONT_FAMILY 

1993 ), 

1994 name=_SIZE_KEY_NAME, 

1995 ) 

1996 _stack_bottom_right(fig) 

1997 

1998 

1999def _size_key_anchor(fig: go.Figure, position: str, w: float, h: float) -> tuple: 

2000 """``(left, bottom, anchor_x, anchor_y)``: where the size key's box goes. 

2001 

2002 ``left`` / ``bottom`` are px from the paper anchor to the box's bottom-left 

2003 corner. An outside spot sits beyond whatever already occupies that margin 

2004 (a colour bar, the transport controls, another legend) and grows it. 

2005 """ 

2006 gap = _LEGEND_GAP_PX 

2007 if position in _CORNERS: 

2008 top = position.startswith("top") 

2009 right = position.endswith("right") 

2010 return ( 

2011 (-w if right else 0.0), 

2012 (-h if top else 0.0), 

2013 (1 if right else 0), 

2014 (1 if top else 0), 

2015 ) 

2016 _plot_w, _plot_h, margin = _plot_px(fig) 

2017 if position == "right": 

2018 _grow(fig, "r", w + gap) 

2019 return margin["r"] + gap, 0.0, 1, 0 

2020 if position == "left": 

2021 _grow(fig, "l", w + gap) 

2022 return -(margin["l"] + gap + w), 0.0, 0, 0 

2023 if position == "above": 

2024 _grow(fig, "t", max(0.0, h + gap - margin["t"])) 

2025 return 0.0, gap, 0, 1 

2026 # below 

2027 _grow(fig, "b", h + gap) 

2028 return -w, -(margin["b"] + gap + h), 1, 0 

2029 

2030 

2031# VIZ-9 "linear reading" mode: draw saccades as upward arcs instead of straight 

2032# connectors. The apex rises by _ARCH_FRAC of the saccade's horizontal span; each 

2033# arc is sampled into _ARCH_SAMPLES points so it stays smooth in the true-scale 

2034# embed. `arch_frac=None` (the default everywhere) keeps the straight connectors. 

2035_ARCH_FRAC = 0.28 

2036_ARCH_SAMPLES = 20 

2037 

2038 

2039def _arch_control_point( 

2040 x0: float, y0: float, x1: float, y1: float, frac: float 

2041) -> tuple[float, float]: 

2042 """Control point of the quadratic Bézier arch drawn between two fixations. 

2043 

2044 Single source of truth for the arch geometry: ``_arch_points`` samples the 

2045 curve it defines and ``_arch_point_and_tangent`` (the arrowheads, BUG-9) 

2046 evaluates the same curve, so the marker can never drift off the drawn line. 

2047 Screen y grows downward, so the control point sits *above* the chord.""" 

2048 return (x0 + x1) / 2.0, min(y0, y1) - frac * abs(x1 - x0) 

2049 

2050 

2051def _arch_points( 

2052 x0: float, y0: float, x1: float, y1: float, frac: float, n: int = _ARCH_SAMPLES 

2053) -> tuple[list, list]: 

2054 """Sample a quadratic Bézier arch from (x0,y0) to (x1,y1), bulging upward. 

2055 

2056 Screen y grows downward, so the control point is *above* the chord (smaller 

2057 y). Returns (xs, ys) of length ``n`` including both endpoints. A NaN endpoint 

2058 propagates to NaN samples, which Plotly simply skips.""" 

2059 cx, cy = _arch_control_point(x0, y0, x1, y1, frac) 

2060 ts = np.linspace(0.0, 1.0, n) 

2061 xs = ((1 - ts) ** 2) * x0 + 2 * (1 - ts) * ts * cx + (ts**2) * x1 

2062 ys = ((1 - ts) ** 2) * y0 + 2 * (1 - ts) * ts * cy + (ts**2) * y1 

2063 return xs.tolist(), ys.tolist() 

2064 

2065 

2066def _arch_apex_y(x0: float, y0: float, x1: float, y1: float, frac: float) -> float: 

2067 """Topmost (smallest) ``y`` reached by the drawn arch — its true apex (BUG-13). 

2068 

2069 Minimises ``y(t)`` over the SAME quadratic ``_arch_points`` samples and 

2070 ``_arch_point_and_tangent`` evaluates, instead of reading off one sampled 

2071 parameter. Writing the curve as ``y(t) = y0 + 2t(cy - y0) + t^2 * d`` with 

2072 ``d = y0 - 2*cy + y1`` gives the closed-form extremum ``t* = (y0 - cy)/d`` and 

2073 the value ``y0 - (y0 - cy)^2 / d``. The control point is at or above both 

2074 endpoints (``cy <= min(y0, y1)``), so ``d >= 0`` and ``t*`` always lands in 

2075 ``[0, 1]``. 

2076 

2077 For a level saccade the peak is at ``t = 0.5`` (the old estimate), but as the 

2078 endpoints diverge in ``y`` it slides toward the higher one and rises well 

2079 above the ``t = 0.5`` point — which is why a wide, steeply-sloped arc used to 

2080 clip against the top of the view. 

2081 """ 

2082 _, cy = _arch_control_point(x0, y0, x1, y1, frac) 

2083 denom = y0 - 2.0 * cy + y1 

2084 if denom <= 0.0: 

2085 # Degenerate: zero rise and level endpoints — the "arch" is a flat line. 

2086 return min(y0, y1) 

2087 t = (y0 - cy) / denom 

2088 if t <= 0.0: 

2089 return y0 

2090 if t >= 1.0: 

2091 return y1 

2092 return y0 - (y0 - cy) ** 2 / denom 

2093 

2094 

2095def _extend_segment( 

2096 xs: list, ys: list, x0, y0, x1, y1, arch_frac: float | None 

2097) -> None: 

2098 """Append one saccade segment (straight, or an arch when ``arch_frac``) to the 

2099 None-separated ``xs``/``ys`` accumulators.""" 

2100 if arch_frac is None: 

2101 xs.extend([x0, x1, None]) 

2102 ys.extend([y0, y1, None]) 

2103 else: 

2104 ax, ay = _arch_points(x0, y0, x1, y1, arch_frac) 

2105 xs.extend(ax + [None]) 

2106 ys.extend(ay + [None]) 

2107 

2108 

2109def _saccade_segments( 

2110 fix_df: pd.DataFrame, 

2111 x_col: str, 

2112 y_col: str, 

2113 arch_frac: float | None = None, 

2114) -> tuple[list, list]: 

2115 """Return concatenated x/y arrays separated by None for a single saccade trace. 

2116 

2117 ``arch_frac`` (VIZ-9) draws each segment as an upward arc instead of a straight 

2118 line.""" 

2119 if len(fix_df) < 2: 

2120 return [], [] 

2121 ordered = fix_df.sort_values("timestamp_ms") 

2122 xs: list = [] 

2123 ys: list = [] 

2124 x_vals = ordered[x_col].tolist() 

2125 y_vals = ordered[y_col].tolist() 

2126 for i in range(len(ordered) - 1): 

2127 _extend_segment( 

2128 xs, ys, x_vals[i], y_vals[i], x_vals[i + 1], y_vals[i + 1], arch_frac 

2129 ) 

2130 return xs, ys 

2131 

2132 

2133def _saccade_segments_by_class( 

2134 fix_df: pd.DataFrame, 

2135 x_col: str, 

2136 y_col: str, 

2137 classes: pd.Series, 

2138 arch_frac: float | None = None, 

2139) -> dict: 

2140 """Group saccade segments by reading class → ``{class: (xs, ys)}`` (VIZ-8). 

2141 

2142 Each segment (fixation i → i+1) takes the class of its *departing* fixation 

2143 (``classes[i]``), so it matches ``measures.classify_saccades``. Segments with 

2144 no class (``None``/NaN — the last fixation, or an unclassifiable one) fall 

2145 into ``"other"`` so they still draw. Each class's arrays are None-separated, 

2146 ready for one Scatter trace per class. ``arch_frac`` (VIZ-9) arcs each 

2147 segment.""" 

2148 if len(fix_df) < 2: 

2149 return {} 

2150 ordered = fix_df.sort_values("timestamp_ms") 

2151 cls = classes.reindex(ordered.index).tolist() 

2152 x_vals = ordered[x_col].tolist() 

2153 y_vals = ordered[y_col].tolist() 

2154 out: dict = {} 

2155 for i in range(len(ordered) - 1): 

2156 c = cls[i] 

2157 if pd.isna(c): 

2158 c = "other" 

2159 xs, ys = out.setdefault(c, ([], [])) 

2160 _extend_segment( 

2161 xs, ys, x_vals[i], y_vals[i], x_vals[i + 1], y_vals[i + 1], arch_frac 

2162 ) 

2163 return out 

2164 

2165 

2166def _snap_fixations_to_words( 

2167 fixations: pd.DataFrame, words: pd.DataFrame, x_field: str, y_field: str 

2168) -> pd.DataFrame: 

2169 """Return a copy of ``fixations`` with each fixation moved to the top-centre of 

2170 the word it lands on (VIZ-9 "linear reading" mode). 

2171 

2172 Fixations with no assigned word keep their raw position. Uses a precomputed 

2173 ``word_id`` column when present, else assigns via bounding-box containment.""" 

2174 out = fixations.copy() 

2175 if "word_id" not in words.columns: 

2176 return out 

2177 if ( 

2178 "word_id" in out.columns 

2179 and pd.to_numeric(out["word_id"], errors="coerce").notna().any() 

2180 ): 

2181 wid = pd.to_numeric(out["word_id"], errors="coerce") 

2182 else: 

2183 from .measures import assign_fixations_to_words 

2184 

2185 wid = pd.to_numeric( 

2186 assign_fixations_to_words(out, words)["word_id"], errors="coerce" 

2187 ) 

2188 # Snap above the middle of the word's box, where its label is drawn (BUG-97). 

2189 # Render-only: which word a fixation belongs to is still 

2190 # `assign_fixations_to_words`, against the same boxes. 

2191 from .measures import word_box_bounds 

2192 

2193 x0, _, x1, _ = word_box_bounds(words) 

2194 cx_by_id = dict(zip(words["word_id"], (x0 + x1) / 2.0)) 

2195 top_by_id = dict(zip(words["word_id"], pd.to_numeric(words["y"], errors="coerce"))) 

2196 snap_x = wid.map(cx_by_id) 

2197 snap_y = wid.map(top_by_id) 

2198 out[x_field] = snap_x.where(snap_x.notna(), out[x_field]) 

2199 out[y_field] = snap_y.where(snap_y.notna(), out[y_field]) 

2200 return out 

2201 

2202 

2203# Saccades shorter than this fraction of the fixation-extent diagonal get no 

2204# direction arrow — their heading is sub-pixel noise (refixations on one word). 

2205_ARROW_MIN_LEN_FRAC = 0.005 

2206 

2207# Bézier parameter at which the arrowhead sits on an arched saccade (BUG-9). 

2208_ARROW_ARCH_T = 0.5 

2209 

2210 

2211def _arch_point_and_tangent( 

2212 x0: float, y0: float, x1: float, y1: float, frac: float, t: float = _ARROW_ARCH_T 

2213) -> tuple[float, float, float, float]: 

2214 """Point on the drawn arch at Bézier parameter ``t``, plus the curve's tangent. 

2215 

2216 Evaluates the very curve ``_arch_points`` samples (same 

2217 ``_arch_control_point``), so an arrowhead placed here lands exactly on the 

2218 rendered line. Returns ``(x, y, dx, dy)`` where ``(dx, dy)`` is the 

2219 (unnormalized) tangent B'(t). 

2220 

2221 Note the quadratic's identity at ``t=0.5``: B'(0.5) == P1 - P0, i.e. the 

2222 tangent at the parameter midpoint is parallel to the chord regardless of the 

2223 control point. So arcing a saccade moves the arrowhead *up onto* the curve 

2224 (the visible defect) without rotating it — which is exactly right, the curve 

2225 really is chord-parallel there.""" 

2226 cx, cy = _arch_control_point(x0, y0, x1, y1, frac) 

2227 u = 1.0 - t 

2228 px = u * u * x0 + 2 * u * t * cx + t * t * x1 

2229 py = u * u * y0 + 2 * u * t * cy + t * t * y1 

2230 tx = 2 * u * (cx - x0) + 2 * t * (x1 - cx) 

2231 ty = 2 * u * (cy - y0) + 2 * t * (y1 - cy) 

2232 return px, py, tx, ty 

2233 

2234 

2235def _saccade_arrow_rows( 

2236 fix_df: pd.DataFrame, 

2237 x_col: str, 

2238 y_col: str, 

2239 arch_frac: float | None = None, 

2240) -> tuple[list, list, list, list]: 

2241 """:func:`_saccade_arrow_markers` plus each arrowhead's saccade index. 

2242 

2243 Returns ``(mid_x, mid_y, angle_deg, segment_index)``. Arrowheads are dropped 

2244 for micro/degenerate saccades, so the arrays are shorter than the saccade 

2245 count and the position alone doesn't say which saccade an arrow belongs to — 

2246 ``segment_index[j]`` is the index of the departing fixation, which is what 

2247 lets the animated replay reveal each arrow with its own saccade (VIZ-23). 

2248 """ 

2249 if len(fix_df) < 2: 

2250 return [], [], [], [] 

2251 ordered = fix_df.sort_values("timestamp_ms") 

2252 xv = pd.to_numeric(ordered[x_col], errors="coerce").to_numpy() 

2253 yv = pd.to_numeric(ordered[y_col], errors="coerce").to_numpy() 

2254 # Suppress arrowheads on micro-saccades: a sub-pixel refixation has a 

2255 # well-defined midpoint but its direction is just noise, so a full-size 

2256 # arrow would point a random way. Threshold scales with the data extent so 

2257 # it's dataset-agnostic. 

2258 finite = np.isfinite(xv) & np.isfinite(yv) 

2259 if finite.any(): 

2260 x_ext = float(np.nanmax(xv[finite]) - np.nanmin(xv[finite])) 

2261 y_ext = float(np.nanmax(yv[finite]) - np.nanmin(yv[finite])) 

2262 min_len = np.hypot(x_ext, y_ext) * _ARROW_MIN_LEN_FRAC 

2263 else: 

2264 min_len = 0.0 

2265 mid_x: list = [] 

2266 mid_y: list = [] 

2267 angles: list = [] 

2268 seg_index: list = [] 

2269 for i in range(len(ordered) - 1): 

2270 x0, y0, x1, y1 = xv[i], yv[i], xv[i + 1], yv[i + 1] 

2271 if not np.isfinite((x0, y0, x1, y1)).all(): 

2272 continue 

2273 dx, dy = x1 - x0, y1 - y0 

2274 seg_len = float(np.hypot(dx, dy)) 

2275 if seg_len == 0.0 or seg_len < min_len: 

2276 continue 

2277 if arch_frac is None: 

2278 mx, my, hx, hy = (x0 + x1) / 2.0, (y0 + y1) / 2.0, dx, dy 

2279 else: 

2280 mx, my, hx, hy = _arch_point_and_tangent(x0, y0, x1, y1, arch_frac) 

2281 mid_x.append(mx) 

2282 mid_y.append(my) 

2283 # marker.angle is clockwise from up; screen-up is decreasing data y 

2284 # (the y-axis is drawn reversed), so negate the heading's dy. 

2285 angles.append(float(np.degrees(np.arctan2(hx, -hy)))) 

2286 seg_index.append(i) 

2287 return mid_x, mid_y, angles, seg_index 

2288 

2289 

2290def _saccade_arrow_markers( 

2291 fix_df: pd.DataFrame, 

2292 x_col: str, 

2293 y_col: str, 

2294 arch_frac: float | None = None, 

2295) -> tuple[list, list, list]: 

2296 """Arrowhead position + rotation for each saccade, for a marker trace. 

2297 

2298 Returns (mid_x, mid_y, angle_deg) with one entry per consecutive-fixation 

2299 segment: a marker at the segment midpoint, rotated to point along the gaze 

2300 direction. Angles follow Plotly's ``marker.angle`` convention (degrees 

2301 clockwise from "up") and account for the reversed y-axis — data y grows 

2302 downward on screen — so they read correctly on the plot. 

2303 

2304 ``arch_frac`` (BUG-9) must be the same value the segment builders got: in Arc 

2305 mode (VIZ-9) the marker moves to the *arch's* midpoint and takes the arc's 

2306 tangent there, instead of floating below the curve at the straight chord's 

2307 midpoint. ``None`` (the default everywhere) keeps the straight-chord 

2308 placement. 

2309 """ 

2310 mid_x, mid_y, angles, _ = _saccade_arrow_rows(fix_df, x_col, y_col, arch_frac) 

2311 return mid_x, mid_y, angles 

2312 

2313 

2314_RGB_FUNCTION = re.compile( 

2315 r"rgba?\(\s*(\d{1,3})\s*,\s*(\d{1,3})\s*,\s*(\d{1,3})\s*(?:,[^)]*)?\)" 

2316) 

2317 

2318 

2319def color_with_alpha(color: str, alpha: float) -> str: 

2320 """``color`` as ``rgba(r,g,b,alpha)`` — a fill drawn at its own opacity. 

2321 

2322 Takes ``#rrggbb``, ``#rgb`` or ``rgb(…)`` / ``rgba(…)`` (whose own alpha is 

2323 replaced). Anything else raises ``ValueError`` naming it, rather than 

2324 quietly drawing some other colour. 

2325 """ 

2326 text = str(color).strip() 

2327 match = re.fullmatch(r"#([0-9A-Fa-f]{3}|[0-9A-Fa-f]{6})", text) 

2328 if match: 

2329 digits = match.group(1) 

2330 if len(digits) == 3: 

2331 digits = "".join(c * 2 for c in digits) 

2332 r, g, b = (int(digits[i : i + 2], 16) for i in (0, 2, 4)) 

2333 return f"rgba({r},{g},{b},{alpha})" 

2334 match = _RGB_FUNCTION.fullmatch(text) 

2335 if match and all(int(v) <= 255 for v in match.groups()): 

2336 r, g, b = match.groups() 

2337 return f"rgba({r},{g},{b},{alpha})" 

2338 raise ValueError( 

2339 f"{color!r} is not a color a fill can take: use #rrggbb, #rgb or rgb(r, g, b)." 

2340 ) 

2341 

2342 

2343def build_word_boxes( 

2344 words: pd.DataFrame, 

2345 color: str = WORD_BOX_COLOR, 

2346 fill_color: str = WORD_BOX_FILL_COLOR, 

2347 fill_opacity: float = WORD_BOX_FILL_OPACITY, 

2348 line_opacity: float = WORD_BOX_LINE_OPACITY, 

2349) -> list: 

2350 """Rectangles for the word interest areas. 

2351 

2352 Drawn from ``measures.word_box_bounds`` — the experiment's own rectangles 

2353 (BUG-83) — so what's on screen is exactly what ``assign_fixations_to_words`` 

2354 assigns against. On a tiling corpus each outline therefore runs on across 

2355 the space after its word, and the word *label* is centred in it (BUG-97). 

2356 ``color`` is the outline, drawn at ``line_opacity`` (0 = no outline); the 

2357 fill is ``fill_color`` at ``fill_opacity`` (0 = no fill). Each is its own 

2358 alpha, so one never fades the other. A fully opaque outline keeps ``color`` 

2359 as given, so any colour Plotly accepts still works there. 

2360 """ 

2361 from .measures import word_box_bounds 

2362 

2363 fill = color_with_alpha(fill_color, fill_opacity) 

2364 if line_opacity < 1: 

2365 color = color_with_alpha(color, line_opacity) 

2366 shapes = [] 

2367 for x0, y0, x1, y1 in zip(*word_box_bounds(words)): 

2368 shapes.append( 

2369 dict( 

2370 type="rect", 

2371 x0=x0, 

2372 y0=y0, 

2373 x1=x1, 

2374 y1=y1, 

2375 line=dict(color=color, width=1), 

2376 fillcolor=fill, 

2377 # VIZ-5: tag the layer so split_scanpath_layers can separate the 

2378 # word boxes from the (visually similar) heatmap rects. 

2379 name=_shape_layer_tag("word_boxes"), 

2380 ) 

2381 ) 

2382 return shapes 

2383 

2384 

2385# Bold-frame overlay for critical-span words; rendered on top of regular word 

2386# boxes only when the trial was shown with a preview question (Hunting condition). 

2387_CRITICAL_FRAME_COLOR = "#000000" # black — high-contrast frame, readable over heatmaps 

2388_CRITICAL_FRAME_WIDTH = 2 

2389_CRITICAL_TEXT_COLOR = ( 

2390 HIGHLIGHTED_TEXT_COLOR # dark pink — used when critical_span_style="Mark text" 

2391) 

2392 

2393 

2394def build_critical_span_overlay( 

2395 words: pd.DataFrame, 

2396 column: str = "is_in_aspan", 

2397 color: str = _CRITICAL_FRAME_COLOR, 

2398) -> list: 

2399 """Return outline shapes for the highlighted span (``column``, default the 

2400 OneStop answer span ``is_in_aspan``), outlined in ``color``. 

2401 

2402 Each visual line that contains highlighted words gets its own outline 

2403 rectangle, going from the *first* to the *last* highlighted word on that 

2404 line (not the whole line). Returns [] when the column is missing or no 

2405 words match. 

2406 """ 

2407 if not column or column not in words.columns: 

2408 return [] 

2409 mask = words[column].fillna(False).astype(bool) 

2410 if not mask.any(): 

2411 return [] 

2412 span = words[mask].copy() 

2413 

2414 # Cluster words into visual lines by y. `line_idx` upstream is often a 

2415 # constant (no real per-word line numbers in OneStop IA exports), so we 

2416 # group by y with a tolerance of ~half a word-height: rows whose y jumps 

2417 # by more than that are on a new line. 

2418 typical_h = float(span["height"].median() or 1.0) 

2419 y_sorted = span["y"].sort_values() 

2420 line_ids = (y_sorted.diff().fillna(0) > typical_h * 0.5).cumsum() 

2421 span["_line_id"] = line_ids.reindex(span.index) 

2422 

2423 from .measures import word_box_bounds 

2424 

2425 span_x0, _, span_x1, _ = word_box_bounds(span) 

2426 span["_box_x0"], span["_box_x1"] = span_x0, span_x1 

2427 

2428 shapes = [] 

2429 for _, group in span.groupby("_line_id"): 

2430 x0 = float(group["_box_x0"].min()) 

2431 x1 = float(group["_box_x1"].max()) 

2432 y0 = float(group["y"].min()) 

2433 y1 = float((group["y"] + group["height"]).max()) 

2434 shapes.append( 

2435 dict( 

2436 type="rect", 

2437 x0=x0, 

2438 y0=y0, 

2439 x1=x1, 

2440 y1=y1, 

2441 line=dict(color=color, width=_CRITICAL_FRAME_WIDTH), 

2442 fillcolor="rgba(0,0,0,0)", 

2443 layer="above", 

2444 # VIZ-5: the critical-span outline rides the word-boxes layer. 

2445 name=_shape_layer_tag("word_boxes"), 

2446 ) 

2447 ) 

2448 return shapes 

2449 

2450 

2451# --- VIZ-5: separable-layer export ------------------------------------------ 

2452# The scanpath figure is a single flattened image, but publication workflows want 

2453# to restyle each layer in Illustrator / Inkscape. `split_scanpath_layers` returns 

2454# one figure per layer, each a copy of the full figure with only that layer's 

2455# elements kept and everything else removed — so the layouts (axis ranges, size, 

2456# equal-aspect scaleanchor) stay byte-identical and the exported files register 

2457# perfectly when stacked. Every element is tagged with its layer: shapes carry a 

2458# `_LAYER_SHAPE_TAG`-prefixed `name`, traces are classified by their (stable) name, 

2459# and the single `layout.image` is the stimulus. 

2460_LAYER_SHAPE_TAG = "__sps_layer:" 

2461# Draw order (bottom → top), matching how make_scanpath_figure stacks them. 

2462SCANPATH_LAYER_ORDER = ( 

2463 "stimulus_image", 

2464 "heatmap", 

2465 "word_boxes", 

2466 "saccades", 

2467 "fixations", 

2468 "raw_gaze", 

2469 "labels", 

2470 "frame", 

2471) 

2472_TRANSPARENT = "rgba(0,0,0,0)" 

2473 

2474 

2475def _shape_layer_tag(layer: str) -> str: 

2476 """The `name` marker stamped on a shape so its layer survives into the figure.""" 

2477 return f"{_LAYER_SHAPE_TAG}{layer}" 

2478 

2479 

2480def _shape_layer(shape) -> str | None: 

2481 """Layer of a tagged shape, or None for an untagged one.""" 

2482 name = getattr(shape, "name", None) or "" 

2483 if name.startswith(_LAYER_SHAPE_TAG): 

2484 return name[len(_LAYER_SHAPE_TAG) :] 

2485 return None 

2486 

2487 

2488def _trace_layer(trace) -> str: 

2489 """Classify a scanpath trace into its layer by its (stable) name. 

2490 

2491 Only a handful of names are fixed — ``words`` (labels), the saccade traces, 

2492 ``Raw gaze``, and any heatmap trace (``… heatmap …``). Every *other* trace the 

2493 scanpath figure draws is a fixation-marker variant with a data-dependent name 

2494 (``Fixations``, per-line ``line: …``, categorical colour-legend entries, the 

2495 PRE-2 flag overlays, the PRE-3 ``drift`` connectors), so they all fall through 

2496 to ``fixations`` — robust to those names changing.""" 

2497 name = trace.name or "" 

2498 low = name.lower() 

2499 if name == "words": 

2500 return "labels" 

2501 if "heatmap" in low: 

2502 return "heatmap" 

2503 if name == "Raw gaze": 

2504 return "raw_gaze" 

2505 if name in ("saccades", "saccade direction") or name in set( 

2506 SACCADE_CLASS_LABELS.values() 

2507 ): 

2508 return "saccades" 

2509 return "fixations" 

2510 

2511 

2512def split_scanpath_layers(fig: go.Figure) -> dict[str, go.Figure]: 

2513 """Split a `make_scanpath_figure` result into one figure per visible layer. 

2514 

2515 Returns ``{layer_name: figure}`` in bottom-to-top draw order, keeping only the 

2516 layers that actually have elements. Each returned figure is a copy of ``fig`` 

2517 with (a) only that layer's traces/shapes/images kept and (b) a transparent 

2518 paper/plot background, so stacking the exported files in a vector editor 

2519 reproduces the combined figure exactly (identical axis ranges + size ⇒ perfect 

2520 registration). See VIZ-5.""" 

2521 # Which layers are present, and each trace's layer (computed once). 

2522 trace_layers = [_trace_layer(tr) for tr in fig.data] 

2523 shape_layers = [_shape_layer(sh) for sh in (fig.layout.shapes or ())] 

2524 present = set(trace_layers) | {s for s in shape_layers if s} 

2525 if fig.layout.images: 

2526 present.add("stimulus_image") 

2527 

2528 out: dict[str, go.Figure] = {} 

2529 for layer in SCANPATH_LAYER_ORDER: 

2530 if layer not in present: 

2531 continue 

2532 g = copy.deepcopy(fig) 

2533 g.data = tuple(tr for tr, tl in zip(g.data, trace_layers) if tl == layer) 

2534 g.layout.shapes = tuple( 

2535 sh for sh, sl in zip(g.layout.shapes or (), shape_layers) if sl == layer 

2536 ) 

2537 g.layout.images = fig.layout.images if layer == "stimulus_image" else () 

2538 # The duration size key's ms labels belong with its circles, which ride 

2539 # the fixations layer; every other annotation (title text, the 

2540 # Illustration stamp) stays on each layer as before. 

2541 if layer != "fixations": 

2542 g.layout.annotations = tuple( 

2543 a for a in (g.layout.annotations or ()) if a.name != _SIZE_KEY_NAME 

2544 ) 

2545 # Transparent background so the layers overlay cleanly when re-stacked. 

2546 g.update_layout(paper_bgcolor=_TRANSPARENT, plot_bgcolor=_TRANSPARENT) 

2547 out[layer] = g 

2548 return out 

2549 

2550 

2551_HOVER_MEASURE_LABELS: dict[str, str] = { 

2552 "total_fixation_duration_ms": "TFD", 

2553 "first_fixation_ms": "FFD", 

2554 "first_pass_gaze_duration_ms": "FPRT", 

2555 "regression_path_duration_ms": "RPD", 

2556 "n_fixations": "Fixations", 

2557} 

2558 

2559#: DATA-66: `FigureSettings.column_labels` for the build in progress. The three 

2560#: builders set it (`_labelled_columns`) and the helpers that write a column's 

2561#: name into the figure read it (`_column_title`, `_hover_label`), so the dozen 

2562#: helpers in between keep their signatures. Empty outside a build. 

2563_COLUMN_LABELS: ContextVar[Mapping[str, str] | None] = ContextVar( 

2564 "scanpath_column_labels", default=None 

2565) 

2566 

2567 

2568@contextmanager 

2569def _labelled_columns(labels: Mapping[str, str] | None) -> Iterator[None]: 

2570 """Make ``labels`` the column names one figure build writes.""" 

2571 token = _COLUMN_LABELS.set(dict(labels or {})) 

2572 try: 

2573 yield 

2574 finally: 

2575 _COLUMN_LABELS.reset(token) 

2576 

2577 

2578def _humanize_column(column: str, *, unit: bool = True) -> str: 

2579 """``total_fixation_duration_ms`` → "Total fixation duration (ms)", and 

2580 ``participant_id`` → "Participant ID". 

2581 

2582 ``unit=False`` drops the unit, for a hover row that writes it after the 

2583 value — which used to read "… Duration Ms: 200 ms".""" 

2584 from .column_names import _CANONICAL_LABELS, canonical_label 

2585 

2586 text = str(column) 

2587 if text in _CANONICAL_LABELS: 

2588 # #374: the app's own columns read as the rail names them. 

2589 label = canonical_label(text) 

2590 return label if unit else label.removesuffix(" (ms)") 

2591 in_ms = text.endswith("_ms") 

2592 if in_ms: 

2593 text = text[: -len("_ms")] 

2594 words = text.replace("_", " ").strip() 

2595 title = re.sub(r"\bid\b", "ID", words[:1].upper() + words[1:], flags=re.IGNORECASE) 

2596 return f"{title} (ms)" if in_ms and unit else title 

2597 

2598 

2599def _column_name(column: str) -> str: 

2600 """A column's name in a legend entry: the dataset's own (DATA-66), else the 

2601 column's own, as the legend has always written it.""" 

2602 labels = _COLUMN_LABELS.get() or {} 

2603 return labels.get(column, str(column)) 

2604 

2605 

2606def _column_title(column: str) -> str: 

2607 """A column's name in a figure's titles: the dataset's own (DATA-66), else 

2608 the column humanized.""" 

2609 labels = _COLUMN_LABELS.get() or {} 

2610 return labels[column] if column in labels else _humanize_column(column) 

2611 

2612 

2613def _table_label(field: str, table: str | None) -> str | None: 

2614 """``field``'s label in ``table`` when the build names it: the 

2615 ``"<table>:<field>"`` entry first — a word table and a fixation table can 

2616 call one canonical column differently (``word_id``) — then the plain one.""" 

2617 labels = _COLUMN_LABELS.get() or {} 

2618 if table is not None and f"{table}:{field}" in labels: 

2619 return labels[f"{table}:{field}"] 

2620 return labels.get(field) 

2621 

2622 

2623def _hover_label(field: str, table: str | None = None) -> str: 

2624 """Readable label for an arbitrary hover column. 

2625 

2626 The dataset's own name when the build has one (DATA-66), as ``table`` 

2627 names it; else a short label for the app's own columns, else the column 

2628 humanized without its unit (the row writes the unit after the value).""" 

2629 own = _table_label(field, table) 

2630 if own is not None: 

2631 return own 

2632 aliases = { 

2633 "text": "Word", 

2634 "word_id": "Word #", 

2635 "line_idx": "Line #", 

2636 "order_in_trial": "Fixation #", 

2637 "duration_ms": "Duration", 

2638 "timestamp_ms": "Timestamp", 

2639 } 

2640 return aliases.get( 

2641 field, 

2642 _HOVER_MEASURE_LABELS.get(field, _humanize_column(field, unit=False)), 

2643 ) 

2644 

2645 

2646def _plotly_literal(value: str) -> str: 

2647 """``value`` as Plotly text that draws its own characters. 

2648 

2649 Plotly reads a text or hover string as its pseudo-HTML, so a stimulus 

2650 token ``<b>bold</b>`` drew bold and ``x<br>y`` broke the line (round-8 

2651 review, finding 6). Escaping ``&``, ``<`` and ``>`` — the entities Plotly 

2652 decodes back — keeps the dataset's characters on screen, and ``%{`` is 

2653 written ``&#37;{`` so a name placed in a hover template is not read as a 

2654 template field (round 9). Applied once, to data values and user text only, 

2655 at the figure boundary: the tables, exports and the app's own markup (a 

2656 hover's ``<br>``) are left as they are.""" 

2657 return html.escape(value, quote=False).replace("%{", "&#37;{") 

2658 

2659 

2660def _plotly_literal_values(series: pd.Series) -> pd.Series: 

2661 """A hover column with its strings made literal (:func:`_plotly_literal`); 

2662 numbers, missing values and anything else pass through untouched.""" 

2663 if pd.api.types.is_numeric_dtype(series) or pd.api.types.is_bool_dtype(series): 

2664 return series 

2665 return series.map( 

2666 lambda value: _plotly_literal(value) if isinstance(value, str) else value, 

2667 na_action="ignore", 

2668 ) 

2669 

2670 

2671def _fixation_order_labels(ordered: pd.DataFrame) -> list[str]: 

2672 """The fixation-number labels of a replay trail, one per row of ``ordered``. 

2673 

2674 The trial's own ``order_in_trial``, as the static figure, the comparison 

2675 and every hover show it — so a later screen of a multipart trial, a 

2676 fixation window or a *Discard* keeps its gaps (501, 502 …) instead of 

2677 renumbering what is left 1..n. A row without an index gets no label, as 

2678 on the static figure. Only a frame with no usable index at all (no column, 

2679 or nothing numeric in it) falls back to the ordinal 1..n.""" 

2680 n = len(ordered) 

2681 if "order_in_trial" in ordered.columns: 

2682 values = pd.to_numeric(ordered["order_in_trial"], errors="coerce") 

2683 if values.notna().any(): 

2684 return [ 

2685 "" 

2686 if pd.isna(value) 

2687 else str(int(value)) 

2688 if float(value).is_integer() 

2689 else f"{value:g}" 

2690 for value in values.tolist() 

2691 ] 

2692 return [str(j + 1) for j in range(n)] 

2693 

2694 

2695#: #374 F7: the fixation hover's lead line is written from these, in this 

2696#: order, as "Fixation 41 · 336 ms · on “Droppings!” (word 27)". 

2697_FIXATION_HEAD_FIELDS = ("order_in_trial", "duration_ms", "word_id") 

2698 

2699 

2700def _hover_number(value) -> str: 

2701 """A hover number without a trailing ``.0`` (``27.0`` → ``27``).""" 

2702 number = pd.to_numeric(value, errors="coerce") 

2703 if pd.isna(number): 

2704 return str(value) 

2705 return f"{number:.0f}" if float(number).is_integer() else f"{number:g}" 

2706 

2707 

2708def _fixation_hover_head( 

2709 frame: pd.DataFrame, head: Sequence[str], words: pd.DataFrame | None 

2710) -> pd.Series: 

2711 """The fixation hover's lead line, one string per row (#374 F7). 

2712 

2713 The word a fixation landed on is named by its text when the trial's word 

2714 table has it (and its word ids are unique); a fixation on no word reads 

2715 "outside the text".""" 

2716 word_text: dict[float, str] = {} 

2717 if ( 

2718 "word_id" in head 

2719 and words is not None 

2720 and not words.empty 

2721 and {"word_id", "text"} <= set(words.columns) 

2722 ): 

2723 ids = pd.to_numeric(words["word_id"], errors="coerce") 

2724 if ids.notna().all() and ids.is_unique: 

2725 word_text = dict(zip(ids.astype(float), words["text"].astype(str))) 

2726 columns = {field: frame[field].tolist() for field in head} 

2727 # A fixation with no word id is "outside the text" only when it is outside 

2728 # every word box: the data's own assignment can be blank inside one. 

2729 outside = [False] * len(frame) 

2730 if ( 

2731 "word_id" in head 

2732 and words is not None 

2733 and not words.empty 

2734 and {"x", "y"} <= set(frame.columns) 

2735 ): 

2736 from .measures import fixation_in_text_mask 

2737 

2738 outside = (~fixation_in_text_mask(frame, words)).tolist() 

2739 lines = [] 

2740 for i in range(len(frame)): 

2741 parts = [] 

2742 if "order_in_trial" in columns: 

2743 value = columns["order_in_trial"][i] 

2744 if pd.notna(value): 

2745 parts.append(f"Fixation {_hover_number(value)}") 

2746 if "duration_ms" in columns: 

2747 value = columns["duration_ms"][i] 

2748 if pd.notna(value): 

2749 parts.append(f"{_hover_number(value)} ms") 

2750 if "word_id" in columns: 

2751 value = pd.to_numeric(columns["word_id"][i], errors="coerce") 

2752 if pd.isna(value): 

2753 if outside[i]: 

2754 parts.append("outside the text") 

2755 else: 

2756 text = word_text.get(float(value)) 

2757 word = f"word {_hover_number(value)}" 

2758 parts.append( 

2759 f"on “{_plotly_literal(text)}” ({word})" if text else f"on {word}" 

2760 ) 

2761 lines.append(" · ".join(parts)) 

2762 return pd.Series(lines, index=frame.index, dtype=object) 

2763 

2764 

2765def _hover_cell(value): 

2766 """One hover value as shown: a missing one is "—" (Plotly printed 

2767 ``null``), a fraction is cut to four significant digits (#374).""" 

2768 if value is None: 

2769 return "—" 

2770 try: 

2771 if pd.isna(value): 

2772 return "—" 

2773 except (TypeError, ValueError): 

2774 return value 

2775 if isinstance(value, (float, np.floating)): 

2776 return f"{value:.0f}" if float(value).is_integer() else f"{value:.4g}" 

2777 return value 

2778 

2779 

2780def _hover_cells(series: pd.Series) -> pd.Series: 

2781 """:func:`_hover_cell` over a hover column.""" 

2782 return series.astype(object).map(_hover_cell) 

2783 

2784 

2785def _hover_payload( 

2786 frame: pd.DataFrame, 

2787 fields: Sequence[str], 

2788 *, 

2789 line_display: pd.Series | None = None, 

2790 table: str | None = None, 

2791 fixation: bool = False, 

2792 words: pd.DataFrame | None = None, 

2793) -> tuple[np.ndarray | None, str]: 

2794 """Plotly customdata + template for a user-selected field list (VIZ-26); 

2795 ``table`` says whose names label the rows (DATA-66). 

2796 

2797 ``fixation=True`` writes the fixation number, duration and word as one 

2798 plain lead line (#374 F7), the word by its text from ``words``; any other 

2799 chosen field follows as a ``Label: value`` row.""" 

2800 valid = [ 

2801 field 

2802 for field in fields 

2803 if field in frame.columns or (field == "line_idx" and line_display is not None) 

2804 ] 

2805 if not valid: 

2806 return None, "<extra></extra>" 

2807 values: list[pd.Series] = [] 

2808 rows: list[str] = [] 

2809 if fixation: 

2810 head = [field for field in _FIXATION_HEAD_FIELDS if field in valid] 

2811 if head: 

2812 values.append(_fixation_hover_head(frame, head, words)) 

2813 rows.append("%{customdata[0]}") 

2814 valid = [field for field in valid if field not in head] 

2815 for idx, field in enumerate(valid, start=len(values)): 

2816 series = ( 

2817 line_display 

2818 if field == "line_idx" and line_display is not None 

2819 else frame[field] 

2820 ) 

2821 values.append(_hover_cells(_plotly_literal_values(series))) 

2822 suffix = " ms" if field.endswith("_ms") else "" 

2823 rows.append(f"{_hover_label(field, table)}: %{{customdata[{idx}]}}{suffix}") 

2824 customdata = pd.concat(values, axis=1).to_numpy(dtype=object) 

2825 return customdata, "<br>".join(rows) + "<extra></extra>" 

2826 

2827 

2828#: The highlight key's annotation (#374 F6), so it is drawn once a figure. 

2829_HIGHLIGHT_KEY_NAME = "highlight_key" 

2830 

2831 

2832def _add_highlight_key( 

2833 fig: go.Figure, column: str, color: str, *, border: bool = False 

2834) -> None: 

2835 """Name the highlighted words on the figure itself (#374 F6): a swatch in 

2836 the highlight's colour and "Highlighted: is_in_aspan (answer span)", in 

2837 the plot's bottom-left corner. Drawn once however many panels mark words.""" 

2838 if any(a.name == _HIGHLIGHT_KEY_NAME for a in fig.layout.annotations or ()): 

2839 return 

2840 from .column_names import HIGHLIGHT_NOTES 

2841 

2842 note = HIGHLIGHT_NOTES.get(str(column)) 

2843 swatch = "▢" if border else "■" 

2844 text = ( 

2845 f'<span style="color:{color}">{swatch}</span> Highlighted: ' 

2846 f"{_plotly_literal(_column_name(column))}" + (f" ({note})" if note else "") 

2847 ) 

2848 fig.add_annotation( 

2849 x=0, 

2850 y=0, 

2851 xref="paper", 

2852 yref="paper", 

2853 xanchor="left", 

2854 yanchor="bottom", 

2855 xshift=6, 

2856 yshift=6, 

2857 text=text, 

2858 showarrow=False, 

2859 align="left", 

2860 font=dict(size=12, color="#444444"), 

2861 bgcolor="rgba(255,255,255,0.75)", 

2862 name=_HIGHLIGHT_KEY_NAME, 

2863 ) 

2864 

2865 

2866def _add_word_label_trace( 

2867 fig: go.Figure, 

2868 words: pd.DataFrame, 

2869 base_font_size: int, 

2870 font_family: str, 

2871 row: int | None = None, 

2872 col: int | None = None, 

2873 highlight_column: str | None = None, 

2874 text_color: str = WORD_LABEL_COLOR, 

2875 highlight_text_color: str = _CRITICAL_TEXT_COLOR, 

2876 word_hover_measure: str | None = None, 

2877 word_hover_fields: Sequence[str] | None = None, 

2878) -> None: 

2879 if words.empty or "text" not in words.columns: 

2880 return 

2881 customdata = None 

2882 hover = "Word: %{text}<extra></extra>" 

2883 if "word_id" in words.columns: 

2884 from .measures import cluster_word_lines 

2885 

2886 # The source ``line_idx`` is often a constant (OneStop IA exports rarely 

2887 # carry a real per-word line number), so infer the visual line from 

2888 # word-box geometry — same clustering the by-line coloring uses — and 

2889 # show it 1-based. 

2890 line_display = (cluster_word_lines(words) + 1).rename("line") 

2891 if word_hover_fields is not None: 

2892 customdata, hover = _hover_payload( 

2893 words, word_hover_fields, line_display=line_display, table="words" 

2894 ) 

2895 else: 

2896 # Legacy API/deep-link behaviour: the three fixed identity lines plus 

2897 # the old single optional measure. 

2898 customdata_parts: list[pd.Series] = [ 

2899 _plotly_literal_values(words["word_id"]), 

2900 line_display, 

2901 ] 

2902 hover = "Word: %{text}<br>Word #%{customdata[0]}<br>Line #%{customdata[1]}" 

2903 if word_hover_measure and word_hover_measure in words.columns: 

2904 label = _table_label( 

2905 word_hover_measure, "words" 

2906 ) or _HOVER_MEASURE_LABELS.get(word_hover_measure, word_hover_measure) 

2907 suffix = " ms" if word_hover_measure.endswith("_ms") else "" 

2908 hover += f"<br>{label}: %{{customdata[2]}}{suffix}" 

2909 customdata_parts.append( 

2910 _plotly_literal_values(words[word_hover_measure]) 

2911 ) 

2912 hover += "<extra></extra>" 

2913 customdata = pd.concat(customdata_parts, axis=1) 

2914 # Per-word text color: the highlight colour for highlighted words when the 

2915 # caller asks for "Mark text" (``highlight_column`` set), the base text 

2916 # colour otherwise. Both are configurable from the plot rail. 

2917 if highlight_column and highlight_column in words.columns: 

2918 critical_mask = words[highlight_column].fillna(False).astype(bool) 

2919 label_color = [ 

2920 highlight_text_color if is_crit else text_color for is_crit in critical_mask 

2921 ] 

2922 if critical_mask.any(): 

2923 _add_highlight_key(fig, highlight_column, highlight_text_color) 

2924 else: 

2925 label_color = text_color 

2926 # BUG-97 — the label is centred in its word's box, as the data defines it 

2927 # (`_word_label_x`: a padded layout's line-start word sits flush left, as it 

2928 # was shown). BUG-30 centred it on the glyph run instead, which on a tiling 

2929 # corpus (the box carries the following space) drew every word flush left. 

2930 # Centred text needs no LTR/RTL anchor; the Unicode direction isolates stay — 

2931 # they are about *shaping* mixed Hebrew/Arabic + punctuation, not placement. 

2932 from .preprocessing import detect_right_to_left 

2933 

2934 rtl = words.get("right_to_left") 

2935 if rtl is None: 

2936 rtl = words["text"].astype(str).map(detect_right_to_left) 

2937 else: 

2938 rtl = rtl.fillna(False).astype(bool) 

2939 label_x = _word_label_x(words) 

2940 # The word drawn as its own characters (finding 6 — not as Plotly markup), 

2941 # escaped before the direction isolates wrap it; the hover's `%{text}` 

2942 # reads this same string, so it shows the word literally too. 

2943 label_text = [ 

2944 f"\u2067{_plotly_literal(value)}\u2069" if is_rtl else _plotly_literal(value) 

2945 for value, is_rtl in zip(words["text"].astype(str), rtl) 

2946 ] 

2947 trace = go.Scatter( 

2948 x=label_x, 

2949 y=words["y"] + words["height"] / 2, 

2950 text=label_text, 

2951 mode="text", 

2952 textposition="middle center", 

2953 showlegend=False, 

2954 textfont=dict(color=label_color, size=base_font_size, family=font_family), 

2955 hovertemplate=hover, 

2956 customdata=customdata, 

2957 name="words", 

2958 ) 

2959 if row is not None and col is not None: 

2960 fig.add_trace(trace, row=row, col=col) 

2961 else: 

2962 fig.add_trace(trace) 

2963 

2964 

2965_IMAGE_MIME = { 

2966 ".png": "image/png", 

2967 ".jpg": "image/jpeg", 

2968 ".jpeg": "image/jpeg", 

2969 ".gif": "image/gif", 

2970 ".webp": "image/webp", 

2971} 

2972 

2973 

2974def _image_to_data_uri(src: str | None) -> str | None: 

2975 """A ``data:`` URI for an image path, or pass through an existing one. 

2976 

2977 Returns None for a missing / unreadable file so the background-image layer 

2978 simply doesn't draw (e.g. an uploaded MultiplEYE dataset has no image path).""" 

2979 if not src: 

2980 return None 

2981 text = str(src) 

2982 if text.startswith("data:"): 

2983 return text 

2984 path = Path(text) 

2985 try: 

2986 raw = path.read_bytes() 

2987 except OSError: 

2988 return None 

2989 mime = _IMAGE_MIME.get(path.suffix.lower(), "image/png") 

2990 return f"data:{mime};base64," + base64.b64encode(raw).decode("ascii") 

2991 

2992 

2993def _png_pixel_size(src: str | None) -> tuple[int, int] | None: 

2994 """(width, height) of a PNG from its header, without Pillow; None otherwise.""" 

2995 if not src: 

2996 return None 

2997 try: 

2998 with open(src, "rb") as fh: 

2999 head = fh.read(24) 

3000 except OSError: 

3001 return None 

3002 if head[:8] == b"\x89PNG\r\n\x1a\n" and head[12:16] == b"IHDR": 

3003 width, height = struct.unpack(">II", head[16:24]) 

3004 return int(width), int(height) 

3005 return None 

3006 

3007 

3008def _background_image_spec( 

3009 background_image: str | None, 

3010 background_image_size: tuple[float, float] | None, 

3011 background_image_origin: tuple[float, float] | None, 

3012 background_image_opacity: float = 1.0, 

3013) -> dict | None: 

3014 """The ``layout.image`` dict for the stimulus-page background (VIZ-4). 

3015 

3016 The rendered page sits at data coordinates ``(origin_x, origin_y)`` → 

3017 ``(+image_w, +image_h)`` UNDER every other layer; the coords are where the 

3018 (centered) stimulus appeared on the monitor, so the image aligns exactly and 

3019 sidesteps CJK/RTL font rendering. ``background_image_origin`` defaults to 

3020 ``(0, 0)``; ``yanchor="top"`` + the reversed y-axis put the image's top-left 

3021 there. ``background_image_opacity`` dims a busy stimulus so the AOIs/scanpath 

3022 read over it (1.0 = opaque). Returns ``None`` when there is nothing to draw 

3023 (no image, no size, or an unreadable path), so callers can just skip it. 

3024 """ 

3025 if not (background_image and background_image_size): 

3026 return None 

3027 uri = _image_to_data_uri(background_image) 

3028 if not uri: 

3029 return None 

3030 image_w, image_h = background_image_size 

3031 origin_x, origin_y = background_image_origin or (0.0, 0.0) 

3032 return dict( 

3033 source=uri, 

3034 xref="x", 

3035 yref="y", 

3036 x=origin_x, 

3037 y=origin_y, 

3038 sizex=image_w, 

3039 sizey=image_h, 

3040 sizing="stretch", 

3041 layer="below", 

3042 xanchor="left", 

3043 yanchor="top", 

3044 opacity=float(background_image_opacity), 

3045 ) 

3046 

3047 

3048def _add_background_image( 

3049 fig: go.Figure, 

3050 background_image: str | None, 

3051 background_image_size: tuple[float, float] | None, 

3052 background_image_origin: tuple[float, float] | None, 

3053 background_image_opacity: float = 1.0, 

3054 *, 

3055 row: int | None = None, 

3056 col: int | None = None, 

3057) -> bool: 

3058 """Add the stimulus-page background image to ``fig``; True when one was added. 

3059 

3060 ``row``/``col`` bind it to one subplot's axes (the split comparison figure); 

3061 without them it goes on the figure's single axis pair. 

3062 """ 

3063 spec = _background_image_spec( 

3064 background_image, 

3065 background_image_size, 

3066 background_image_origin, 

3067 background_image_opacity, 

3068 ) 

3069 if spec is None: 

3070 return False 

3071 if row is not None and col is not None: 

3072 fig.add_layout_image(spec, row=row, col=col) 

3073 else: 

3074 fig.add_layout_image(spec) 

3075 return True 

3076 

3077 

3078def _add_saccade_layer( 

3079 fig: go.Figure, 

3080 fixations: pd.DataFrame, 

3081 *, 

3082 x_field: str, 

3083 y_field: str, 

3084 color: str, 

3085 width: float, 

3086 style: str, 

3087 show_arrows: bool, 

3088 saccade_classes: pd.Series | None = None, 

3089 color_by_class: bool = True, 

3090 class_colors: dict | None = None, 

3091 class_legend: bool = True, 

3092 visible_classes: Iterable[str] | None = None, 

3093 render_mode: str = "Straight", 

3094 two_way: bool = False, 

3095) -> bool: 

3096 """Add one scanpath's saccade lines (+ optional direction arrowheads) to ``fig``. 

3097 

3098 Connects consecutive fixations in time order. When ``saccade_classes`` is 

3099 given (the per-fixation reading class from ``measures.classify_saccades``) 

3100 the segments are grouped by class; with ``color_by_class`` (VIZ-8 "By type") 

3101 each group becomes its own sub-trace in its colour from ``class_colors`` 

3102 plus a small legend, and ``two_way`` (VIZ-19) first folds the five classes 

3103 down to forward vs. regression. Otherwise a single uniform-``color`` trace is 

3104 drawn. ``visible_classes`` (VIZ-31) is the reading-class **filter**: segments 

3105 whose class isn't listed are not drawn at all — so it needs the 

3106 classification even in uniform-colour mode, which is why ``color_by_class`` 

3107 exists separately. Filtering happens on the *unfolded* classes, before the 

3108 two-way fold, so "show only regressions" means the same thing in every 

3109 colour mode. It is read literally: an **empty** ``visible_classes`` draws no 

3110 saccades. (The rail never sends one — ``controls._collect_viz_settings`` 

3111 reads a cleared multiselect as "no filter", since hiding the layer outright 

3112 is what its toggle is for.) ``render_mode="Arc"`` (VIZ-9) draws each saccade as an upward 

3113 arch instead of a straight connector. The arrowheads are a separate, 

3114 independently-toggled trace drawn before the fixation markers so the dots sit 

3115 on top (uniform ``color`` either way — they encode direction, the line colour 

3116 encodes type), and they honour the same filter. The caller gates this on 

3117 spatial axes, ``show_saccades`` and at least two fixations. 

3118 

3119 Returns ``True`` when legend entries were added (by-type mode), so the caller 

3120 reserves margin for the legend (mirrors ``_add_raw_gaze_layer``). 

3121 """ 

3122 legend_added = False 

3123 arch_frac = _ARCH_FRAC if render_mode == "Arc" else None 

3124 keep = None if visible_classes is None else set(visible_classes) 

3125 # Group once, on the raw (unfolded) classes, then filter — the two-way fold 

3126 # and the uniform-colour merge below both work off the same grouping. 

3127 segs = None 

3128 if saccade_classes is not None: 

3129 segs = _saccade_segments_by_class( 

3130 fixations, x_field, y_field, saccade_classes, arch_frac 

3131 ) 

3132 if keep is not None: 

3133 segs = {c: s for c, s in segs.items() if c in keep} 

3134 

3135 if segs is not None and color_by_class: 

3136 # Merge over the full defaults so classes the caller omits (notably the 

3137 # non-editable "other" catch-all — the UI palette only carries the five 

3138 # reading classes) still get their intended colour instead of falling 

3139 # back to the uniform line colour. 

3140 palette = {**SACCADE_CLASS_COLORS, **(class_colors or {})} 

3141 # VIZ-19: the two-way mode is the five-way one with the classes folded 

3142 # into forward/regression buckets, so everything below — segments, 

3143 # colours, legend — is shared. "Forward" takes the forward class's 

3144 # colour, which is what the picker shows for it. 

3145 if two_way: 

3146 folded: dict = {} 

3147 for cls_name, (fx, fy) in segs.items(): 

3148 bucket = SACCADE_DIRECTION_FOLD.get(cls_name, "other") 

3149 bx, by = folded.setdefault(bucket, ([], [])) 

3150 bx.extend(fx) 

3151 by.extend(fy) 

3152 segs = folded 

3153 draw_order = [*SACCADE_DIRECTION_CLASSES, "other"] 

3154 labels = SACCADE_DIRECTION_LABELS 

3155 legend_title = "Saccade direction" 

3156 else: 

3157 draw_order = SACCADE_CLASS_ORDER 

3158 labels = SACCADE_CLASS_LABELS 

3159 legend_title = "Saccade type" 

3160 for cls_name in draw_order: 

3161 seg = segs.get(cls_name) 

3162 if not seg or not seg[0]: 

3163 continue 

3164 sx, sy = seg 

3165 fig.add_trace( 

3166 go.Scatter( 

3167 x=sx, 

3168 y=sy, 

3169 mode="lines", 

3170 line=dict( 

3171 color=palette.get(cls_name, color), width=width, dash=style 

3172 ), 

3173 hoverinfo="skip", 

3174 # VIZ-8: the colour key is optional — hide it (but keep the 

3175 # coloured sub-traces) when class_legend is off. 

3176 showlegend=class_legend, 

3177 legendgroup="saccade_type", 

3178 legendgrouptitle_text=legend_title, 

3179 name=labels.get(cls_name, cls_name), 

3180 ) 

3181 ) 

3182 # Reserve legend margin only when the key is actually shown. 

3183 legend_added = class_legend 

3184 else: 

3185 if segs is None: 

3186 sx, sy = _saccade_segments(fixations, x_field, y_field, arch_frac) 

3187 else: 

3188 # Filtered, but drawn in one uniform colour: concatenate the classes 

3189 # that survived. Each class's arrays are already None-separated, so 

3190 # joining them is the same trace the unfiltered path builds minus the 

3191 # hidden segments (their order within the trace doesn't render). 

3192 sx, sy = [], [] 

3193 for cls_name in [ 

3194 *SACCADE_CLASS_ORDER, 

3195 *(c for c in segs if c not in SACCADE_CLASS_ORDER), 

3196 ]: 

3197 seg = segs.get(cls_name) 

3198 if seg: 

3199 sx.extend(seg[0]) 

3200 sy.extend(seg[1]) 

3201 if sx: 

3202 fig.add_trace( 

3203 go.Scatter( 

3204 x=sx, 

3205 y=sy, 

3206 mode="lines", 

3207 line=dict(color=color, width=width, dash=style), 

3208 hoverinfo="skip", 

3209 showlegend=False, 

3210 name="saccades", 

3211 ) 

3212 ) 

3213 if show_arrows: 

3214 # BUG-9: same arch_frac as the segments above, so the arrowheads sit on 

3215 # the drawn line (arched or straight) instead of on the chord. 

3216 amx, amy, aang, aseg = _saccade_arrow_rows( 

3217 fixations, x_field, y_field, arch_frac 

3218 ) 

3219 if amx and keep is not None and saccade_classes is not None: 

3220 mask = _arrow_class_mask(fixations, saccade_classes, keep, aseg) 

3221 amx = [v for v, m in zip(amx, mask) if m] 

3222 amy = [v for v, m in zip(amy, mask) if m] 

3223 aang = [v for v, m in zip(aang, mask) if m] 

3224 if amx: 

3225 fig.add_trace( 

3226 go.Scatter( 

3227 x=amx, 

3228 y=amy, 

3229 mode="markers", 

3230 marker=dict( 

3231 symbol="arrow", 

3232 size=12, 

3233 angle=aang, 

3234 angleref="up", 

3235 color=color, 

3236 line=dict(width=0), 

3237 ), 

3238 hoverinfo="skip", 

3239 showlegend=False, 

3240 name="saccade direction", 

3241 ) 

3242 ) 

3243 return legend_added 

3244 

3245 

3246def _add_raw_gaze_layer( 

3247 fig: go.Figure, 

3248 raw_gaze: pd.DataFrame | None, 

3249 *, 

3250 show_raw_gaze: bool, 

3251 raw_gaze_color: str = "#888888", 

3252 raw_gaze_marker_size: float = 4.0, 

3253 raw_gaze_opacity: float = 0.6, 

3254) -> bool: 

3255 """Add the raw-gaze sample-point scatter (time-coloured when available). 

3256 

3257 Returns ``True`` if a trace was added, so the caller can mark the legend 

3258 active (it reserves margin for the legend). Self-gates on ``show_raw_gaze`` 

3259 and a non-empty frame. **UX-86**: ``raw_gaze_color`` is the flat colour 

3260 used when the data carries no ``timestamp_ms`` — when it does, fixations 

3261 stay time-mapped (Viridis) since that is the more informative default and 

3262 a flat colour would throw it away. Samples imported without a clock carry 

3263 ``sample_index`` instead: coloured by that order, with the legend titled 

3264 *Sample order* and the hover saying ``sample n`` — never milliseconds. 

3265 """ 

3266 if not (show_raw_gaze and raw_gaze is not None and not raw_gaze.empty): 

3267 return False 

3268 legend_title = None 

3269 if "timestamp_ms" in raw_gaze.columns: 

3270 color_vals = raw_gaze["timestamp_ms"] 

3271 colorscale = "Viridis" 

3272 customdata = raw_gaze["timestamp_ms"] 

3273 when = "<br>Timestamp: %{customdata} ms" 

3274 elif SAMPLE_INDEX in raw_gaze.columns: 

3275 # No clock (the import mapped none): coloured by the samples' order, 

3276 # and said so — a ramp with no title would read as time. 

3277 color_vals = raw_gaze[SAMPLE_INDEX] 

3278 colorscale = "Viridis" 

3279 customdata = raw_gaze[SAMPLE_INDEX] 

3280 when = "<br>Sample #: %{customdata}" 

3281 legend_title = "Sample order" 

3282 else: 

3283 color_vals = raw_gaze_color 

3284 colorscale = None 

3285 customdata = None 

3286 when = "" 

3287 fig.add_trace( 

3288 go.Scatter( 

3289 x=raw_gaze["x"], 

3290 y=raw_gaze["y"], 

3291 mode="markers", 

3292 marker=dict( 

3293 size=raw_gaze_marker_size, 

3294 color=color_vals, 

3295 colorscale=colorscale, 

3296 opacity=raw_gaze_opacity, 

3297 showscale=False, 

3298 ), 

3299 hovertemplate=( 

3300 "Raw gaze<br>x: %{x:.1f}<br>y: %{y:.1f}" + when + "<extra></extra>" 

3301 ), 

3302 customdata=customdata, 

3303 legendgroup="raw_gaze" if legend_title else None, 

3304 legendgrouptitle_text=legend_title, 

3305 name="Raw gaze", 

3306 showlegend=True, 

3307 ) 

3308 ) 

3309 return True 

3310 

3311 

3312# PRE-2 fixation classification (viz-only): SHORT / LONG / OUT-OF-BOUNDS, each 

3313# Off / Highlight / Discard. Thresholds come from the caller's flags dict; these 

3314# are the fallbacks the app's own controls default to. 

3315_FIX_FLAG_SHORT_MS = 80.0 

3316_FIX_FLAG_LONG_MS = 800.0 

3317_FIX_FLAG_CATEGORIES = ("short", "long", "oob", "blink") 

3318_FIX_FLAG_LABELS = { 

3319 "short": "Short", 

3320 "long": "Long", 

3321 "oob": "Out of bounds", 

3322 "blink": "Blink-adjacent", 

3323} 

3324 

3325 

3326def _fixation_flag_masks( 

3327 fixations: pd.DataFrame, 

3328 words: pd.DataFrame, 

3329 flags: dict | None, 

3330 *, 

3331 spatial_axes: bool = True, 

3332) -> dict[str, pd.Series]: 

3333 """Boolean mask per PRE-2 fixation-flag category, over ``fixations``' index. 

3334 

3335 ``{"short": …, "long": …, "oob": …}``, or ``{}`` when nothing is flagged. 

3336 Out-of-bounds needs word boxes on spatial axes — without them no fixation 

3337 counts as out of bounds. Shared by the static figure and the animated replay 

3338 so the two classify identically (VIZ-23). 

3339 """ 

3340 if not flags or fixations.empty: 

3341 return {} 

3342 dur = pd.to_numeric(fixations.get("duration_ms"), errors="coerce") 

3343 oob = pd.Series(False, index=fixations.index) 

3344 if spatial_axes and not words.empty: 

3345 from .measures import fixation_in_text_mask 

3346 

3347 oob = ~fixation_in_text_mask(fixations, words) 

3348 short_ms = float(flags.get("short", {}).get("threshold_ms", _FIX_FLAG_SHORT_MS)) 

3349 long_ms = float(flags.get("long", {}).get("threshold_ms", _FIX_FLAG_LONG_MS)) 

3350 blink = pd.Series(False, index=fixations.index) 

3351 for column in ("is_blink", "blink_before", "blink_after", "blink"): 

3352 if column in fixations: 

3353 blink |= fixations[column].fillna(False).astype(bool) 

3354 return { 

3355 "short": (dur < short_ms).fillna(False).astype(bool), 

3356 "long": (dur > long_ms).fillna(False).astype(bool), 

3357 "oob": oob.fillna(False).astype(bool), 

3358 "blink": blink, 

3359 } 

3360 

3361 

3362def _discard_flagged_fixations( 

3363 fixations: pd.DataFrame, 

3364 words: pd.DataFrame, 

3365 flags: dict | None, 

3366 *, 

3367 spatial_axes: bool = True, 

3368) -> pd.DataFrame: 

3369 """Drop the fixations whose PRE-2 flag mode is *Discard* (viz-only). 

3370 

3371 Changes only what is DRAWN — the returned frame feeds the markers, the 

3372 fixation-index labels and the marker-size scaling; reading measures and 

3373 exports are untouched. Returns ``fixations`` itself when nothing is dropped. 

3374 """ 

3375 masks = _fixation_flag_masks(fixations, words, flags, spatial_axes=spatial_axes) 

3376 if not masks: 

3377 return fixations 

3378 drop = pd.Series(False, index=fixations.index) 

3379 for category, mask in masks.items(): 

3380 if (flags or {}).get(category, {}).get("mode") == "Discard": 

3381 drop = drop | mask 

3382 return fixations[~drop] if bool(drop.any()) else fixations 

3383 

3384 

3385def _render_scanpath_figure( 

3386 words: pd.DataFrame, 

3387 fixations: pd.DataFrame, 

3388 *, 

3389 settings: FigureSettings, 

3390 raw_gaze: pd.DataFrame | None = None, 

3391) -> go.Figure: 

3392 canvas_width = settings.canvas_width 

3393 canvas_height = settings.canvas_height 

3394 base_font_size = settings.base_font_size 

3395 font_family = settings.font_family 

3396 x_field = settings.x_field 

3397 y_field = settings.y_field 

3398 show_words = settings.show_words 

3399 show_word_labels = settings.show_word_labels 

3400 show_fixations = settings.show_fixations 

3401 show_order = settings.show_order 

3402 show_saccades = settings.show_saccades 

3403 show_heatmap = settings.show_heatmap 

3404 color_by = settings.color_by 

3405 heatmap_metric = settings.heatmap_metric 

3406 show_saccade_arrows = settings.show_saccade_arrows 

3407 heatmap_style = settings.heatmap_style 

3408 heatmap_norm = settings.heatmap_norm 

3409 heatmap_sigma_px = settings.heatmap_sigma_px 

3410 marker_size_range = settings.marker_size_range 

3411 order_font_size = settings.order_font_size 

3412 order_font_color = settings.order_font_color 

3413 show_colorbars = settings.show_fixation_colorbar 

3414 show_heatmap_colorbar = settings.show_heatmap_colorbar 

3415 fixation_color_range = settings.fixation_color_range 

3416 heatmap_range = settings.heatmap_range 

3417 fixation_colorscale = settings.fixation_colorscale 

3418 heatmap_colorscale = settings.heatmap_colorscale 

3419 show_raw_gaze = settings.show_raw_gaze 

3420 raw_gaze_color = settings.raw_gaze_color 

3421 raw_gaze_marker_size = settings.raw_gaze_marker_size 

3422 raw_gaze_opacity = settings.raw_gaze_opacity 

3423 critical_span_style = settings.critical_span_style 

3424 highlight_column = settings.highlight_column 

3425 saccade_color = settings.saccade_color 

3426 saccade_style = settings.saccade_style 

3427 saccade_width = settings.saccade_width 

3428 saccade_color_mode = settings.saccade_color_mode 

3429 saccade_class_colors = settings.saccade_class_colors 

3430 saccade_type_legend = settings.saccade_type_legend 

3431 saccade_classes = settings.saccade_classes 

3432 saccade_render_mode = settings.saccade_render_mode 

3433 fixation_snap_to_word = settings.fixation_snap_to_word 

3434 hollow_fixations = settings.hollow_fixations 

3435 fixation_opacity = settings.fixation_opacity 

3436 fixation_color = settings.fixation_color 

3437 fixation_symbol = settings.fixation_symbol 

3438 text_color = settings.text_color 

3439 highlight_text_color = settings.highlight_text_color 

3440 background_color = settings.background_color 

3441 # BUG-85: `color_by="line"` is the rail's own spelling of this (its "line" 

3442 # option, a share link's `color_by=line`), so it colours by line from the 

3443 # API and CLI too rather than falling through to a missing column. 

3444 color_by_line = settings.color_by_line or settings.color_by == "line" 

3445 fixation_flags = settings.fixation_flags 

3446 span_border_color = settings.span_border_color 

3447 # The fixations' colour bar; the heatmap's is `cb_style` below. 

3448 colorbar_orientation = settings.fixation_colorbar_orientation 

3449 colorbar_tickangle = settings.fixation_colorbar_tickangle 

3450 colorbar_tickfont_size = settings.fixation_colorbar_tickfont_size 

3451 line_spacing = settings.line_spacing 

3452 scale_text_to_boxes = settings.scale_text_to_boxes 

3453 background_image = settings.background_image 

3454 background_image_size = settings.background_image_size 

3455 background_image_origin = settings.background_image_origin 

3456 background_image_opacity = settings.background_image_opacity 

3457 fit_to_monitor = settings.fit_to_monitor 

3458 show_coordinate_grid = settings.show_coordinate_grid 

3459 coordinate_grid_spacing = settings.coordinate_grid_spacing 

3460 word_heatmap_col = settings.word_heatmap_col 

3461 word_heatmap_title = settings.word_heatmap_title 

3462 word_hover_measure = settings.word_hover_measure 

3463 word_hover_fields = settings.word_hover_fields 

3464 fixation_hover_fields = settings.fixation_hover_fields 

3465 show_connectors = settings.show_connectors 

3466 connector_y = settings.connector_y 

3467 illustration_reasons = settings.illustration_reasons 

3468 fig = go.Figure() 

3469 spatial_axes = x_field == "x" and y_field == "y" 

3470 # Track whether a colorbar / legend will render, to reserve margin for them 

3471 # below (so they don't shrink the equal-aspect plot). See _decoration_margins. 

3472 legend_active = False 

3473 heatmap_rendered = False 

3474 is_numeric_color = False 

3475 # The heatmap's colour-bar styling (orientation / tick angle / tick size). 

3476 cb_style = dict( 

3477 orientation=settings.heatmap_colorbar_orientation, 

3478 tickangle=settings.heatmap_colorbar_tickangle, 

3479 tickfont_size=settings.heatmap_colorbar_tickfont_size, 

3480 ) 

3481 font_settings = dict(family=font_family or FONT_FAMILY, size=base_font_size) 

3482 

3483 raw_for_range = raw_gaze if (show_raw_gaze and raw_gaze is not None) else None 

3484 if spatial_axes: 

3485 x_range, y_range, x_min_data, x_max_data, y_min_data, y_max_data = ( 

3486 _compute_axis_ranges( 

3487 canvas_width, 

3488 canvas_height, 

3489 (fixations, x_field, y_field), 

3490 (raw_for_range, "x", "y"), 

3491 word_frames=[words] if not words.empty else [], 

3492 fit_to_monitor=fit_to_monitor, 

3493 ) 

3494 ) 

3495 else: 

3496 x_range = [0, canvas_width] 

3497 y_range = [canvas_height, 0] 

3498 x_min_data = x_max_data = y_min_data = y_max_data = None 

3499 

3500 # VIZ-9 "linear reading" mode: snap each fixation above the word it lands on, 

3501 # so the saccade layer AND the fixation markers below draw from the snapped 

3502 # positions. Off by default. The axis ranges above and the heatmaps below keep 

3503 # the RECORDED positions (raw gaze density); only the drawn connectors and 

3504 # markers move — and the Arc headroom just below, which must follow them. 

3505 render_fix = fixations 

3506 if ( 

3507 fixation_snap_to_word 

3508 and spatial_axes 

3509 and not fixations.empty 

3510 and not words.empty 

3511 ): 

3512 render_fix = _snap_fixations_to_words(fixations, words, x_field, y_field) 

3513 

3514 # VIZ-9 arc mode: the saccade arches rise ABOVE the fixations, so reserve 

3515 # headroom at the top of the view (smaller y — the axis is inverted) or a wide 

3516 # top-line saccade's apex gets clipped. Computed from the exact Bézier apex of 

3517 # each segment so it's tight; only in Arc mode, so the default view is 

3518 # unchanged. The apexes come from ``render_fix`` — the coordinates the 

3519 # connectors are actually drawn from — because Snap to word can both widen a 

3520 # saccade (two near-edge fixations jump to their words' centres) and lift its 

3521 # endpoints (to the box tops), so an arc over the recorded positions would 

3522 # under-reserve and clip the snapped curve. 

3523 # Whole-monitor view (``fit_to_monitor``): the range still starts as the full 

3524 # screen, and this only ever *grows* it — past the screen's top edge when an 

3525 # arc would reach it — so a schematic arc is never silently cut off; the 

3526 # screen itself is always shown whole. 

3527 if ( 

3528 spatial_axes 

3529 and saccade_render_mode == "Arc" 

3530 and show_saccades 

3531 and len(render_fix) > 1 

3532 ): 

3533 fo = render_fix.sort_values("timestamp_ms") 

3534 fxv = pd.to_numeric(fo[x_field], errors="coerce").to_numpy(dtype=float) 

3535 fyv = pd.to_numeric(fo[y_field], errors="coerce").to_numpy(dtype=float) 

3536 apex = np.inf 

3537 for i in range(len(fo) - 1): 

3538 x0, y0, x1, y1 = fxv[i], fyv[i], fxv[i + 1], fyv[i + 1] 

3539 if not np.isfinite((x0, y0, x1, y1)).all(): 

3540 continue 

3541 # BUG-13: the curve's true peak, not its t=0.5 point — for endpoints 

3542 # that differ in y the parabola crests higher than the midpoint 

3543 # sample, and a wide top-line arc then clipped. 

3544 apex = min(apex, _arch_apex_y(x0, y0, x1, y1, _ARCH_FRAC)) 

3545 if np.isfinite(apex): 

3546 margin = 0.02 * abs(y_range[0] - y_range[1]) 

3547 y_range = [y_range[0], min(y_range[1], apex - margin)] 

3548 

3549 # Stimulus-page background image (MultiplEYE / VIZ-4) — see 

3550 # `_background_image_spec` for the placement contract. 

3551 if spatial_axes: 

3552 _add_background_image( 

3553 fig, 

3554 background_image, 

3555 background_image_size, 

3556 background_image_origin, 

3557 background_image_opacity, 

3558 ) 

3559 

3560 # Fix the display size up front so the data->screen scale is known: word 

3561 # labels are then sized in that scale (true-to-scale text), and the same 

3562 # fitted_w/fitted_h drive the final layout below. 

3563 fitted_w, fitted_h = _fit_display_size( 

3564 canvas_width, canvas_height, x_range, y_range, spatial_axes 

3565 ) 

3566 scale = ( 

3567 _display_scale(x_range, y_range, fitted_w, fitted_h) if spatial_axes else 1.0 

3568 ) 

3569 label_font_px = _word_label_font_px( 

3570 words, 

3571 scale=scale, 

3572 line_spacing=line_spacing, 

3573 manual_font_px=base_font_size, 

3574 scale_text_to_boxes=scale_text_to_boxes, 

3575 ) 

3576 

3577 # Words flagged by ``highlight_column`` (default the OneStop answer span 

3578 # ``is_in_aspan``) get marked one of two ways: 

3579 # - "Mark text": color the highlighted words dark pink (no border). 

3580 # - "Mark border": draw a thin black outline around the span. 

3581 has_highlight = ( 

3582 bool(highlight_column) 

3583 and highlight_column in words.columns 

3584 and not words.empty 

3585 and bool(words[highlight_column].fillna(False).astype(bool).any()) 

3586 ) 

3587 highlight_text = has_highlight and critical_span_style == "Mark text" 

3588 

3589 if spatial_axes and not words.empty: 

3590 # Word-box grid (the "Bounding boxes" layer) and the "Mark border" span 

3591 # overlay are independent: the span borders show even when the boxes are 

3592 # off (then only the span outline is drawn). 

3593 shapes = ( 

3594 build_word_boxes( 

3595 words, 

3596 color=settings.word_box_color, 

3597 fill_color=settings.word_box_fill_color, 

3598 fill_opacity=settings.word_box_fill_opacity, 

3599 line_opacity=settings.word_box_line_opacity, 

3600 ) 

3601 if show_words 

3602 else [] 

3603 ) 

3604 if has_highlight and critical_span_style == "Mark border": 

3605 shapes = shapes + build_critical_span_overlay( 

3606 words, highlight_column, color=span_border_color 

3607 ) 

3608 _add_highlight_key(fig, highlight_column, span_border_color, border=True) 

3609 if shapes: 

3610 fig.update_layout(shapes=shapes) 

3611 if show_word_labels: 

3612 _add_word_label_trace( 

3613 fig, 

3614 words, 

3615 label_font_px, 

3616 font_settings["family"], 

3617 highlight_column=highlight_column if highlight_text else None, 

3618 text_color=text_color, 

3619 highlight_text_color=highlight_text_color, 

3620 word_hover_measure=word_hover_measure, 

3621 word_hover_fields=word_hover_fields, 

3622 ) 

3623 

3624 if _add_raw_gaze_layer( 

3625 fig, 

3626 raw_gaze, 

3627 show_raw_gaze=show_raw_gaze, 

3628 raw_gaze_color=raw_gaze_color, 

3629 raw_gaze_marker_size=raw_gaze_marker_size, 

3630 raw_gaze_opacity=raw_gaze_opacity, 

3631 ): 

3632 legend_active = True 

3633 

3634 if spatial_axes and show_heatmap and not fixations.empty: 

3635 heatmap_rendered = True 

3636 weights = fixations["duration_ms"] if heatmap_metric == "duration_ms" else None 

3637 x_min = ( 

3638 x_min_data if x_min_data is not None else float(fixations[x_field].min()) 

3639 ) 

3640 x_max = ( 

3641 x_max_data if x_max_data is not None else float(fixations[x_field].max()) 

3642 ) 

3643 y_min = ( 

3644 y_min_data if y_min_data is not None else float(fixations[y_field].min()) 

3645 ) 

3646 y_max = ( 

3647 y_max_data if y_max_data is not None else float(fixations[y_field].max()) 

3648 ) 

3649 if heatmap_style == "Interpolated": 

3650 _add_interpolated_heatmap( 

3651 fig, 

3652 fixations, 

3653 x_field=x_field, 

3654 y_field=y_field, 

3655 x_min=x_min, 

3656 x_max=x_max, 

3657 y_min=y_min, 

3658 y_max=y_max, 

3659 weights=weights, 

3660 heatmap_colorscale=heatmap_colorscale, 

3661 show_colorbars=show_heatmap_colorbar, 

3662 heatmap_norm=heatmap_norm, 

3663 colorbar_style=cb_style, 

3664 sigma_px=heatmap_sigma_px, 

3665 ) 

3666 elif not words.empty: 

3667 _add_word_level_heatmap( 

3668 fig, 

3669 words, 

3670 fixations, 

3671 x_field=x_field, 

3672 y_field=y_field, 

3673 weights=weights, 

3674 heatmap_colorscale=heatmap_colorscale, 

3675 heatmap_range=heatmap_range, 

3676 show_colorbars=show_heatmap_colorbar, 

3677 heatmap_norm=heatmap_norm, 

3678 colorbar_style=cb_style, 

3679 ) 

3680 else: 

3681 _add_density_heatmap( 

3682 fig, 

3683 fixations, 

3684 x_field=x_field, 

3685 y_field=y_field, 

3686 x_min=x_min, 

3687 x_max=x_max, 

3688 y_min=y_min, 

3689 y_max=y_max, 

3690 weights=weights, 

3691 heatmap_colorscale=heatmap_colorscale, 

3692 heatmap_range=heatmap_range, 

3693 show_colorbars=show_heatmap_colorbar, 

3694 heatmap_norm=heatmap_norm, 

3695 colorbar_style=cb_style, 

3696 ) 

3697 elif spatial_axes and show_heatmap and not words.empty: 

3698 # Words-only dataset (no fixation report): fall back to the words 

3699 # frame's own pre-aggregated reading measures for the box heatmap. The 

3700 # corpus "word difficulty on the stimulus" view (AN-4) passes an explicit 

3701 # ``word_heatmap_col`` + title so it can tint by any aggregate or rate. 

3702 if word_heatmap_col is not None and word_heatmap_col in words.columns: 

3703 heatmap_rendered = True 

3704 values = pd.to_numeric(words[word_heatmap_col], errors="coerce").fillna(0.0) 

3705 _draw_word_value_heatmap( 

3706 fig, 

3707 words, 

3708 [float(v) for v in values], 

3709 heatmap_colorscale=heatmap_colorscale, 

3710 heatmap_range=heatmap_range, 

3711 show_colorbars=show_heatmap_colorbar, 

3712 heatmap_norm=heatmap_norm, 

3713 colorbar_title=_plotly_literal(word_heatmap_title or "Value"), 

3714 colorbar_style=cb_style, 

3715 ) 

3716 else: 

3717 measure = ( 

3718 "total_fixation_duration_ms" 

3719 if heatmap_metric == "duration_ms" 

3720 else "n_fixations" 

3721 ) 

3722 if measure in words.columns: 

3723 heatmap_rendered = True 

3724 _add_word_measure_heatmap( 

3725 fig, 

3726 words, 

3727 measure, 

3728 heatmap_colorscale=heatmap_colorscale, 

3729 heatmap_range=heatmap_range, 

3730 show_colorbars=show_heatmap_colorbar, 

3731 heatmap_norm=heatmap_norm, 

3732 colorbar_style=cb_style, 

3733 ) 

3734 

3735 # Saccade lines + optional direction arrowheads (drawn before the fixation 

3736 # markers so the dots sit on top). 

3737 if spatial_axes and show_saccades and len(fixations) > 1: 

3738 # VIZ-8: "By type" colours each saccade by its reading class. The class is 

3739 # geometry-derived per trial (like color-by-line above), so compute it 

3740 # here from the trial's fixations + words — reuse a precomputed 

3741 # `saccade_class` column if the pipeline already added one. Classify on the 

3742 # RAW fixations (word_id is unchanged by the snap). 

3743 # VIZ-19: "Forward / regression" is the same classification folded into 

3744 # two buckets, so both non-uniform modes take this path. 

3745 # VIZ-31: the same classification also backs the reading-class *filter* 

3746 # (`saccade_classes` = the classes to draw, ``None`` meaning all), so 

3747 # classify whenever either the colour mode or the filter needs it. A 

3748 # filter naming every class is a no-op and takes the cheaper raw path. 

3749 class_series = None 

3750 two_way_saccades = saccade_color_mode == "Forward / regression" 

3751 color_by_class = saccade_color_mode in ("By type", "Forward / regression") 

3752 visible_classes = ( 

3753 None 

3754 if saccade_classes is None 

3755 or set(saccade_classes) >= set(SACCADE_CLASS_ORDER) 

3756 else set(saccade_classes) 

3757 ) 

3758 if color_by_class or visible_classes is not None: 

3759 existing = fixations.get("saccade_class") 

3760 if existing is not None: 

3761 class_series = existing 

3762 else: 

3763 from .measures import classify_saccades 

3764 

3765 class_series = classify_saccades(fixations, words) 

3766 if _add_saccade_layer( 

3767 fig, 

3768 render_fix, 

3769 x_field=x_field, 

3770 y_field=y_field, 

3771 color=saccade_color, 

3772 width=saccade_width, 

3773 style=saccade_style, 

3774 show_arrows=show_saccade_arrows, 

3775 saccade_classes=class_series, 

3776 color_by_class=color_by_class, 

3777 class_colors=saccade_class_colors, 

3778 class_legend=saccade_type_legend, 

3779 visible_classes=visible_classes, 

3780 render_mode=saccade_render_mode, 

3781 two_way=two_way_saccades, 

3782 ): 

3783 legend_active = True 

3784 

3785 # Drift connectors (PRE-3): one faint grey vertical segment per fixation from 

3786 # its original y (`connector_y`) to its drift-corrected y (the already-snapped 

3787 # `fixations["y"]`). A SINGLE Scatter with None separators (scales with the 

3788 # true-scale embed), drawn before the fixation markers so the dots sit on top. 

3789 if ( 

3790 spatial_axes 

3791 and show_connectors 

3792 and connector_y is not None 

3793 and not fixations.empty 

3794 ): 

3795 xs = pd.to_numeric(fixations[x_field], errors="coerce").to_numpy(dtype=float) 

3796 y_corr = pd.to_numeric(fixations[y_field], errors="coerce").to_numpy( 

3797 dtype=float 

3798 ) 

3799 y_orig = np.asarray(connector_y, dtype=float) 

3800 seg_x: list = [] 

3801 seg_y: list = [] 

3802 for xi, yo, yc in zip(xs, y_orig, y_corr): 

3803 if not (np.isfinite(xi) and np.isfinite(yo) and np.isfinite(yc)): 

3804 continue 

3805 seg_x += [xi, xi, None] 

3806 seg_y += [yo, yc, None] 

3807 if seg_x: 

3808 fig.add_trace( 

3809 go.Scatter( 

3810 x=seg_x, 

3811 y=seg_y, 

3812 mode="lines", 

3813 line=dict(color="rgba(110,110,110,0.9)", width=1), 

3814 opacity=0.3, 

3815 hoverinfo="skip", 

3816 showlegend=False, 

3817 name="drift", 

3818 ) 

3819 ) 

3820 

3821 if show_fixations and not fixations.empty: 

3822 # ``render_fix`` == fixations unless VIZ-9 snap-to-word is on, in which 

3823 # case the markers, order labels and colour-by-line use the snapped x/y. 

3824 ordered = render_fix.sort_values("timestamp_ms") 

3825 # Fixation classification (PRE-2, viz-only): SHORT / LONG / OUT-OF-BOUNDS, 

3826 # each Off / Highlight / Discard. Apply Discard here — drop those rows from 

3827 # `ordered` so they vanish from the markers, fixation-index labels and 

3828 # marker-size scaling (the saccade layer above still bridges across them). 

3829 # This changes only what's DRAWN; reading measures and exports are untouched. 

3830 flags = fixation_flags or {} 

3831 if flags: 

3832 ordered = _discard_flagged_fixations( 

3833 ordered, words, flags, spatial_axes=spatial_axes 

3834 ) 

3835 # "Color by line" overrides the chosen color field: each fixation is 

3836 # tinted by the text line it lands on (lines inferred from word 

3837 # geometry). Rendered as discrete categories so the legend reads 

3838 # "line: Line 1", "line: Line 2", … 

3839 if color_by_line and spatial_axes and not words.empty: 

3840 from .measures import assign_fixation_lines 

3841 

3842 line_ids = assign_fixation_lines(ordered, words) 

3843 color_data = line_ids.map( 

3844 lambda v: f"Line {int(v) + 1}" if pd.notna(v) else "Out of bounds" 

3845 ) 

3846 color_label = "line" 

3847 is_numeric_color = False 

3848 elif color_by == UNIFORM_COLOR_FIELD: 

3849 # VIZ-17: no variable mapped to hue — size already encodes duration. 

3850 color_data = None 

3851 color_label = color_by 

3852 is_numeric_color = False 

3853 else: 

3854 color_data = ordered[color_by] if color_by in ordered.columns else None 

3855 color_label = color_by 

3856 is_numeric_color = color_data is not None and pd.api.types.is_numeric_dtype( 

3857 color_data 

3858 ) 

3859 marker_color, category_legend = _resolve_marker_colors( 

3860 color_data, is_numeric_color, fixation_color 

3861 ) 

3862 sizes = _compute_marker_sizes( 

3863 ordered["duration_ms"], marker_size_range, **_settings_size_scale(settings) 

3864 ) 

3865 marker = dict( 

3866 size=sizes, 

3867 symbol=fixation_symbol or DEFAULT_FIXATION_SYMBOL, 

3868 color=marker_color, 

3869 colorscale=fixation_colorscale if is_numeric_color else None, 

3870 showscale=show_colorbars and is_numeric_color, 

3871 colorbar=_colorbar_dict( 

3872 _column_title(color_label), 

3873 orientation=colorbar_orientation, 

3874 tickangle=colorbar_tickangle, 

3875 tickfont_size=colorbar_tickfont_size, 

3876 ) 

3877 if show_colorbars and is_numeric_color 

3878 else None, 

3879 cmin=fixation_color_range[0] if fixation_color_range else None, 

3880 cmax=fixation_color_range[1] if fixation_color_range else None, 

3881 line=dict(color=FIX_MARKER_OUTLINE, width=0.5), 

3882 ) 

3883 # Marker alpha (VIZ-6): lower it so overlapping fixations show through. 

3884 # Always set it (even at 1.0) so the slider is authoritative — Plotly's 

3885 # variable-size scatter markers otherwise render at a ~0.7 default, so an 

3886 # unset 1.0 looked translucent ("opacity at 1 wasn't really 1"). 

3887 marker["opacity"] = float( 

3888 fixation_opacity if fixation_opacity is not None else 1.0 

3889 ) 

3890 if hollow_fixations: 

3891 marker = _make_hollow(marker) 

3892 hover_fields = ( 

3893 ["order_in_trial", "duration_ms", "word_id"] 

3894 if fixation_hover_fields is None 

3895 else list(fixation_hover_fields) 

3896 ) 

3897 customdata, hovertemplate = _hover_payload( 

3898 ordered, hover_fields, fixation=True, words=words 

3899 ) 

3900 glyph = FIXATION_GLYPH_SYMBOLS.get(fixation_symbol or "") 

3901 if glyph: 

3902 # VIZ-15: a shape Plotly's marker enum doesn't carry (♥), drawn as 

3903 # text from the same marker dict (`_glyph_scatter_traces`); the 

3904 # fixation-index labels move to their own trace since one Scatter has 

3905 # only one text field, and a numeric colour bar to its own. 

3906 for trace in _glyph_scatter_traces( 

3907 ordered[x_field], 

3908 ordered[y_field], 

3909 marker, 

3910 glyph, 

3911 hovertemplate=hovertemplate, 

3912 customdata=customdata, 

3913 name="Fixations", 

3914 showlegend=False, 

3915 ): 

3916 fig.add_trace(trace) 

3917 bar = _glyph_colorbar_trace(marker, color_data if is_numeric_color else ()) 

3918 if bar is not None: 

3919 fig.add_trace(bar) 

3920 if show_order: 

3921 fig.add_trace( 

3922 go.Scatter( 

3923 x=ordered[x_field], 

3924 y=ordered[y_field], 

3925 mode="text", 

3926 text=ordered["order_in_trial"], 

3927 textfont=dict( 

3928 color=order_font_color, 

3929 size=order_font_size, 

3930 family=font_settings["family"], 

3931 ), 

3932 textposition="top center", 

3933 hoverinfo="skip", 

3934 name="Fixation index", 

3935 showlegend=False, 

3936 ) 

3937 ) 

3938 else: 

3939 fig.add_trace( 

3940 go.Scatter( 

3941 x=ordered[x_field], 

3942 y=ordered[y_field], 

3943 mode="markers+text" if show_order else "markers", 

3944 marker=marker, 

3945 text=ordered["order_in_trial"] if show_order else None, 

3946 textfont=dict( 

3947 color=order_font_color, 

3948 size=order_font_size, 

3949 family=font_settings["family"], 

3950 ), 

3951 textposition="top center", 

3952 hovertemplate=hovertemplate, 

3953 customdata=customdata, 

3954 name="Fixations", 

3955 showlegend=False, 

3956 ) 

3957 ) 

3958 legend_limit = len(_QUALITATIVE_PALETTE) 

3959 truncated_legend = category_legend[:legend_limit] 

3960 if category_legend: 

3961 legend_active = True 

3962 for category, color in truncated_legend: 

3963 fig.add_trace( 

3964 go.Scatter( 

3965 x=[None], 

3966 y=[None], 

3967 mode="markers", 

3968 marker=dict( 

3969 size=10, 

3970 color=color, 

3971 line=dict(color=FIX_MARKER_OUTLINE, width=0.5), 

3972 ), 

3973 name=category 

3974 if color_label == "line" 

3975 else f"{_column_name(color_label)}: {category}", 

3976 showlegend=True, 

3977 hoverinfo="skip", 

3978 ) 

3979 ) 

3980 if len(category_legend) > legend_limit: 

3981 fig.add_trace( 

3982 go.Scatter( 

3983 x=[None], 

3984 y=[None], 

3985 mode="markers", 

3986 marker=dict(size=10, color="#cccccc"), 

3987 name=f"… +{len(category_legend) - legend_limit} more", 

3988 showlegend=True, 

3989 hoverinfo="skip", 

3990 ) 

3991 ) 

3992 

3993 # Highlight overlays (PRE-2): mark SHORT / LONG / OUT-OF-BOUNDS fixations 

3994 # in the chosen marker + colour, on top of the regular markers. Masks are 

3995 # recomputed on the (post-discard) `ordered`; out-of-bounds needs word 

3996 # boxes + spatial axes, short/long are duration-based and apply anywhere. 

3997 if flags: 

3998 _overlay = _fixation_flag_masks( 

3999 ordered, words, flags, spatial_axes=spatial_axes 

4000 ) 

4001 for _cat in _FIX_FLAG_CATEGORIES: 

4002 _spec = flags.get(_cat, {}) 

4003 if _spec.get("mode") != "Highlight": 

4004 continue 

4005 _name = _FIX_FLAG_LABELS[_cat] 

4006 hits = ordered[_overlay[_cat]] 

4007 if hits.empty: 

4008 continue 

4009 legend_active = True 

4010 fig.add_trace( 

4011 go.Scatter( 

4012 x=hits[x_field], 

4013 y=hits[y_field], 

4014 mode="markers", 

4015 marker=dict( 

4016 symbol=_spec.get("symbol") or "x", 

4017 size=13, 

4018 color=_spec.get("color") or OUT_OF_TEXT_COLOR, 

4019 line=dict(color="#ffffff", width=1), 

4020 ), 

4021 name=_name, 

4022 showlegend=True, 

4023 hovertemplate=( 

4024 f"{_name} fixation<br>x %{{x:.0f}}, y %{{y:.0f}}" 

4025 "<extra></extra>" 

4026 ), 

4027 ) 

4028 ) 

4029 

4030 xaxis_cfg = dict(showticklabels=False, showgrid=False, zeroline=False, title=None) 

4031 yaxis_cfg = dict(showticklabels=False, showgrid=False, zeroline=False, title=None) 

4032 if spatial_axes: 

4033 # automargin off: the colorbar/legend live in the reserved margin we size 

4034 # below (_decoration_margins), so Plotly must not also shrink the 

4035 # equal-aspect plot domain to fit them. 

4036 xaxis_cfg.update(range=x_range, constrain="domain", automargin=False) 

4037 yaxis_cfg.update( 

4038 range=y_range, 

4039 constrain="domain", 

4040 scaleanchor="x", 

4041 scaleratio=1, 

4042 automargin=False, 

4043 ) 

4044 _apply_coordinate_grid_axes( 

4045 xaxis_cfg, 

4046 yaxis_cfg, 

4047 show=show_coordinate_grid, 

4048 spacing=coordinate_grid_spacing, 

4049 x_range=x_range, 

4050 y_range=y_range, 

4051 rendered_width=fitted_w, 

4052 rendered_height=fitted_h, 

4053 ) 

4054 else: 

4055 xaxis_cfg.update( 

4056 showticklabels=True, showgrid=True, title=_column_title(x_field) 

4057 ) 

4058 yaxis_cfg.update( 

4059 showticklabels=True, showgrid=True, title=_column_title(y_field) 

4060 ) 

4061 

4062 shapes = list(fig.layout.shapes) if fig.layout.shapes else [] 

4063 if spatial_axes: 

4064 shapes.append( 

4065 dict( 

4066 type="rect", 

4067 x0=x_range[0], 

4068 y0=y_range[1], 

4069 x1=x_range[1], 

4070 y1=y_range[0], 

4071 line=dict(color="#000000", width=1), 

4072 fillcolor="rgba(0,0,0,0)", 

4073 # VIZ-5: the plot border is its own layer (a registration guide). 

4074 name=_shape_layer_tag("frame"), 

4075 ) 

4076 ) 

4077 

4078 # fitted_w / fitted_h were computed up front (so the label scale matched). A 

4079 # colorbar (numeric colour / heatmap) or legend (discrete colour categories, 

4080 # out-of-text, raw gaze) is given reserved margin so it never shrinks the 

4081 # equal-aspect plot region — keeping the word labels matched to the boxes. 

4082 decoration = ( 

4083 _decoration_margins( 

4084 fitted_w, 

4085 fitted_h, 

4086 **_colorbar_reserves( 

4087 (show_colorbars and is_numeric_color, colorbar_orientation), 

4088 (show_heatmap_colorbar and heatmap_rendered, cb_style["orientation"]), 

4089 ), 

4090 legend=legend_active, 

4091 coordinate_grid=show_coordinate_grid, 

4092 ) 

4093 if spatial_axes 

4094 else {"width": fitted_w, "height": fitted_h, "margin": dict(l=0, r=0, t=0, b=0)} 

4095 ) 

4096 fig.update_layout( 

4097 height=decoration["height"], 

4098 width=decoration["width"], 

4099 autosize=False, 

4100 margin=decoration["margin"], 

4101 xaxis=xaxis_cfg, 

4102 yaxis=yaxis_cfg, 

4103 legend=dict(orientation="h", yanchor="bottom", y=1.02, xanchor="right", x=1), 

4104 template="plotly_white", 

4105 # None leaves the template's default white; a hex value paints both the 

4106 # plotting area and the surrounding paper (e.g. a neutral gray). 

4107 plot_bgcolor=background_color, 

4108 paper_bgcolor=background_color, 

4109 font=font_settings, 

4110 shapes=shapes, 

4111 ) 

4112 add_illustration_label(fig, illustration_reasons, text=settings.illustration_text) 

4113 return fig 

4114 

4115 

4116def add_illustration_label( 

4117 fig: go.Figure, reasons: Sequence[str] | None, *, text: str = "" 

4118) -> go.Figure: 

4119 """Stamp a figure and its metadata when it is schematic or transformed. 

4120 

4121 ``text`` replaces the drawn wording; empty draws "Illustration · <reasons>". 

4122 The reasons are recorded in the metadata either way.""" 

4123 reasons = [str(reason) for reason in (reasons or []) if reason] 

4124 if not reasons: 

4125 return fig 

4126 fig.add_annotation( 

4127 x=1, 

4128 y=0, 

4129 xref="paper", 

4130 yref="paper", 

4131 xanchor="right", 

4132 yanchor="bottom", 

4133 text=_plotly_literal(str(text).strip()) 

4134 # #374: a label the user switched on with nothing detected says 

4135 # "Illustration" alone; "· manual label" told a reader nothing. 

4136 or ( 

4137 "Illustration" 

4138 if reasons == [MANUAL_LABEL_REASON] 

4139 else "Illustration · " + "; ".join(reasons) 

4140 ), 

4141 showarrow=False, 

4142 font=dict(size=10, color="#5f6368"), 

4143 bgcolor="rgba(255,255,255,0.82)", 

4144 borderpad=3, 

4145 name=_ILLUSTRATION_LABEL_NAME, 

4146 ) 

4147 _stack_bottom_right(fig) 

4148 metadata = dict(fig.layout.meta or {}) 

4149 metadata["illustration"] = True 

4150 metadata["illustration_reasons"] = reasons 

4151 fig.update_layout(meta=metadata) 

4152 return fig 

4153 

4154 

4155# VIZ-3: alternative heatmap normalization. The colour of a heatmap cell maps 

4156# LINEARLY between its z-range endpoints, so a few very-hot words (dwell times are 

4157# heavy-tailed) can wash out the rest. "Log" instead maps colour to log1p(value), 

4158# compressing the top of the range so mid-range words stay distinguishable. The 

4159# transform is applied to the *values and the range endpoints together*, so the 

4160# raw-unit `heatmap_range` slider keeps its meaning; only the colour curve changes. 

4161_HEATMAP_NORMS = ("Linear", "Log") 

4162 

4163 

4164def _apply_heatmap_norm(values, norm: str): 

4165 """Transform heatmap values for the chosen normalization (VIZ-3). 

4166 

4167 ``Log`` returns ``log1p(max(value, 0))`` (heavy-tail compression); anything 

4168 else is the identity. Accepts a scalar or an array; returns the same shape.""" 

4169 arr = np.asarray(values, dtype=float) 

4170 if norm == "Log": 

4171 return np.log1p(np.clip(arr, 0.0, None)) 

4172 return arr 

4173 

4174 

4175#: The duration-weighted word-box heatmap's colour-bar title: a box maps the 

4176#: *summed* duration of its fixations, which a fixation-duration colour bar 

4177#: beside it must not be mistaken for (round-7 review, findings 11 and 16). 

4178_WORD_DWELL_TITLE = "Dwell time per word (ms)" 

4179 

4180 

4181def _heatmap_title(base: str, norm: str) -> str: 

4182 """Colour-bar title, marked ``(log)`` when the log normalization is active.""" 

4183 return f"{base} (log)" if norm == "Log" else base 

4184 

4185 

4186def _add_word_level_heatmap( 

4187 fig: go.Figure, 

4188 words: pd.DataFrame, 

4189 fixations: pd.DataFrame, 

4190 *, 

4191 x_field: str, 

4192 y_field: str, 

4193 weights: pd.Series | None, 

4194 heatmap_colorscale: str, 

4195 heatmap_range: tuple[float, float] | None, 

4196 show_colorbars: bool, 

4197 heatmap_norm: str = "Linear", 

4198 colorbar_style: dict | None = None, 

4199) -> None: 

4200 # Pull the fixation coordinates (and optional weights) into numpy arrays once, 

4201 # then test box membership per word against the arrays. Same O(words × fix) 

4202 # work as before but without rebuilding pandas Series each iteration, and with 

4203 # O(fix) memory (no full words × fix matrix). 

4204 fx = pd.to_numeric(fixations[x_field], errors="coerce").to_numpy(dtype=float) 

4205 fy = pd.to_numeric(fixations[y_field], errors="coerce").to_numpy(dtype=float) 

4206 w_arr = ( 

4207 pd.to_numeric(weights, errors="coerce").to_numpy(dtype=float) 

4208 if weights is not None 

4209 else None 

4210 ) 

4211 # Bin against the experiment's boxes (BUG-83) with the assignment's own 

4212 # containment rule — the same boundary it uses and the heatmap then draws, 

4213 # and half-open, so a fixation on a shared edge counts towards one word. 

4214 from .measures import word_box_bounds, word_box_contains 

4215 

4216 word_values = [] 

4217 for wx0, wy0, wx1, wy1 in zip(*word_box_bounds(words)): 

4218 in_word = word_box_contains(fx, fy, wx0, wy0, wx1, wy1) 

4219 val = ( 

4220 float(np.nansum(w_arr[in_word])) 

4221 if w_arr is not None 

4222 else float(in_word.sum()) 

4223 ) 

4224 word_values.append(val) 

4225 

4226 _draw_word_value_heatmap( 

4227 fig, 

4228 words, 

4229 word_values, 

4230 heatmap_colorscale=heatmap_colorscale, 

4231 heatmap_range=heatmap_range, 

4232 show_colorbars=show_colorbars, 

4233 heatmap_norm=heatmap_norm, 

4234 colorbar_title="Fixation count" if weights is None else _WORD_DWELL_TITLE, 

4235 colorbar_style=colorbar_style, 

4236 ) 

4237 

4238 

4239def _add_word_measure_heatmap( 

4240 fig: go.Figure, 

4241 words: pd.DataFrame, 

4242 measure: str, 

4243 *, 

4244 heatmap_colorscale: str, 

4245 heatmap_range: tuple[float, float] | None, 

4246 show_colorbars: bool, 

4247 heatmap_norm: str = "Linear", 

4248 colorbar_style: dict | None = None, 

4249) -> None: 

4250 """Word-box heatmap from a pre-aggregated per-word measure column. 

4251 

4252 Used for words-only datasets (IA report without a fixation report): the 

4253 usual heatmap aggregates fixation durations/counts into the boxes, but 

4254 with no fixations the dataset's own reading measures (e.g. total fixation 

4255 duration) carry the same information.""" 

4256 values = pd.to_numeric(words[measure], errors="coerce").fillna(0.0) 

4257 _draw_word_value_heatmap( 

4258 fig, 

4259 words, 

4260 [float(v) for v in values], 

4261 heatmap_colorscale=heatmap_colorscale, 

4262 heatmap_range=heatmap_range, 

4263 show_colorbars=show_colorbars, 

4264 heatmap_norm=heatmap_norm, 

4265 colorbar_title="Fixation count" 

4266 if measure == "n_fixations" 

4267 else _WORD_DWELL_TITLE, 

4268 colorbar_style=colorbar_style, 

4269 ) 

4270 

4271 

4272def _draw_word_value_heatmap( 

4273 fig: go.Figure, 

4274 words: pd.DataFrame, 

4275 word_values: list, 

4276 *, 

4277 heatmap_colorscale: str, 

4278 heatmap_range: tuple[float, float] | None, 

4279 show_colorbars: bool, 

4280 heatmap_norm: str = "Linear", 

4281 colorbar_title: str, 

4282 colorbar_style: dict | None = None, 

4283) -> None: 

4284 from plotly.colors import sample_colorscale 

4285 

4286 from .measures import word_box_bounds 

4287 

4288 # Nonzero test on the RAW values (a word with no dwell stays uncoloured); the 

4289 # colour position then maps through the chosen normalization (VIZ-3). Boxes 

4290 # come from word_box_bounds so the tinted rects sit exactly on the outlines 

4291 # build_word_boxes draws. 

4292 boxes = zip(*word_box_bounds(words)) 

4293 nonzero_rows = [(box, v) for box, v in zip(boxes, word_values) if v > 0] 

4294 if not nonzero_rows: 

4295 return 

4296 vals = [v for _, v in nonzero_rows] 

4297 # Auto starts at 0: an empty word is the bottom of the scale. 

4298 z_min_raw = heatmap_range[0] if heatmap_range else 0.0 

4299 z_max_raw = heatmap_range[1] if heatmap_range else float(max(vals)) 

4300 z_min = float(_apply_heatmap_norm(z_min_raw, heatmap_norm)) 

4301 z_max = float(_apply_heatmap_norm(z_max_raw, heatmap_norm)) 

4302 z_span = max(z_max - z_min, 1e-9) 

4303 

4304 heatmap_shapes = [] 

4305 for (x0, y0, x1, y1), v in nonzero_rows: 

4306 tv = float(_apply_heatmap_norm(v, heatmap_norm)) 

4307 norm = max(0.0, min(1.0, (tv - z_min) / z_span)) 

4308 color = sample_colorscale(heatmap_colorscale, [norm])[0] 

4309 heatmap_shapes.append( 

4310 dict( 

4311 type="rect", 

4312 x0=x0, 

4313 y0=y0, 

4314 x1=x1, 

4315 y1=y1, 

4316 line=dict(width=0), 

4317 fillcolor=color, 

4318 opacity=0.5, 

4319 layer="below", 

4320 # VIZ-5: word-box heatmap rects belong to the heatmap layer. 

4321 name=_shape_layer_tag("heatmap"), 

4322 ) 

4323 ) 

4324 existing = list(fig.layout.shapes) if fig.layout.shapes else [] 

4325 fig.update_layout(shapes=existing + heatmap_shapes) 

4326 if show_colorbars: 

4327 fig.add_trace( 

4328 go.Scatter( 

4329 x=[None], 

4330 y=[None], 

4331 mode="markers", 

4332 marker=dict( 

4333 colorscale=heatmap_colorscale, 

4334 showscale=True, 

4335 cmin=z_min, 

4336 cmax=z_max, 

4337 colorbar=_colorbar_dict( 

4338 _heatmap_title(colorbar_title, heatmap_norm), 

4339 **(colorbar_style or {}), 

4340 ), 

4341 ), 

4342 showlegend=False, 

4343 hoverinfo="skip", 

4344 # VIZ-5: the word-box heatmap's colorbar-carrier rides the heatmap 

4345 # layer (name contains "heatmap" → classified there). 

4346 name="heatmap colorbar", 

4347 ) 

4348 ) 

4349 

4350 

4351def _add_density_heatmap( 

4352 fig: go.Figure, 

4353 fixations: pd.DataFrame, 

4354 *, 

4355 x_field: str, 

4356 y_field: str, 

4357 x_min: float, 

4358 x_max: float, 

4359 y_min: float, 

4360 y_max: float, 

4361 weights: pd.Series | None, 

4362 heatmap_colorscale: str, 

4363 heatmap_range: tuple[float, float] | None, 

4364 show_colorbars: bool, 

4365 heatmap_norm: str = "Linear", 

4366 colorbar_style: dict | None = None, 

4367) -> None: 

4368 # A 40×40 count/duration grid drawn as a go.Heatmap (rather than 

4369 # go.Histogram2d) so the colour mapping can go through _apply_heatmap_norm 

4370 # (VIZ-3) — Plotly's Histogram2d bins internally and can't be log-scaled. 

4371 xs = pd.to_numeric(fixations[x_field], errors="coerce") 

4372 ys = pd.to_numeric(fixations[y_field], errors="coerce") 

4373 valid = xs.notna() & ys.notna() 

4374 if not valid.any(): 

4375 return 

4376 if weights is not None: 

4377 w = ( 

4378 pd.to_numeric(weights, errors="coerce") 

4379 .reindex(fixations.index) 

4380 .fillna(0.0)[valid] 

4381 .to_numpy() 

4382 ) 

4383 else: 

4384 w = np.ones(int(valid.sum())) 

4385 xv = xs[valid].to_numpy() 

4386 yv = ys[valid].to_numpy() 

4387 

4388 x_edges = np.linspace(x_min, x_max, 41) 

4389 y_edges = np.linspace(y_min, y_max, 41) 

4390 hist, _, _ = np.histogram2d(xv, yv, bins=[x_edges, y_edges], weights=w) 

4391 grid = hist.T # rows index y, cols index x — the orientation go.Heatmap wants 

4392 if grid.max() <= 0: 

4393 return 

4394 z = np.where(grid > 0, _apply_heatmap_norm(grid, heatmap_norm), np.nan) 

4395 z_range = ( 

4396 _apply_heatmap_norm(np.asarray(heatmap_range, dtype=float), heatmap_norm) 

4397 if heatmap_range 

4398 else (None, None) 

4399 ) 

4400 base_title = "Fixation density" if weights is None else "Dwell time per cell (ms)" 

4401 fig.add_trace( 

4402 go.Heatmap( 

4403 x=(x_edges[:-1] + x_edges[1:]) / 2.0, 

4404 y=(y_edges[:-1] + y_edges[1:]) / 2.0, 

4405 z=z, 

4406 colorscale=heatmap_colorscale, 

4407 opacity=0.35, 

4408 showscale=show_colorbars, 

4409 colorbar=_colorbar_dict( 

4410 _heatmap_title(base_title, heatmap_norm), **(colorbar_style or {}) 

4411 ), 

4412 zmin=z_range[0], 

4413 zmax=z_range[1], 

4414 hoverinfo="skip", 

4415 name="Fixation heatmap", 

4416 ) 

4417 ) 

4418 

4419 

4420# Past this radius the 3-sigma kernel's normalising sum is taken from the 

4421# Gaussian integral (exact to float precision at that sigma) instead of an array. 

4422_KERNEL_EXACT_SUM_RADIUS = 100_000 

4423 

4424 

4425def _gaussian_kernel_1d(sigma: float, max_radius: int | None = None) -> np.ndarray: 

4426 """Normalized 1-D Gaussian kernel, truncated at 3 sigma. 

4427 

4428 ``max_radius`` builds only the central taps, still normalized over the full 

4429 3-sigma kernel, so a short axis pays for the taps it can reach rather than 

4430 for sigma. 

4431 """ 

4432 radius = max(1, round(sigma * 3)) 

4433 keep = radius if max_radius is None else max(0, min(radius, max_radius)) 

4434 offsets = np.arange(-keep, keep + 1) 

4435 kernel = np.exp(-(offsets**2) / (2.0 * sigma * sigma)) 

4436 if keep == radius: 

4437 total = kernel.sum() 

4438 elif radius <= _KERNEL_EXACT_SUM_RADIUS: 

4439 full = np.arange(-radius, radius + 1) 

4440 total = np.exp(-(full**2) / (2.0 * sigma * sigma)).sum() 

4441 else: 

4442 total = ( 

4443 sigma 

4444 * math.sqrt(2.0 * math.pi) 

4445 * math.erf((radius + 0.5) / (sigma * math.sqrt(2.0))) 

4446 ) 

4447 return kernel / total 

4448 

4449 

4450def _blur_axis(grid: np.ndarray, sigma: float, axis: int) -> np.ndarray: 

4451 """Zero-padded Gaussian blur along one axis; the output keeps ``grid``'s shape. 

4452 

4453 ``np.convolve(mode="same")`` returns the *kernel's* length when the kernel 

4454 is longer than the axis, which shifted the density off its coordinates on a 

4455 short grid. Padding by the kernel radius and taking the ``valid`` part pins 

4456 the length. Taps further out than the axis is long only ever meet the zero 

4457 padding, so they are never built: the result is identical and the cost is 

4458 bounded by the axis length, not by sigma. 

4459 """ 

4460 kernel = _gaussian_kernel_1d(sigma, max_radius=grid.shape[axis] - 1) 

4461 radius = len(kernel) // 2 

4462 pad = [(0, 0)] * grid.ndim 

4463 pad[axis] = (radius, radius) 

4464 padded = np.pad(grid, pad) 

4465 return np.apply_along_axis( 

4466 lambda v: np.convolve(v, kernel, mode="valid"), axis, padded 

4467 ) 

4468 

4469 

4470def _gaussian_blur_2d( 

4471 grid: np.ndarray, sigma_rows: float, sigma_cols: float 

4472) -> np.ndarray: 

4473 """Separable Gaussian blur (a numpy-only stand-in for scipy.ndimage). 

4474 

4475 Zero padding beyond the grid; the output always has the input's shape. 

4476 """ 

4477 out = grid.astype(float) 

4478 if out.size == 0: 

4479 return out 

4480 if sigma_rows and sigma_rows > 0: 

4481 out = _blur_axis(out, sigma_rows, 0) 

4482 if sigma_cols and sigma_cols > 0: 

4483 out = _blur_axis(out, sigma_cols, 1) 

4484 return out 

4485 

4486 

4487# Interpolated-heatmap tuning. The Gaussian sigma defaults to a fraction of the 

4488# larger data span — enough to merge a fixation cluster into one smooth blob 

4489# without bleeding across neighbouring text lines. 

4490_INTERP_GRID = 240 # cells along the wider axis 

4491_INTERP_SIGMA_FRAC = 0.02 # sigma as a fraction of the larger data span 

4492_INTERP_MIN_SIGMA_PX = 8.0 

4493 

4494 

4495def interpolated_sigma_px(x_span: float, y_span: float) -> float: 

4496 """The Interpolated heatmap's automatic Gaussian σ, in px: 2% of the data's 

4497 larger span (fixations, word boxes and shown raw gaze), at least 8 px.""" 

4498 return max(_INTERP_MIN_SIGMA_PX, _INTERP_SIGMA_FRAC * max(x_span, y_span)) 

4499 

4500 

4501_INTERP_OPACITY = 0.45 

4502_INTERP_FLOOR_FRAC = 0.02 # cells below this fraction of the peak render transparent 

4503_INTERP_MIN_CELLS = 10 # cells along the narrower axis, at least 

4504_INTERP_MIN_SPAN_PX = 2.0 * _INTERP_MIN_SIGMA_PX # narrower than this is widened 

4505 

4506 

4507def _interp_grid_shape(x_span: float, y_span: float) -> tuple[int, int]: 

4508 """(nx, ny) cells: _INTERP_GRID on the wider axis, a proportional share on 

4509 the other, each within [_INTERP_MIN_CELLS, _INTERP_GRID].""" 

4510 wide = max(x_span, y_span) 

4511 narrow = min(x_span, y_span) 

4512 share = round(_INTERP_GRID * narrow / wide) if wide > 0 else _INTERP_GRID 

4513 n_narrow = int(min(_INTERP_GRID, max(_INTERP_MIN_CELLS, share))) 

4514 if x_span >= y_span: 

4515 return _INTERP_GRID, n_narrow 

4516 return n_narrow, _INTERP_GRID 

4517 

4518 

4519def _add_interpolated_heatmap( 

4520 fig: go.Figure, 

4521 fixations: pd.DataFrame, 

4522 *, 

4523 x_field: str, 

4524 y_field: str, 

4525 x_min: float, 

4526 x_max: float, 

4527 y_min: float, 

4528 y_max: float, 

4529 weights: pd.Series | None, 

4530 heatmap_colorscale: str, 

4531 show_colorbars: bool, 

4532 heatmap_norm: str = "Linear", 

4533 colorbar_style: dict | None = None, 

4534 sigma_px: float | None = None, 

4535 title: str | None = None, 

4536) -> None: 

4537 """Smooth, word-box-independent fixation heatmap (Gaussian-interpolated). 

4538 

4539 Bins the fixations onto a fine grid (weighted by duration when ``weights`` 

4540 is given), then blurs with a Gaussian — the classic eye-movement heatmap 

4541 (cf. PyGaze's gaze plotter). Empty cells render transparent so the reading 

4542 text stays legible underneath. 

4543 """ 

4544 xs = pd.to_numeric(fixations[x_field], errors="coerce") 

4545 ys = pd.to_numeric(fixations[y_field], errors="coerce") 

4546 valid = xs.notna() & ys.notna() 

4547 if not valid.any(): 

4548 return 

4549 if weights is not None: 

4550 w = ( 

4551 pd.to_numeric(weights, errors="coerce") 

4552 .reindex(fixations.index) 

4553 .fillna(0.0)[valid] 

4554 .to_numpy() 

4555 ) 

4556 else: 

4557 w = np.ones(int(valid.sum())) 

4558 xs = xs[valid].to_numpy() 

4559 ys = ys[valid].to_numpy() 

4560 

4561 x_span = max(x_max - x_min, 1.0) 

4562 y_span = max(y_max - y_min, 1.0) 

4563 sigma_px = float(sigma_px or interpolated_sigma_px(x_span, y_span)) 

4564 # An axis with (next to) no extent — coincident fixations, one row or one 

4565 # column of a fixation-only import — is widened around its centre to the 

4566 # blob's own size (±3 sigma), so the edges match the span the cells and the 

4567 # sigma are computed from, and the blob is round rather than a sliver. 

4568 min_span = max(_INTERP_MIN_SPAN_PX, 6.0 * sigma_px) 

4569 if x_max - x_min < _INTERP_MIN_SPAN_PX: 

4570 x_mid = (x_min + x_max) / 2.0 

4571 x_min, x_max, x_span = x_mid - min_span / 2, x_mid + min_span / 2, min_span 

4572 if y_max - y_min < _INTERP_MIN_SPAN_PX: 

4573 y_mid = (y_min + y_max) / 2.0 

4574 y_min, y_max, y_span = y_mid - min_span / 2, y_mid + min_span / 2, min_span 

4575 # The cell budget sits on the wider axis and the other gets its share of it, 

4576 # so neither axis, nor the grid, outgrows _INTERP_GRID (squared): a tall, 

4577 # narrow reading must not ask for 240 cells across and 96,000 down. 

4578 nx, ny = _interp_grid_shape(x_span, y_span) 

4579 x_edges = np.linspace(x_min, x_max, nx + 1) 

4580 y_edges = np.linspace(y_min, y_max, ny + 1) 

4581 # histogram2d returns shape (nx, ny); transpose so rows index y, cols index x 

4582 # (the orientation go.Heatmap's z expects). 

4583 hist, _, _ = np.histogram2d(xs, ys, bins=[x_edges, y_edges], weights=w) 

4584 grid = hist.T 

4585 

4586 blurred = _gaussian_blur_2d( 

4587 grid, sigma_rows=sigma_px / (y_span / ny), sigma_cols=sigma_px / (x_span / nx) 

4588 ) 

4589 peak = float(blurred.max()) 

4590 if peak <= 0: 

4591 return 

4592 # Near-zero cells -> NaN so Plotly renders them transparent (only populated 

4593 # regions get tinted, keeping the text readable). The remaining density maps 

4594 # through the chosen normalization (VIZ-3; log1p(0)=0 keeps the floor at 0). 

4595 z = np.where(blurred < peak * _INTERP_FLOOR_FRAC, np.nan, blurred) 

4596 z = _apply_heatmap_norm(z, heatmap_norm) 

4597 

4598 base_title = title or ( 

4599 "Dwell-time density" if weights is not None else "Fixation density" 

4600 ) 

4601 fig.add_trace( 

4602 go.Heatmap( 

4603 x=(x_edges[:-1] + x_edges[1:]) / 2.0, 

4604 y=(y_edges[:-1] + y_edges[1:]) / 2.0, 

4605 z=z, 

4606 colorscale=heatmap_colorscale, 

4607 opacity=_INTERP_OPACITY, 

4608 showscale=show_colorbars, 

4609 # z is a Gaussian-smoothed density in arbitrary (weighted) units, not 

4610 # the per-word counts/ms the `heatmap_range` slider is calibrated for, 

4611 # so it autoscales from 0 rather than borrowing that range. 

4612 zmin=0.0, 

4613 colorbar=_colorbar_dict( 

4614 _heatmap_title(base_title, heatmap_norm), **(colorbar_style or {}) 

4615 ), 

4616 hoverinfo="skip", 

4617 name="Fixation heatmap", 

4618 ) 

4619 ) 

4620 

4621 

4622# ============================================================================= 

4623# Scanpath animation — one or two scanpaths on a shared real reading-time clock 

4624# ============================================================================= 

4625 

4626# Floor on the ▶ Play button's own per-frame duration: ~one 60 fps display frame. 

4627# That duration only drives Plotly's frame queue, which is what a figure plays on 

4628# where the wall-clock player isn't embedded (`fig.show()`); every HTML surface 

4629# replays on `animation_player_post_script` instead (BUG-93). 

4630_ANIM_MIN_FRAME_MS = 16 

4631# VIZ-11: animation frames sit on a UNIFORM time grid (one every 

4632# _ANIM_GRID_STEP_MS of reading) rather than one per fixation onset, so the 

4633# slider scrubs linearly through seconds regardless of how fixations cluster or 

4634# how many scanpaths overlay. The grid coarsens past _ANIM_MAX_FRAMES so a long 

4635# reading doesn't emit thousands of frames (which would balloon the GIF/MP4 

4636# export); quantization is then at most one grid step. 

4637_ANIM_GRID_STEP_MS = 100.0 

4638_ANIM_MAX_FRAMES = 360 

4639 

4640# Vertical space (px) reserved BELOW the animation plot for the transport 

4641# controls (play / pause / restart buttons + the time slider with its "Elapsed" 

4642# readout). The figure is grown by this much (plus a small safety buffer) and 

4643# the controls are placed in the bottom margin, so Plotly's automargin never has 

4644# to shrink the equal-aspect plot to fit them — which would make the word boxes 

4645# smaller than the true-to-scale label font computed for fitted_h (the 

4646# text-too-large bug). Keeps the animation plot the SAME size as the static one. 

4647_CONTROLS_MARGIN_PX = 116 

4648_CONTROLS_SAFETY_PX = 24 

4649 

4650 

4651def _scanpath_anim_specs( 

4652 entries, 

4653 marker_size_range, 

4654 scale: str = DEFAULT_MARKER_SIZE_SCALE, 

4655 duration_range=DEFAULT_MARKER_DURATION_RANGE, 

4656 *, 

4657 size_ranges=None, 

4658): 

4659 """Build per-scanpath animation specs from (fixations, color, label) entries. 

4660 

4661 Empty/None fixations are skipped. Onsets are the recorded ``timestamp_ms`` 

4662 rebased to each reading's first fixation, so multiple scanpaths share one 

4663 *real reading-time* clock. When timestamps aren't real times — missing, or 

4664 the 0,1,2,… row index ``data.normalize_fixations`` synthesises when the 

4665 source has no timestamp column — fixations are instead laid out back-to-back 

4666 by their durations. Marker sizes use the figure's duration scale; under the 

4667 relative scale they span the COMBINED durations, so equal durations still 

4668 render at equal sizes across the two scanpaths. 

4669 

4670 ``size_ranges`` (one per entry, default ``marker_size_range`` for each) 

4671 gives each scanpath its own size range, as Compare's per-scanpath *Size* 

4672 does: the duration scale stays shared — one duration is one *fraction* of 

4673 the range on either side — and each side maps that fraction onto its own. 

4674 """ 

4675 from .measures import rebased_fixation_onsets 

4676 

4677 if size_ranges is None: 

4678 size_ranges = [marker_size_range] * len(entries) 

4679 specs = [] 

4680 for (fix_df, color, label), size_range in zip(entries, size_ranges): 

4681 if fix_df is None or fix_df.empty: 

4682 continue 

4683 ordered = fix_df.sort_values("timestamp_ms").reset_index(drop=True) 

4684 dur = pd.to_numeric(ordered["duration_ms"], errors="coerce").fillna(0) 

4685 # Recorded-timestamp-vs-synthetic-index heuristic (shared with the 

4686 # similarity time-curve): trust recorded timestamps only when they look 

4687 # like real times, else lay fixations back-to-back by their durations. 

4688 onsets = rebased_fixation_onsets(ordered) 

4689 specs.append( 

4690 dict( 

4691 ordered=ordered, 

4692 dur=dur, 

4693 onsets=onsets, 

4694 end=float(onsets[-1] + dur.iloc[-1]), 

4695 color=color, 

4696 label=label, 

4697 size_range=tuple(size_range), 

4698 ) 

4699 ) 

4700 if specs: 

4701 # Each duration's place on the shared scale, 0..1, then sized in its 

4702 # own scanpath's range — identical to sizing the combined durations in 

4703 # one range whenever the two ranges agree. 

4704 fractions = _compute_marker_sizes( 

4705 pd.concat([s["dur"] for s in specs], ignore_index=True), 

4706 (0.0, 1.0), 

4707 scale, 

4708 duration_range, 

4709 ) 

4710 cursor = 0 

4711 for s in specs: 

4712 n = len(s["dur"]) 

4713 low, high = s["size_range"] 

4714 s["sizes"] = low + np.asarray( 

4715 fractions[cursor : cursor + n], dtype=float 

4716 ) * (high - low) 

4717 cursor += n 

4718 return specs 

4719 

4720 

4721def _anim_frame_duration_ms(frame_step_ms: float, playback_speed: float) -> int: 

4722 """▶ Play's own per-frame duration: one grid step at the playback speed. 

4723 

4724 Floored at ``_ANIM_MIN_FRAME_MS``. The only part of a replay the speed 

4725 changes besides ``layout.meta`` — which is what lets `set_replay_clock` 

4726 re-time a built replay (PERF-15).""" 

4727 return int(max(frame_step_ms / max(playback_speed, 1e-6), _ANIM_MIN_FRAME_MS)) 

4728 

4729 

4730def _anim_timeline(specs, *, grid_step_ms=None, max_frames=None): 

4731 """Uniform time-grid frame timeline across all scanpaths (VIZ-11). 

4732 

4733 Returns ``(frame_times, frame_step_ms, reading_span_ms)``. Frames are 

4734 emitted on a **uniform time grid** — one every ``step`` ms, where ``step`` is 

4735 ``grid_step_ms`` unless that would exceed ``max_frames`` frames (then it 

4736 coarsens) — so the slider scrubs linearly through reading time no matter how 

4737 fixations cluster or how many scanpaths overlay (the union of onset sets is 

4738 meaningless for >1 reader). ``frame_step_ms`` is that exact step (``0.0`` with 

4739 no frames). None of it depends on the playback speed: ▶ Play's own per-frame 

4740 duration is :func:`_anim_frame_duration_ms` of the step, used only where the 

4741 wall-clock player isn't embedded, and the player shows frame k once 

4742 ``frame_times[k] / playback_speed`` has elapsed, so a replay takes 

4743 ``reading_span_ms / playback_speed`` (BUG-93). Frame *content* is 

4744 unchanged — every fixation whose onset ≤ t shows at time t. All readings are 

4745 rebased to t=0; ``reading_span_ms`` is the longest reading's span. Returns an 

4746 empty grid when there is nothing to animate. 

4747 

4748 Both knobs are user-facing (VIZ-11 follow-up): they trade smoothness against 

4749 frame count, which is what the GIF/MP4 export size and render time are made 

4750 of. Defaults are ``_ANIM_GRID_STEP_MS`` / ``_ANIM_MAX_FRAMES``. 

4751 """ 

4752 step_pref = float(grid_step_ms if grid_step_ms else _ANIM_GRID_STEP_MS) 

4753 cap = int(max_frames if max_frames else _ANIM_MAX_FRAMES) 

4754 reading_span_ms = max((s["end"] for s in specs), default=0.0) 

4755 if not specs or reading_span_ms <= 0: 

4756 return [], 0.0, reading_span_ms 

4757 step = max(step_pref, reading_span_ms / max(cap, 1)) 

4758 frame_times = [ 

4759 min(k * step, reading_span_ms) for k in range(int(reading_span_ms // step) + 1) 

4760 ] 

4761 # Land the final frame exactly on the reading end so it reveals everything. 

4762 if frame_times[-1] < reading_span_ms: 

4763 frame_times.append(reading_span_ms) 

4764 return frame_times, step, reading_span_ms 

4765 

4766 

4767def _revealed_xy(all_x, all_y, kk): 

4768 """Full-length x/y with only the first ``kk`` fixations revealed. 

4769 

4770 Not-yet-reached fixations are masked to ``None`` so Plotly draws nothing 

4771 there. The array length is the SAME in every frame — the replay reveals a 

4772 fixation by un-masking its coordinate, never by growing the array — which is 

4773 what lets the Play button animate with ``redraw=False`` (only positions 

4774 change, so Plotly skips redrawing the static word boxes/labels each frame). 

4775 """ 

4776 n = len(all_x) 

4777 xs = [all_x[i] if i < kk else None for i in range(n)] 

4778 ys = [all_y[i] if i < kk else None for i in range(n)] 

4779 return xs, ys 

4780 

4781 

4782def _revealed_saccade_xy(all_x, all_y, kk): 

4783 """Constant-length saccade polyline for the first ``kk`` fixations. 

4784 

4785 Every consecutive fixation pair occupies a fixed ``(x0, x1, None)`` slot; 

4786 segments past the ``kk``-th fixation are blanked to ``None`` so the trace 

4787 length never changes frame to frame (same ``redraw=False`` requirement as 

4788 :func:`_revealed_xy`). Only which segments are drawn changes. 

4789 """ 

4790 sx, sy = [], [] 

4791 for j in range(len(all_x) - 1): 

4792 if j < kk - 1: 

4793 sx.extend([all_x[j], all_x[j + 1], None]) 

4794 sy.extend([all_y[j], all_y[j + 1], None]) 

4795 else: 

4796 sx.extend([None, None, None]) 

4797 sy.extend([None, None, None]) 

4798 return sx, sy 

4799 

4800 

4801def _revealed_arrow_xy(all_x, all_y, seg_index, kk): 

4802 """Constant-length saccade-arrow positions for the first ``kk`` fixations. 

4803 

4804 An arrowhead belongs to the saccade leaving fixation ``seg_index[j]``, so it 

4805 appears exactly when :func:`_revealed_saccade_xy` draws that segment — the 

4806 arrows reveal *with* their saccades instead of all standing there from frame 

4807 zero. Hidden arrows are masked to ``None`` rather than dropped, keeping the 

4808 array (and its ``marker.angle``) the same length every frame, which is what 

4809 the ``redraw=False`` playback needs. 

4810 """ 

4811 xs = [x if seg_index[j] < kk - 1 else None for j, x in enumerate(all_x)] 

4812 ys = [y if seg_index[j] < kk - 1 else None for j, y in enumerate(all_y)] 

4813 return xs, ys 

4814 

4815 

4816def animation_playback_ms( 

4817 fixations_list, playback_speed, *, grid_step_ms=None, max_frames=None 

4818): 

4819 """Reading span and *actual* animation runtime for the given scanpath(s). 

4820 

4821 Returns ``(reading_span_ms, playback_ms)``. ``playback_ms`` is what the replay 

4822 takes: the wall-clock player (:func:`animation_player_post_script`) reaches the 

4823 last frame once ``reading_span_ms / playback_speed`` has elapsed, so that is 

4824 the time the side panel quotes and a GIF/MP4 lasts (BUG-93). Both 0 when there 

4825 are no fixations. 

4826 """ 

4827 summary = animation_timeline_summary( 

4828 fixations_list, playback_speed, grid_step_ms=grid_step_ms, max_frames=max_frames 

4829 ) 

4830 return summary["reading_span_ms"], summary["playback_ms"] 

4831 

4832 

4833def animation_timeline_summary( 

4834 fixations_list, playback_speed, *, grid_step_ms=None, max_frames=None 

4835) -> dict: 

4836 """What the chosen frame grid actually produces, without building the figure. 

4837 

4838 VIZ-11 follow-up: the grid step and the frame cap are user controls, so the UI 

4839 has to show their consequence — frame count, the effective step, and whether 

4840 the cap *coarsened* the requested step. Silently coarsening is the thing that 

4841 made the old hard-coded behaviour opaque. 

4842 

4843 Returns ``{"n_frames", "step_ms", "requested_step_ms", "coarsened", 

4844 "frame_duration_ms", "reading_span_ms", "playback_ms"}``. 

4845 """ 

4846 requested = float(grid_step_ms if grid_step_ms else _ANIM_GRID_STEP_MS) 

4847 specs = _scanpath_anim_specs( 

4848 [(f, None, None) for f in fixations_list], DEFAULT_MARKER_SIZE_RANGE 

4849 ) 

4850 frame_times, frame_step_ms, reading_span_ms = _anim_timeline( 

4851 specs, grid_step_ms=grid_step_ms, max_frames=max_frames 

4852 ) 

4853 n_frames = len(frame_times) 

4854 step = (frame_times[1] - frame_times[0]) if n_frames > 1 else float(reading_span_ms) 

4855 return { 

4856 "n_frames": n_frames, 

4857 "step_ms": float(step), 

4858 "requested_step_ms": requested, 

4859 "coarsened": bool(n_frames > 1 and step > requested + 1e-6), 

4860 "frame_duration_ms": _anim_frame_duration_ms(frame_step_ms, playback_speed), 

4861 "reading_span_ms": float(reading_span_ms), 

4862 "playback_ms": float(reading_span_ms) / max(playback_speed, 1e-6), 

4863 } 

4864 

4865 

4866# BUG-93 — the replay's clock. Plotly's own ▶ Play steps a frame on the first 

4867# display tick *after* its duration and restarts the next frame's clock from 

4868# there, so every hold rounds up to whole ticks and the rounding accumulates: on 

4869# a 60 Hz screen a 40 ms frame lasts 50 ms, and a 20.8 s reading replayed in 26 s. 

4870# No duration can fix that from here — the tick is the viewer's. So every HTML 

4871# surface (`tabs._true_scale_plot_html`, `tabs._animation_html`, 

4872# `api.save_figure`) embeds a small player that shows whichever frame the wall 

4873# clock has reached, reading the frame times, the speed and VIZ-10's autoplay 

4874# intent off `fig.layout.meta`, where `make_scanpath_animation` stamps them. 

4875_AUTOPLAY_META_FLAG = "scanpath_autoplay" 

4876_REPLAY_META_TIMES = "scanpath_frame_times_ms" 

4877_REPLAY_META_SPEED = "scanpath_playback_speed" 

4878 

4879# `{plot_id}` stays literal: plotly.py substitutes it (a plain `str.replace`, so 

4880# the braces need no escaping) and runs the script in a `.then()` after `newPlot`. 

4881_REPLAY_PLAYER_JS = """(function () { 

4882 var gd = document.getElementById('{plot_id}'); 

4883 if (!gd) { return; } 

4884 var tries = 0; 

4885 // Frames live on gd._transitionData._frames (gd.frames is undefined) and are 

4886 // attached after newPlot resolves, so wait for them rather than a fixed delay. 

4887 (function init() { 

4888 var meta = gd.layout && gd.layout.meta; 

4889 var td = gd._transitionData; 

4890 if (typeof Plotly === 'undefined' || !gd.on || !meta || 

4891 !(td && td._frames && td._frames.length)) { 

4892 if (++tries < 200) { setTimeout(init, 50); } 

4893 return; 

4894 } 

4895 run(meta); 

4896 })(); 

4897 

4898 function run(meta) { 

4899 var times = meta.scanpath_frame_times_ms; 

4900 var speed = meta.scanpath_playback_speed; 

4901 if (!times || !times.length || !(speed > 0)) { return; } 

4902 var last = times.length - 1; 

4903 var jump = {mode: 'immediate', frame: {duration: 0, redraw: false}, 

4904 transition: {duration: 0}}; 

4905 var shown = 0, raf = null, t0 = 0, resume = false; 

4906 

4907 function frameAt(ms) { // the last frame whose reading time has been reached 

4908 var lo = 0, hi = last; 

4909 while (lo < hi) { 

4910 var mid = (lo + hi + 1) >> 1; 

4911 if (times[mid] <= ms) { lo = mid; } else { hi = mid - 1; } 

4912 } 

4913 return lo; 

4914 } 

4915 function show(k) { 

4916 shown = k; 

4917 Plotly.animate(gd, [String(k)], jump); 

4918 } 

4919 // A late tick skips frames rather than falling behind the clock. 

4920 function tick() { 

4921 if (!gd.isConnected) { raf = null; leave(); return; } 

4922 var k = frameAt((performance.now() - t0) * speed); 

4923 if (k !== shown) { show(k); } 

4924 raf = k < last ? requestAnimationFrame(tick) : null; 

4925 } 

4926 function stop() { 

4927 if (raf !== null) { cancelAnimationFrame(raf); raf = null; } 

4928 } 

4929 function play() { 

4930 if (raf !== null) { return; } // already playing: keep the clock 

4931 show(shown < last ? shown : 0); // at the end, Play starts over 

4932 t0 = performance.now() - times[shown] / speed; 

4933 raf = requestAnimationFrame(tick); 

4934 } 

4935 

4936 // Whatever put a frame on screen — this clock, the slider, Restart — Play 

4937 // resumes from it. 

4938 gd.on('plotly_animatingframe', function (e) { 

4939 var k = parseInt(e && e.name, 10); 

4940 if (k >= 0 && k <= last) { shown = k; } 

4941 }); 

4942 gd.on('plotly_buttonclicked', function (e) { 

4943 if (e && e.button && e.button.name === 'play') { play(); } else { stop(); } 

4944 }); 

4945 gd.on('plotly_sliderstart', stop); 

4946 gd.on('plotly_sliderchange', function (e) { if (e && e.interaction) { stop(); } }); 

4947 // A background tab gets no ticks; carry on from the same frame on return. 

4948 function onVisibility() { 

4949 if (!gd.isConnected) { stop(); leave(); return; } 

4950 if (document.hidden) { resume = raf !== null; stop(); } 

4951 else if (resume) { resume = false; play(); } 

4952 } 

4953 // A page that swaps content without reloading (the docs site) can drop the 

4954 // plot; let go of the document then, so the plot can be collected. 

4955 function leave() { 

4956 document.removeEventListener('visibilitychange', onVisibility); 

4957 } 

4958 document.addEventListener('visibilitychange', onVisibility); 

4959 

4960 // Plotly still draws the ▶ Play button; this clock takes over what it does. 

4961 var edit = {}; 

4962 (gd.layout.updatemenus || []).forEach(function (menu, i) { 

4963 (menu.buttons || []).forEach(function (button, j) { 

4964 if (button.name === 'play') { 

4965 edit['updatemenus[' + i + '].buttons[' + j + '].execute'] = false; 

4966 } 

4967 }); 

4968 }); 

4969 Promise.resolve(Plotly.relayout(gd, edit)).then(function () { 

4970 if (meta.scanpath_autoplay) { play(); } 

4971 }); 

4972 } 

4973})();""" 

4974 

4975 

4976def animation_player_post_script(fig) -> str | None: 

4977 """The replay player for an animated scanpath, or ``None`` if there is none. 

4978 

4979 Pass it to ``fig.to_html(post_script=…)`` / ``write_html(post_script=…)`` 

4980 (with ``auto_play=False``) for any figure :func:`make_scanpath_animation` 

4981 built — or its ``to_dict()``; ``None`` — for a static figure, or a replay 

4982 with no frames — is what those calls take for "no script" (BUG-93). 

4983 

4984 The player keeps the replay on the wall clock: at every display tick it 

4985 shows the last frame whose reading time ``elapsed × playback_speed`` has 

4986 reached, so a replay takes ``reading span / playback_speed`` exactly — a slow 

4987 tick skips frames instead of pushing every later one back — and a background 

4988 tab pauses it. It takes the ▶ Play button over (the button's own command is 

4989 switched off with ``execute: false``, Plotly's hook for exactly this, and the 

4990 click still arrives as ``plotly_buttonclicked``); Pause, Restart and the time 

4991 slider keep their own commands, and any of them stops the clock. VIZ-10's 

4992 autoplay starts it on load, from the first frame. 

4993 

4994 It **polls** for Plotly and the figure's frames before starting. Two things 

4995 made a one-shot kick-off silently never fire (VIZ-10), both confirmed 

4996 against a live Plotly build: frames live on ``gd._transitionData._frames``, 

4997 **not** ``gd.frames`` (``undefined``), and the library can arrive late (CDN 

4998 latency, the true-scale iframe mount) and attaches its frames only after 

4999 ``newPlot`` resolves. Polling every 50 ms (capped at ~10 s) covers all of it, 

5000 on the live embed and saved HTML alike. 

5001 

5002 Without the script — ``fig.show()``, or a plain ``write_html`` — the figure 

5003 still plays on Plotly's own queue, at the frame duration the ▶ Play button 

5004 carries. 

5005 """ 

5006 if isinstance(fig, dict): # a figure's `to_dict()` 

5007 meta = (fig.get("layout") or {}).get("meta") 

5008 else: 

5009 meta = getattr(fig.layout, "meta", None) 

5010 if not isinstance(meta, dict) or not meta.get(_REPLAY_META_TIMES): 

5011 return None 

5012 return _REPLAY_PLAYER_JS 

5013 

5014 

5015# PERF-17 — the replay's frames, as its HTML page carries them. Every frame 

5016# restates every animated trace at full length (see `_revealed_xy`), ~33 KB a 

5017# frame however little changed, so a 2,001-frame replay was a 66 MB page. The 

5018# page instead carries frame 0 whole and, for each later frame, only what changed 

5019# since the one before; this decoder rebuilds the exact frame list in the browser 

5020# and hands it to `Plotly.addFrames`, ahead of the player (which already polls for 

5021# the frames). Delta-against-the-previous-frame can't be Plotly's own frames: the 

5022# slider jumps from any frame to any other, so each frame has to be complete. 

5023# 

5024# A delta node is `null` (unchanged), `[0, value]` (replaced), `[1, {key: node}, 

5025# [dropped keys]?]` (an object's changed keys), or `[2, [indices], [values]]` (a 

5026# same-length array's changed slots). Unchanged values are shared between frames 

5027# rather than copied: Plotly copies a frame's objects before applying it and 

5028# never writes into a frame. `__SCANPATH_PACKED_FRAMES__` is replaced by the 

5029# JSON; `{plot_id}` stays literal for plotly.py, as in the player. 

5030_PACKED_FRAMES_TOKEN = "__SCANPATH_PACKED_FRAMES__" 

5031_REPLAY_FRAMES_JS = """(function () { 

5032 var gd = document.getElementById('{plot_id}'); 

5033 if (!gd || typeof Plotly === 'undefined') { return; } 

5034 var packed = __SCANPATH_PACKED_FRAMES__; 

5035 var own = Object.prototype.hasOwnProperty; 

5036 function patch(prev, node) { 

5037 if (node === null) { return prev; } 

5038 var out, i, k; 

5039 if (node[0] === 0) { return node[1]; } 

5040 if (node[0] === 1) { 

5041 out = {}; 

5042 for (k in prev) { if (own.call(prev, k)) { out[k] = prev[k]; } } 

5043 for (k in node[1]) { if (own.call(node[1], k)) { out[k] = patch(prev[k], node[1][k]); } } 

5044 for (i = 0; node[2] && i < node[2].length; i++) { delete out[node[2][i]]; } 

5045 return out; 

5046 } 

5047 out = prev.slice(); 

5048 for (i = 0; i < node[1].length; i++) { out[node[1][i]] = node[2][i]; } 

5049 return out; 

5050 } 

5051 var state = {}; 

5052 var frames = packed.map(function (f) { 

5053 var frame = {}, k; 

5054 for (k in f) { if (own.call(f, k) && k !== 'p') { frame[k] = f[k]; } } 

5055 if (f.p) { 

5056 frame.data = f.p.map(function (node, i) { 

5057 var slot = f.traces ? f.traces[i] : i; 

5058 state[slot] = patch(state[slot], node); 

5059 return state[slot]; 

5060 }); 

5061 } 

5062 return frame; 

5063 }); 

5064 Plotly.addFrames(gd, frames); 

5065})();""" 

5066 

5067_ABSENT = object() 

5068 

5069 

5070def _same_value(a, b) -> bool: 

5071 """Whether two figure values serialize to the same JSON value. 

5072 

5073 Strict where JSON is: ``True`` is not ``1``. Errs towards *different* — 

5074 a value judged different is merely carried again, never lost. 

5075 """ 

5076 if type(a) is not type(b): 

5077 return False 

5078 if isinstance(a, np.ndarray): 

5079 if a.dtype != b.dtype or a.shape != b.shape: 

5080 return False 

5081 if a.dtype.hasobject: 

5082 return _same_value(a.tolist(), b.tolist()) 

5083 return a.tobytes() == b.tobytes() 

5084 if isinstance(a, dict): 

5085 return a.keys() == b.keys() and all(_same_value(a[k], b[k]) for k in a) 

5086 if isinstance(a, (list, tuple)): 

5087 if len(a) != len(b): 

5088 return False 

5089 try: 

5090 if a != b: # C speed for the common case: plain numbers and strings 

5091 return False 

5092 except ValueError: # an ndarray inside: no truth value 

5093 pass 

5094 except TypeError: # `pd.NA` against anything else: no truth value either 

5095 return False 

5096 kinds = list(map(type, a)) 

5097 if kinds != list(map(type, b)): 

5098 return False 

5099 kind_set = set(kinds) 

5100 if any(k in (list, tuple, dict, np.ndarray) for k in kind_set): 

5101 return all(_same_value(x, y) for x, y in zip(a, b)) 

5102 # `==` holds -0.0 equal to 0.0, which JSON writes apart. Only an array 

5103 # holding a zero pays for the sign check. 

5104 if any(issubclass(k, float) for k in kind_set) and 0.0 in a: 

5105 return all( 

5106 not isinstance(x, float) or x != 0.0 or _same_sign(x, y) 

5107 for x, y in zip(a, b) 

5108 ) 

5109 return True 

5110 try: 

5111 same = bool(a == b) 

5112 except (TypeError, ValueError): 

5113 return False 

5114 return same and (not isinstance(a, float) or a != 0.0 or _same_sign(a, b)) 

5115 

5116 

5117def _same_sign(a: float, b: float) -> bool: 

5118 return math.copysign(1.0, a) == math.copysign(1.0, b) 

5119 

5120 

5121def _frame_delta(prev, cur): 

5122 """``cur`` as a change to ``prev`` — the delta node `_REPLAY_FRAMES_JS` applies.""" 

5123 if isinstance(cur, dict) and isinstance(prev, dict): 

5124 changed = {} 

5125 for key, value in cur.items(): 

5126 node = _frame_delta(prev.get(key, _ABSENT), value) 

5127 if node is not None: 

5128 changed[key] = node 

5129 dropped = [key for key in prev if key not in cur] 

5130 if dropped: 

5131 return [1, changed, dropped] 

5132 return [1, changed] if changed else None 

5133 if prev is not _ABSENT and _same_value(prev, cur): 

5134 return None 

5135 if isinstance(cur, list) and isinstance(prev, list) and len(cur) == len(prev): 

5136 slots = [i for i, (a, b) in enumerate(zip(prev, cur)) if not _same_value(a, b)] 

5137 # A patch costs an index per slot; past half the array, restate it. 

5138 if 2 * len(slots) < len(cur): 

5139 return [2, slots, [cur[i] for i in slots]] 

5140 return [0, cur] 

5141 

5142 

5143def pack_replay_frames(frames: Sequence[Mapping]) -> list[dict]: 

5144 """Delta-encode a replay's frames for `_REPLAY_FRAMES_JS` (PERF-17). 

5145 

5146 ``frames`` are the figure's ``to_dict()["frames"]`` (or the same as plain 

5147 JSON values). Each packed frame keeps its own keys (``name``, ``traces``, …) 

5148 and replaces ``data`` with ``p``: one delta node per trace against that 

5149 trace's state in the frame before — the trace index is ``traces[i]``, or 

5150 ``i`` without it. Frame 0 has no frame before it, so it is carried whole. 

5151 Values are carried as they are, so serializing the result the way 

5152 ``to_html`` serializes frames (``to_json_plotly``) writes each exactly as 

5153 the frames would have. 

5154 """ 

5155 state: dict = {} 

5156 packed = [] 

5157 for frame in frames: 

5158 data = frame.get("data") 

5159 # A frame whose `data` is null keeps it as it was. 

5160 out = {k: v for k, v in frame.items() if k != "data" or data is None} 

5161 if data is not None: 

5162 traces = frame.get("traces") 

5163 nodes = [] 

5164 for i, trace in enumerate(data): 

5165 slot = traces[i] if traces is not None else i 

5166 nodes.append(_frame_delta(state.get(slot, _ABSENT), trace)) 

5167 state[slot] = trace 

5168 out["p"] = nodes 

5169 packed.append(out) 

5170 return packed 

5171 

5172 

5173def replay_page(fig) -> tuple[dict, str] | None: 

5174 """A replay as its HTML page carries it: ``(figure dict, post_script)``. 

5175 

5176 PERF-17: the dict is the figure without its ``frames``, and the script 

5177 rebuilds them in the browser from :func:`pack_replay_frames`' deltas, then 

5178 runs :func:`animation_player_post_script`'s player — a 2,001-frame replay's 

5179 page falls from 66 MB to under 1 MB. Serialize the dict with 

5180 ``to_html(…, validate=False, post_script=script)``; ``auto_play`` no longer 

5181 matters, since plotly.py sees no frames. ``fig`` may be a figure or its 

5182 ``to_dict()``, which is not modified. ``None`` for a figure with no player (a 

5183 static figure, or a replay with no frames): serialize that one as it is. 

5184 

5185 The figure itself keeps its frames: `api.animate_scanpath`, the GIF/MP4 

5186 export and ``fig.show()`` use them as they are. 

5187 """ 

5188 player = animation_player_post_script(fig) 

5189 if player is None: 

5190 return None 

5191 if not isinstance(fig, dict) and not fig.frames: 

5192 return None 

5193 fig_dict = fig if isinstance(fig, dict) else fig.to_dict() 

5194 frames = fig_dict.get("frames") or [] 

5195 if not frames: 

5196 return None 

5197 if not all(_packable(frame) for frame in frames): 

5198 # Frames this encoding can't address (a typed-array `traces`) travel as 

5199 # plotly.py writes them, still on the player. 

5200 return fig_dict, player 

5201 page = {key: value for key, value in fig_dict.items() if key != "frames"} 

5202 return page, _packed_frames_script(frames) + "\n" + player 

5203 

5204 

5205def _packable(frame: Mapping) -> bool: 

5206 """Whether `pack_replay_frames` can address ``frame``'s traces by index.""" 

5207 data = frame.get("data") 

5208 traces = frame.get("traces") 

5209 if data is not None and not isinstance(data, (list, tuple)): 

5210 return False 

5211 return traces is None or ( 

5212 isinstance(traces, (list, tuple)) 

5213 and all(isinstance(i, int) and not isinstance(i, bool) for i in traces) 

5214 ) 

5215 

5216 

5217def _packed_frames_script(frames: Sequence[Mapping]) -> str: 

5218 """`_REPLAY_FRAMES_JS` carrying ``frames``, packed (PERF-17).""" 

5219 from plotly.io.json import to_json_plotly 

5220 

5221 packed = to_json_plotly(pack_replay_frames(frames)) 

5222 # Inside a <script>: no `<` may close it, and plotly.py substitutes 

5223 # `{plot_id}` across the whole script — both only ever occur inside a JSON 

5224 # string, where the escapes decode to the same text. 

5225 packed = packed.replace("<", "\\u003c").replace("{plot_id}", "\\u007bplot_id}") 

5226 return _REPLAY_FRAMES_JS.replace(_PACKED_FRAMES_TOKEN, packed) 

5227 

5228 

5229def animation_clip_frame_ms(fig) -> float | None: 

5230 """How long a GIF/MP4 of ``fig`` holds each frame to last as long as its replay. 

5231 

5232 The replay takes ``reading span / playback_speed`` — its last frame's 

5233 reading time over the speed stamped on ``layout.meta`` — and a clip spreads 

5234 that evenly over the frames (BUG-93). ``None`` for a figure 

5235 :func:`make_scanpath_animation` didn't build.""" 

5236 meta = getattr(fig.layout, "meta", None) 

5237 if not isinstance(meta, dict): 

5238 return None 

5239 times = meta.get(_REPLAY_META_TIMES) 

5240 speed = meta.get(_REPLAY_META_SPEED) 

5241 if not times or not speed or speed <= 0: 

5242 return None 

5243 return float(times[-1]) / float(speed) / len(times) 

5244 

5245 

5246def _replay_clock_meta(frame_times, playback_speed: float, autoplay: bool) -> dict: 

5247 """The replay's ``layout.meta``: the clock the player reads, and autoplay.""" 

5248 return { 

5249 _AUTOPLAY_META_FLAG: bool(autoplay and frame_times), 

5250 _REPLAY_META_TIMES: frame_times, 

5251 _REPLAY_META_SPEED: float(playback_speed), 

5252 } 

5253 

5254 

5255def set_replay_clock( 

5256 fig: go.Figure, frame_step_ms: float, *, playback_speed: float, autoplay: bool 

5257) -> None: 

5258 """Re-time a replay in place: a new playback speed and autoplay, same frames. 

5259 

5260 PERF-15: the frames depend on neither (BUG-93), only ▶ Play's own frame 

5261 duration and the clock on ``layout.meta`` do, so the app builds a replay once 

5262 and stamps these onto the copy each cache hit returns. ``frame_step_ms`` is 

5263 the exact grid step :func:`build_scanpath_replay` returned with the figure — 

5264 the rounded frame times on ``layout.meta`` could truncate Play's duration to 

5265 a different whole millisecond. The result is byte-identical to building the 

5266 replay at that speed and autoplay. Only the clock's keys change: anything 

5267 else on ``layout.meta`` (the Illustration label's) stays where it is. 

5268 """ 

5269 meta = fig.layout.meta if isinstance(fig.layout.meta, dict) else {} 

5270 times = list(meta.get(_REPLAY_META_TIMES) or []) 

5271 if times: 

5272 fig.layout.updatemenus = _animation_play_buttons( 

5273 _anim_frame_duration_ms(frame_step_ms, playback_speed) 

5274 ) 

5275 fig.layout.meta = {**meta, **_replay_clock_meta(times, playback_speed, autoplay)} 

5276 

5277 

5278#: The replay's transport controls are app chrome, drawn in the app's font. 

5279_REPLAY_UI_FONT = APP_THEME["font"] 

5280 

5281 

5282def _animation_play_buttons(frame_duration): 

5283 """Play / Pause / Restart buttons. 

5284 

5285 Each carries a ``name`` the replay player looks for: on every HTML surface 

5286 :func:`animation_player_post_script` takes ▶ Play over and runs the frames on 

5287 the wall clock (BUG-93), so Play's own ``frame_duration`` drives only a 

5288 figure shown without it (``fig.show()``). 

5289 

5290 Frames step with ``redraw=False``: every animated trace is full length with 

5291 not-yet-reached fixations masked to ``None`` (see :func:`_revealed_xy`), so 

5292 advancing a frame only changes point positions — Plotly updates just those 

5293 few traces instead of redrawing the whole figure (the static word boxes + 

5294 labels) every frame. A full redraw of the scanpath figure costs ~50 ms, which 

5295 on a long trial dwarfed the per-frame budget. Transitions are 0 so frames snap 

5296 into place (no tweening), and the constant array length means a new 

5297 fixation/number appears on its mark instead of gliding in from the corner. 

5298 """ 

5299 return [ 

5300 dict( 

5301 type="buttons", 

5302 showactive=False, 

5303 # A horizontal row above the plot. The scrubber shares this top 

5304 # transport band to keep playback controls together. 

5305 direction="right", 

5306 y=1.0, 

5307 x=0.0, 

5308 xanchor="left", 

5309 yanchor="bottom", 

5310 pad=dict(b=12, l=8), 

5311 # #374 F23: the app's font, not the figure's (often a monospace 

5312 # stimulus font), so the buttons read as the app's own. 

5313 font=dict(family=_REPLAY_UI_FONT), 

5314 buttons=[ 

5315 dict( 

5316 label="▶ Play", 

5317 name="play", 

5318 method="animate", 

5319 args=[ 

5320 None, 

5321 dict( 

5322 frame=dict(duration=frame_duration, redraw=False), 

5323 fromcurrent=True, 

5324 transition=dict(duration=0), 

5325 ), 

5326 ], 

5327 ), 

5328 dict( 

5329 label="⏸ Pause", 

5330 name="pause", 

5331 method="animate", 

5332 args=[ 

5333 [None], 

5334 dict( 

5335 frame=dict(duration=0, redraw=False), 

5336 mode="immediate", 

5337 transition=dict(duration=0), 

5338 ), 

5339 ], 

5340 ), 

5341 dict( 

5342 label="⟲ Restart", 

5343 name="restart", 

5344 method="animate", 

5345 args=[ 

5346 ["0"], 

5347 dict( 

5348 frame=dict(duration=0, redraw=True), 

5349 mode="immediate", 

5350 transition=dict(duration=0), 

5351 ), 

5352 ], 

5353 ), 

5354 ], 

5355 ) 

5356 ] 

5357 

5358 

5359def _animation_time_slider(frame_times, total_ms): 

5360 """Linear time-scrubber slider (VIZ-11). 

5361 

5362 Frame times sit on a uniform grid, so the handle moves linearly through 

5363 reading time. Each step's label is **"elapsed / total s"** (e.g. "1.2 / 

5364 30.0s"), surfaced in the single ``currentvalue`` readout — meaningful for any 

5365 number of overlaid scanpaths, unlike a fixation index. A long reading would 

5366 render a wall of overlapping numbers if every step drew a tick + label, so the 

5367 per-step tick ruler (``ticklen``/``minorticklen`` = 0) and per-step labels 

5368 (transparent ``font``) are hidden; the readout is the one time display. 

5369 """ 

5370 total_s = total_ms / 1000.0 

5371 return [ 

5372 dict( 

5373 active=0, 

5374 # Share the top transport band with Play / Pause / Restart. 

5375 yanchor="bottom", 

5376 xanchor="left", 

5377 ticklen=0, 

5378 minorticklen=0, 

5379 # Per-step labels feed the readout but must not pile up under the 

5380 # track, so draw them fully transparent. 

5381 font=dict(color="rgba(0,0,0,0)"), 

5382 currentvalue=dict( 

5383 font=dict(size=14, color="#444", family=_REPLAY_UI_FONT), 

5384 # #374 F23/F8: the trial's own clock, first fixation onward. 

5385 prefix="Trial time ", 

5386 visible=True, 

5387 xanchor="right", 

5388 ), 

5389 transition=dict(duration=0), 

5390 pad=dict(b=12), 

5391 len=0.6, 

5392 x=0.38, 

5393 y=1.0, 

5394 steps=[ 

5395 dict( 

5396 args=[ 

5397 [str(k)], 

5398 dict( 

5399 frame=dict(duration=0, redraw=True), 

5400 mode="immediate", 

5401 transition=dict(duration=0), 

5402 ), 

5403 ], 

5404 label=f"{frame_times[k] / 1000:.1f} / {total_s:.1f} s", 

5405 method="animate", 

5406 ) 

5407 for k in range(len(frame_times)) 

5408 ], 

5409 ) 

5410 ] 

5411 

5412 

5413def _render_scanpath_animation( 

5414 words: pd.DataFrame, 

5415 fixations: pd.DataFrame, 

5416 *, 

5417 settings: FigureSettings, 

5418 fixations_b: pd.DataFrame | None = None, 

5419 words_b: pd.DataFrame | None = None, 

5420) -> tuple[go.Figure, float]: 

5421 """Frame-by-frame scanpath replay on a real reading-time clock. 

5422 

5423 Returns the figure and the exact grid step its frames sit on, which 

5424 :func:`set_replay_clock` needs to re-time it (PERF-15). 

5425 

5426 Pass ``fixations_b`` (and optionally ``words_b``) to overlay a SECOND 

5427 scanpath animated on the same clock. Every scanpath is rebased to its first 

5428 fixation's ``timestamp_ms``, so they share *real reading time* including the 

5429 saccade/blink gaps between fixations; a frame is emitted at every fixation 

5430 onset across all scanpaths, and the shorter reading finishes first and holds 

5431 while the longer keeps going. The wall-clock player every HTML surface embeds 

5432 (:func:`animation_player_post_script`) shows each frame once its reading time 

5433 over ``playback_speed`` has elapsed, so the whole replay takes 

5434 ``reading_span / playback_speed`` — exactly what 

5435 :func:`animation_playback_ms` reports (and the side panel quotes). 

5436 

5437 With two scanpaths each trail wears its own style — ``style_a`` / 

5438 ``style_b``, resolved exactly as :func:`make_comparison_figure` resolves 

5439 them (colour, size range, opacity, hollow markers and the saccade line's 

5440 colour, dash and width; the comparison palette where a style names none) — 

5441 order numbers are tinted per-scanpath, and an optional A/B legend 

5442 (``show_legend``) names them; word boxes/labels come from 

5443 ``words`` (scanpath A), so the overlay is meaningful for two readings of the 

5444 same text. With one scanpath the behaviour matches the classic single replay 

5445 (order numbers honour ``order_font_color``, no legend). 

5446 

5447 The single replay honours the same fixation-colouring options as 

5448 :func:`make_scanpath_figure`: ``color_by`` (numeric → ``fixation_colorscale`` 

5449 pinned to the whole trial's range so colours stay stable as the trail grows, 

5450 categorical → discrete palette + legend), ``color_by_line``, and an optional 

5451 colorbar (styled by the ``fixation_colorbar_*`` settings, like the static 

5452 figure). The dual overlay 

5453 colours as :func:`make_comparison_figure` does: the metric (on one range 

5454 shared by both readings) or one shared category→colour mapping fills the 

5455 markers, and each reading's flat A/B colour becomes its marker outline. 

5456 

5457 VIZ-23 brought the remaining word-label, arrow and flag options across from 

5458 the static figure, each defaulting to the replay's previous behaviour: 

5459 

5460 - the word labels take ``text_color`` / ``highlight_column`` / 

5461 ``highlight_text_color`` / ``word_hover_measure`` (``highlight_column`` is 

5462 the *text*-marking channel — there is no border-overlay style here); 

5463 - ``show_saccade_arrows`` adds the direction arrowheads, each revealed with 

5464 the saccade it belongs to rather than all at frame zero; 

5465 - ``fixation_flags`` applies the PRE-2 short/long/out-of-bounds 

5466 classification: *Discard* drops those fixations from the replay entirely, 

5467 *Highlight* overlays them in their flag marker as the replay reaches them. 

5468 

5469 ``layout.meta`` carries the replay's clock (each frame's reading time and the 

5470 speed) and, with ``autoplay`` (default on, VIZ-10), the intent to start on 

5471 load *at the configured playback speed*; the player reads both. The figure 

5472 itself is always built paused — autoplay is the embedder's to start. 

5473 """ 

5474 canvas_width = settings.canvas_width 

5475 canvas_height = settings.canvas_height 

5476 base_font_size = settings.base_font_size 

5477 font_family = settings.font_family 

5478 playback_speed = settings.playback_speed 

5479 show_words = settings.show_words 

5480 show_word_labels = settings.show_word_labels 

5481 show_saccades = settings.show_saccades 

5482 show_saccade_arrows = settings.show_saccade_arrows 

5483 show_order = settings.show_order 

5484 marker_size_range = settings.marker_size_range 

5485 order_font_size = settings.order_font_size 

5486 order_font_color = settings.order_font_color 

5487 color_by = settings.color_by 

5488 # BUG-85: as in `make_scanpath_figure` — "line" is colour-by-line. 

5489 color_by_line = settings.color_by_line or color_by == "line" 

5490 fixation_colorscale = settings.fixation_colorscale 

5491 fixation_color_range = settings.fixation_color_range 

5492 fixation_flags = settings.fixation_flags 

5493 fixation_flags_b = settings.fixation_flags_b 

5494 # The replay has no heatmap: its one bar is the fixations'. 

5495 show_colorbars = settings.show_fixation_colorbar 

5496 colorbar_orientation = settings.fixation_colorbar_orientation 

5497 colorbar_tickangle = settings.fixation_colorbar_tickangle 

5498 colorbar_tickfont_size = settings.fixation_colorbar_tickfont_size 

5499 saccade_color = settings.saccade_color 

5500 saccade_style = settings.saccade_style 

5501 saccade_width = settings.saccade_width 

5502 hollow_fixations = settings.hollow_fixations 

5503 fixation_opacity = settings.fixation_opacity 

5504 fixation_color = settings.fixation_color 

5505 fixation_symbol = settings.fixation_symbol 

5506 text_color = settings.text_color 

5507 highlight_column = settings.highlight_column 

5508 highlight_text_color = settings.highlight_text_color 

5509 word_hover_measure = settings.word_hover_measure 

5510 word_hover_fields = settings.word_hover_fields 

5511 fixation_hover_fields = settings.fixation_hover_fields 

5512 background_color = settings.background_color 

5513 # Drawn as written: a label is a name, not Plotly markup (round 9). 

5514 label_a = _plotly_literal(settings.label_a) 

5515 label_b = _plotly_literal(settings.label_b) 

5516 show_legend = settings.show_legend 

5517 line_spacing = settings.line_spacing 

5518 scale_text_to_boxes = settings.scale_text_to_boxes 

5519 background_image = settings.background_image 

5520 background_image_size = settings.background_image_size 

5521 background_image_origin = settings.background_image_origin 

5522 background_image_opacity = settings.background_image_opacity 

5523 fit_to_monitor = settings.fit_to_monitor 

5524 show_coordinate_grid = settings.show_coordinate_grid 

5525 coordinate_grid_spacing = settings.coordinate_grid_spacing 

5526 autoplay = settings.autoplay 

5527 anim_grid_step_ms = settings.anim_grid_step_ms 

5528 anim_max_frames = settings.anim_max_frames 

5529 fig = go.Figure() 

5530 font_settings = dict(family=font_family or FONT_FAMILY, size=base_font_size) 

5531 

5532 word_frames = [w for w in (words, words_b) if w is not None and not w.empty] 

5533 x_range, y_range, *_ = _compute_axis_ranges( 

5534 canvas_width, 

5535 canvas_height, 

5536 (fixations, "x", "y"), 

5537 (fixations_b, "x", "y"), 

5538 word_frames=word_frames, 

5539 fit_to_monitor=fit_to_monitor, 

5540 ) 

5541 

5542 # Fix the display size first so word labels are sized in the data->screen 

5543 # scale (true-to-scale text); the same fitted_w/fitted_h drive the layout. 

5544 fitted_w, fitted_h = _fit_display_size( 

5545 canvas_width, canvas_height, x_range, y_range, spatial_axes=True 

5546 ) 

5547 scale = _display_scale(x_range, y_range, fitted_w, fitted_h) 

5548 label_font_px = _word_label_font_px( 

5549 words, 

5550 scale=scale, 

5551 line_spacing=line_spacing, 

5552 manual_font_px=base_font_size, 

5553 scale_text_to_boxes=scale_text_to_boxes, 

5554 ) 

5555 

5556 # CMP-11: which reading's stimulus the replay draws. The replay has only ever 

5557 # had ONE stimulus layer, so "both" keeps meaning A's here rather than 

5558 # stacking a second identical set of rectangles onto every existing 

5559 # same-dataset co-animation. "b" is the one that matters: a cross-dataset 

5560 # co-animation would otherwise run B's trace over A's text. 

5561 stimulus_words = words 

5562 if ( 

5563 _compare_stimulus_sides(settings.compare_stimulus) == (False, True) 

5564 and words_b is not None 

5565 and not words_b.empty 

5566 ): 

5567 stimulus_words = words_b 

5568 shapes = ( 

5569 build_word_boxes( 

5570 stimulus_words, 

5571 color=settings.word_box_color, 

5572 fill_color=settings.word_box_fill_color, 

5573 fill_opacity=settings.word_box_fill_opacity, 

5574 line_opacity=settings.word_box_line_opacity, 

5575 ) 

5576 if show_words and not stimulus_words.empty 

5577 else [] 

5578 ) 

5579 if show_word_labels and not stimulus_words.empty: 

5580 _add_word_label_trace( 

5581 fig, 

5582 stimulus_words, 

5583 label_font_px, 

5584 font_settings["family"], 

5585 highlight_column=highlight_column, 

5586 text_color=text_color, 

5587 highlight_text_color=highlight_text_color, 

5588 word_hover_measure=word_hover_measure, 

5589 word_hover_fields=word_hover_fields, 

5590 ) 

5591 

5592 # PRE-2 fixation flags (VIZ-23), mirroring the static figure: *Discard* drops 

5593 # the flagged rows before the specs are built, so they vanish from the trail, 

5594 # the saccade polyline, the index labels and the marker-size scaling. Applied 

5595 # AFTER the axis ranges above, exactly as in `make_scanpath_figure` — the view 

5596 # is framed on the trial as recorded, not on what survives the filter. 

5597 flags = fixation_flags or {} 

5598 # B is flagged against its own word boxes when it brought them, else against 

5599 # A's (the overlay draws A's boxes, and two readings of one text share them). 

5600 entry_words = [ 

5601 words, 

5602 words_b if (words_b is not None and not words_b.empty) else words, 

5603 ] 

5604 # CMP-24: B carries its own flags when it was given any — as 

5605 # `fixation_flags_b`, or (the comparison's spelling) on its `style_b`. 

5606 if fixation_flags_b is None and isinstance(settings.style_b, dict): 

5607 fixation_flags_b = settings.style_b.get("fixation_flags") 

5608 flags_b = flags if fixation_flags_b is None else (fixation_flags_b or {}) 

5609 entry_flags = [flags, flags_b] 

5610 if flags: 

5611 fixations = _discard_flagged_fixations(fixations, entry_words[0], flags) 

5612 if flags_b and fixations_b is not None: 

5613 fixations_b = _discard_flagged_fixations(fixations_b, entry_words[1], flags_b) 

5614 

5615 # The co-animation draws each scanpath in its own style, exactly as the 

5616 # static comparison resolves it (`_comparison_scanpath_style`: the rail's 

5617 # per-scanpath colour, size range, opacity, hollow and saccade line). A lone 

5618 # scanpath keeps the figure-wide settings below. 

5619 dual_input = all(f is not None and not f.empty for f in (fixations, fixations_b)) 

5620 styles = [ 

5621 _comparison_scanpath_style( 

5622 idx, style, default_marker_size_range=marker_size_range 

5623 ) 

5624 for idx, style in enumerate((settings.style_a, settings.style_b)) 

5625 ] 

5626 entries = [ 

5627 (frame, style["fix_color"], label) 

5628 for frame, style, label in zip( 

5629 (fixations, fixations_b), styles, (label_a, label_b) 

5630 ) 

5631 ] 

5632 specs = _scanpath_anim_specs( 

5633 entries, 

5634 marker_size_range, 

5635 size_ranges=[ 

5636 style["marker_size_range"] if dual_input else marker_size_range 

5637 for style in styles 

5638 ], 

5639 **_settings_size_scale(settings), 

5640 ) 

5641 # The words frame each surviving scanpath is flagged against (the highlight 

5642 # overlay's out-of-bounds test). `_scanpath_anim_specs` skips empty 

5643 # scanpaths, so apply the same skip rule here to stay aligned with `specs`. 

5644 surviving = [ 

5645 (w, f, style) 

5646 for (fix_df, _color, _label), w, f, style in zip( 

5647 entries, entry_words, entry_flags, styles 

5648 ) 

5649 if fix_df is not None and not fix_df.empty 

5650 ] 

5651 for spec, (spec_words, spec_flags, spec_style) in zip(specs, surviving): 

5652 spec["words"] = spec_words 

5653 spec["flags"] = spec_flags 

5654 spec["style"] = spec_style 

5655 dual = len(specs) > 1 

5656 if not dual and specs: 

5657 # A lone scanpath always wears the canonical single-replay colour, 

5658 # whether it arrived as `fixations` or (degenerately) only as 

5659 # `fixations_b`, so the trail never silently renders in the B colour. 

5660 # VIZ-17/18: honour the caller's uniform fixation colour when one is 

5661 # given, so the replay matches the static figure (and the palette). 

5662 specs[0]["color"] = fixation_color or COMPARISON_PALETTE[0] 

5663 

5664 # Metric colouring, mirroring the static figure's fixation trace (the dual 

5665 # overlay's, mirroring the comparison figure's, comes first). Numeric 

5666 # metrics map through 

5667 # `fixation_colorscale` with cmin/cmax pinned to the WHOLE trial (or the 

5668 # caller's range) up front — otherwise the scale would renormalise to the 

5669 # partial trail on every frame and colours would drift during playback. 

5670 for s in specs: 

5671 s["marker_colors"] = None 

5672 s["marker_extra"] = {} 

5673 s["marker_line"] = None 

5674 category_legend: list = [] 

5675 color_label = color_by or "" 

5676 # VIZ-17: the uniform sentinel means "no variable mapped to hue" — leave the 

5677 # trail on its flat colour rather than looking for a column by that name. 

5678 if color_by == UNIFORM_COLOR_FIELD: 

5679 color_by, color_label = None, "" 

5680 if dual and (color_by or color_by_line): 

5681 # The co-animation colours as the static comparison does: the metric 

5682 # (one shared range) or the shared category colours fill each marker, 

5683 # and each scanpath's own colour becomes its outline — the A/B cue. 

5684 labels = [ 

5685 _fixation_category_labels(s["ordered"], s["words"], color_by, color_by_line) 

5686 for s in specs 

5687 ] 

5688 colors, category_legend = _shared_category_colors( 

5689 labels, avoid=[s["color"] for s in specs] 

5690 ) 

5691 if category_legend: 

5692 color_label = "line" if (color_by_line or color_by == "line") else color_by 

5693 for s, spec_colors in zip(specs, colors): 

5694 if spec_colors is not None: 

5695 s["marker_colors"] = spec_colors 

5696 s["marker_line"] = dict(color=s["color"], width=1.4) 

5697 elif color_by and all( 

5698 color_by in s["ordered"].columns 

5699 and pd.api.types.is_numeric_dtype(s["ordered"][color_by]) 

5700 for s in specs 

5701 ): 

5702 values = pd.concat([s["ordered"][color_by] for s in specs]) 

5703 if values.notna().any(): 

5704 rng = fixation_color_range or ( 

5705 float(values.min()), 

5706 float(values.max()), 

5707 ) 

5708 for i, s in enumerate(specs): 

5709 # One bar, on A's trail — the scale is shared. 

5710 bar = show_colorbars and i == 0 

5711 s["marker_colors"] = list(s["ordered"][color_by]) 

5712 s["marker_line"] = dict(color=s["color"], width=1.4) 

5713 s["marker_extra"] = dict( 

5714 colorscale=fixation_colorscale, 

5715 cmin=rng[0], 

5716 cmax=rng[1], 

5717 showscale=bar, 

5718 colorbar=_colorbar_dict( 

5719 _column_title(color_label), 

5720 orientation=colorbar_orientation, 

5721 tickangle=colorbar_tickangle, 

5722 tickfont_size=colorbar_tickfont_size, 

5723 ) 

5724 if bar 

5725 else None, 

5726 ) 

5727 if not dual and specs and (color_by or color_by_line): 

5728 ordered0 = specs[0]["ordered"] 

5729 if color_by_line and not words.empty: 

5730 from .measures import assign_fixation_lines 

5731 

5732 line_ids = assign_fixation_lines(ordered0, words) 

5733 color_data = line_ids.map( 

5734 lambda v: f"Line {int(v) + 1}" if pd.notna(v) else "Out of bounds" 

5735 ) 

5736 color_label = "line" 

5737 is_numeric_color = False 

5738 else: 

5739 color_data = ordered0[color_by] if color_by in ordered0.columns else None 

5740 is_numeric_color = color_data is not None and pd.api.types.is_numeric_dtype( 

5741 color_data 

5742 ) 

5743 if color_data is not None: 

5744 marker_color, category_legend = _resolve_marker_colors( 

5745 color_data, is_numeric_color 

5746 ) 

5747 specs[0]["marker_colors"] = list(marker_color) 

5748 if is_numeric_color: 

5749 rng = fixation_color_range or ( 

5750 float(color_data.min()), 

5751 float(color_data.max()), 

5752 ) 

5753 # VIZ-23: the same styled colorbar the static figure builds, so 

5754 # orientation / tick angle / tick size apply here too. 

5755 colorbar = None 

5756 if show_colorbars: 

5757 colorbar = _colorbar_dict( 

5758 _column_title(color_label), 

5759 orientation=colorbar_orientation, 

5760 tickangle=colorbar_tickangle, 

5761 tickfont_size=colorbar_tickfont_size, 

5762 ) 

5763 specs[0]["marker_extra"] = dict( 

5764 colorscale=fixation_colorscale, 

5765 cmin=rng[0], 

5766 cmax=rng[1], 

5767 showscale=show_colorbars, 

5768 colorbar=colorbar, 

5769 ) 

5770 

5771 def _trail_marker(s): 

5772 """Marker dict for a (full-length) trail trace. 

5773 

5774 Every animated trace is full length with not-yet-reached fixations masked 

5775 to ``None`` positions (see :func:`_revealed_xy`), so the size/colour 

5776 arrays are stated once at full length and never change frame to frame — 

5777 only which positions are revealed does. Restating the whole marker keeps 

5778 the colorscale/cmin/cmax/colorbar attached to the trail.""" 

5779 colors = s["marker_colors"] 

5780 marker = dict( 

5781 size=list(s["sizes"]), 

5782 # A glyph shape (♥) has no Plotly symbol: `_trail_traces` draws this 

5783 # dict as text instead, so the symbol here only needs to be valid. 

5784 symbol=_marker_symbol(fixation_symbol), 

5785 color=colors if colors is not None else s["color"], 

5786 line=s["marker_line"] or dict(color=FIX_MARKER_OUTLINE, width=0.5), 

5787 **s["marker_extra"], 

5788 ) 

5789 # Always set the alpha (even 1.0) so the control overrides Plotly's ~0.7 

5790 # default for variable-size scatter markers (VIZ-6). 

5791 marker["opacity"] = float(s["opacity"] if s["opacity"] is not None else 1.0) 

5792 if s["hollow"]: 

5793 marker = _make_hollow(marker) 

5794 return marker 

5795 

5796 glyph = FIXATION_GLYPH_SYMBOLS.get(fixation_symbol or "") 

5797 

5798 def _trail_style(s) -> tuple[dict, list[dict] | None]: 

5799 """The trail's marker dict and, for a glyph shape, its text layers — 

5800 stated once per scanpath and reused by every frame, which only moves 

5801 positions. Building them per frame re-sampled the colorscale (glyph and 

5802 hollow markers) for every fixation on each of ~360 frames.""" 

5803 if "trail_style" not in s: 

5804 marker = _trail_marker(s) 

5805 layers = _glyph_layers(marker, glyph, s["n_total"]) if glyph else None 

5806 s["trail_style"] = (marker, layers) 

5807 return s["trail_style"] 

5808 

5809 def _trail_traces(s, x, y, *, make: Callable = go.Scatter, **top) -> list: 

5810 """The trail as drawn at positions ``x``/``y`` — one marker trace, or 

5811 for a glyph shape (♥) its text layers (`_glyph_scatter_traces`), which 

5812 un-mask exactly like the markers do. Full length either way, so the 

5813 frames still only move positions. ``make=dict`` for a frame.""" 

5814 marker, layers = _trail_style(s) 

5815 if layers is not None: 

5816 return _glyph_layer_traces( 

5817 x, y, layers, make=make, customdata=s["customdata"], **top 

5818 ) 

5819 return [ 

5820 make( 

5821 x=x, 

5822 y=y, 

5823 mode="markers", 

5824 marker=marker, 

5825 text=s["order_text"], 

5826 customdata=s["customdata"], 

5827 **top, 

5828 ) 

5829 ] 

5830 

5831 # Base traces, with stable indices the frames update by position. Each 

5832 # animated trace is built at FULL length (one slot per fixation); the replay 

5833 # reveals a fixation by un-masking its x/y, never by growing the array or 

5834 # rewriting `text`. Constant length + position-only changes are what let the 

5835 # Play button animate with `redraw=False` (see `_animation_play_buttons`): 

5836 # Plotly then re-renders only these few traces per frame instead of redrawing 

5837 # the static word boxes + labels every time — the redraw cost that made a 

5838 # long replay run far slower than its quoted time. It also keeps the trail's 

5839 # fixation number in `text` (hover only); the visible order numbers live in a 

5840 # separate text trace (below). 

5841 # Scanpath legend entries of their own (dual + legend only): see the trail. 

5842 own_entries = bool(category_legend) or bool(glyph) 

5843 for s in specs: 

5844 ordered = s["ordered"] 

5845 n_total = len(ordered) 

5846 all_x = ordered["x"].tolist() 

5847 all_y = ordered["y"].tolist() 

5848 s["all_x"] = all_x 

5849 s["all_y"] = all_y 

5850 s["n_total"] = n_total 

5851 hover_fields = ( 

5852 ["order_in_trial", "duration_ms"] 

5853 if fixation_hover_fields is None 

5854 else list(fixation_hover_fields) 

5855 ) 

5856 s["customdata"], s["hovertemplate"] = _hover_payload( 

5857 ordered, hover_fields, fixation=True, words=s.get("words") 

5858 ) 

5859 # The trial's own fixation numbers, as the static figure and the hover 

5860 # show them — never a 1..n renumbering of what survived the filters. 

5861 s["order_text"] = _fixation_order_labels(ordered) 

5862 s["text_color"] = s["color"] if dual else order_font_color 

5863 # The co-animation draws each scanpath's saccades, opacity and hollow 

5864 # markers from its own style, as the static comparison does; a lone 

5865 # replay keeps the figure-wide settings. 

5866 style = s["style"] 

5867 s["sac_color"] = style["saccade_color"] if dual else saccade_color 

5868 s["sac_width"] = style["saccade_width"] if dual else saccade_width 

5869 s["sac_dash"] = style["saccade_style"] if dual else saccade_style 

5870 s["opacity"] = style["opacity"] if dual else fixation_opacity 

5871 s["hollow"] = bool(style["hollow"]) if dual else hollow_fixations 

5872 s["curr_outline"] = s["color"] if dual else CURRENT_FIX_OUTLINE 

5873 s["curr_outline_w"] = 2.5 if dual else 2 

5874 

5875 base_x, base_y = _revealed_xy(all_x, all_y, 1) 

5876 trail = _trail_traces( 

5877 s, 

5878 base_x, 

5879 base_y, 

5880 # A/B legend on the dual overlay only — off by default, honours the 

5881 # compare-legend toggle (CMP-2). The single-replay colour-by legend 

5882 # below is separate and unaffected. Under shared category colours 

5883 # the swatch would show a category, and a glyph has no swatch, so a 

5884 # separate entry (below) names the scanpath instead. 

5885 showlegend=dual and show_legend and not own_entries, 

5886 name=s["label"], 

5887 legendgroup=s["label"], 

5888 hovertemplate=(s["label"] + "<br>" if dual else "") + s["hovertemplate"], 

5889 ) 

5890 s["idx_trails"] = list(range(len(fig.data), len(fig.data) + len(trail))) 

5891 for trace in trail: 

5892 fig.add_trace(trace) 

5893 # Order numbers: a text trace holding EVERY fixation's final position, 

5894 # with not-yet-reached fixations masked to None x/y (so nothing is drawn 

5895 # there). A number snaps on at its fixation when that position un-masks — 

5896 # no gliding in from the (0,0) corner — and because the `text` strings 

5897 # never change frame to frame, `redraw=False` renders the reveal purely 

5898 # from the position change. 

5899 if show_order: 

5900 s["idx_order"] = len(fig.data) 

5901 fig.add_trace( 

5902 go.Scatter( 

5903 x=base_x, 

5904 y=base_y, 

5905 mode="text", 

5906 text=s["order_text"], 

5907 textfont=dict( 

5908 color=s["text_color"], 

5909 size=order_font_size, 

5910 family=font_settings["family"], 

5911 ), 

5912 textposition="top center", 

5913 showlegend=False, 

5914 legendgroup=s["label"], 

5915 hoverinfo="skip", 

5916 ) 

5917 ) 

5918 else: 

5919 s["idx_order"] = None 

5920 if show_saccades: 

5921 sac_x, sac_y = _revealed_saccade_xy(all_x, all_y, 1) 

5922 s["idx_sac"] = len(fig.data) 

5923 fig.add_trace( 

5924 go.Scatter( 

5925 x=sac_x, 

5926 y=sac_y, 

5927 mode="lines", 

5928 line=dict( 

5929 color=s["sac_color"], width=s["sac_width"], dash=s["sac_dash"] 

5930 ), 

5931 showlegend=False, 

5932 legendgroup=s["label"], 

5933 hoverinfo="skip", 

5934 ) 

5935 ) 

5936 else: 

5937 s["idx_sac"] = None 

5938 # Saccade direction arrowheads (VIZ-23). Same marker as the static 

5939 # figure's, but each arrow is revealed with its own saccade (its position 

5940 # un-masks when `_revealed_saccade_xy` draws that segment) rather than the 

5941 # whole set standing there from frame zero. Angles are stated once at full 

5942 # length and never change, so `redraw=False` still applies. 

5943 arrow_x, arrow_y, arrow_angle, arrow_seg = ( 

5944 _saccade_arrow_rows(ordered, "x", "y") 

5945 if (show_saccades and show_saccade_arrows) 

5946 else ([], [], [], []) 

5947 ) 

5948 s["arrow_x"], s["arrow_y"], s["arrow_seg"] = arrow_x, arrow_y, arrow_seg 

5949 s["idx_arrow"] = None 

5950 if arrow_x: 

5951 ax0, ay0 = _revealed_arrow_xy(arrow_x, arrow_y, arrow_seg, 1) 

5952 s["idx_arrow"] = len(fig.data) 

5953 fig.add_trace( 

5954 go.Scatter( 

5955 x=ax0, 

5956 y=ay0, 

5957 mode="markers", 

5958 marker=dict( 

5959 symbol="arrow", 

5960 size=12, 

5961 angle=arrow_angle, 

5962 angleref="up", 

5963 color=s["sac_color"], 

5964 line=dict(width=0), 

5965 ), 

5966 showlegend=False, 

5967 legendgroup=s["label"], 

5968 hoverinfo="skip", 

5969 name="saccade direction", 

5970 ) 

5971 ) 

5972 s["idx_curr"] = len(fig.data) 

5973 fig.add_trace( 

5974 go.Scatter( 

5975 x=[all_x[0]], 

5976 y=[all_y[0]], 

5977 mode="markers", 

5978 marker=dict( 

5979 size=[float(s["sizes"][0]) + 8], 

5980 color=CURRENT_FIX_COLOR, 

5981 line=dict(color=s["curr_outline"], width=s["curr_outline_w"]), 

5982 ), 

5983 showlegend=False, 

5984 legendgroup=s["label"], 

5985 hoverinfo="skip", 

5986 ) 

5987 ) 

5988 # PRE-2 *Highlight* overlays (VIZ-23): one trace per flagged category, in 

5989 # that category's marker + colour, drawn over the trail. Full-length like 

5990 # every animated trace — fixations that aren't flagged are masked out 

5991 # permanently, the rest un-mask as the replay reaches them. 

5992 s["flag_overlays"] = [] 

5993 s_flags = s.get("flags", flags) 

5994 if s_flags: 

5995 overlay_masks = _fixation_flag_masks(ordered, s["words"], s_flags) 

5996 for category in _FIX_FLAG_CATEGORIES: 

5997 spec_flags = s_flags.get(category, {}) 

5998 if spec_flags.get("mode") != "Highlight": 

5999 continue 

6000 hit = overlay_masks[category].to_numpy() 

6001 if not hit.any(): 

6002 continue 

6003 hx = [all_x[j] if hit[j] else None for j in range(n_total)] 

6004 hy = [all_y[j] if hit[j] else None for j in range(n_total)] 

6005 label = _FIX_FLAG_LABELS[category] 

6006 name = f"{s['label']} · {label}" if dual else label 

6007 fx0, fy0 = _revealed_xy(hx, hy, 1) 

6008 s["flag_overlays"].append( 

6009 dict( 

6010 idx=len(fig.data), 

6011 x=hx, 

6012 y=hy, 

6013 marker=dict( 

6014 symbol=spec_flags.get("symbol") or "x", 

6015 size=13, 

6016 color=spec_flags.get("color") or OUT_OF_TEXT_COLOR, 

6017 line=dict(color="#ffffff", width=1), 

6018 ), 

6019 name=name, 

6020 ) 

6021 ) 

6022 fig.add_trace( 

6023 go.Scatter( 

6024 x=fx0, 

6025 y=fy0, 

6026 mode="markers", 

6027 marker=s["flag_overlays"][-1]["marker"], 

6028 name=name, 

6029 legendgroup=s["label"], 

6030 showlegend=True, 

6031 hovertemplate=( 

6032 f"{label} fixation<br>x %{{x:.0f}}, y %{{y:.0f}}" 

6033 "<extra></extra>" 

6034 ), 

6035 ) 

6036 ) 

6037 

6038 # Categorical colour legend, as in the static figure. These dummy traces sit 

6039 # AFTER the per-scanpath traces so the frame indices recorded above stay 

6040 # valid; frames never touch them. 

6041 if dual and show_legend and own_entries: 

6042 for s in specs: 

6043 coloured = s["marker_line"] is not None 

6044 fig.add_trace( 

6045 go.Scatter( 

6046 x=[None], 

6047 y=[None], 

6048 mode="markers", 

6049 marker=dict( 

6050 size=10, 

6051 symbol=_marker_symbol(fixation_symbol), 

6052 color="#ffffff" if coloured else s["color"], 

6053 line=dict( 

6054 color=s["color"] if coloured else FIX_MARKER_OUTLINE, 

6055 width=2 if coloured else 0.5, 

6056 ), 

6057 ), 

6058 name=s["label"], 

6059 legendgroup=s["label"], 

6060 showlegend=True, 

6061 hoverinfo="skip", 

6062 ) 

6063 ) 

6064 _add_category_legend(fig, category_legend, color_label) 

6065 if glyph and specs: 

6066 # A glyph carries no colorscale, so A's numeric colour bar rides on a 

6067 # trace of its own (after the animated ones, like the legend entries). 

6068 bar = _glyph_colorbar_trace( 

6069 _trail_style(specs[0])[0], specs[0]["marker_colors"] or () 

6070 ) 

6071 if bar is not None: 

6072 fig.add_trace(bar) 

6073 

6074 frame_times, frame_step_ms, reading_span_ms = _anim_timeline( 

6075 specs, 

6076 grid_step_ms=anim_grid_step_ms, 

6077 max_frames=anim_max_frames, 

6078 ) 

6079 

6080 # Frames are plain dicts, validated once, by the `fig.frames` assignment 

6081 # below. Built from `go.Scatter`/`go.Frame` they were validated (and deep- 

6082 # copied) three times over — each trace, each frame, then the figure — 

6083 # which on a long replay is most of the build: every frame restates the 

6084 # trail's full-length sizes and colours (see `_trail_marker`). 

6085 frames = [] 

6086 n_frames = len(frame_times) 

6087 for k, t in enumerate(frame_times): 

6088 # UX-169: the card's "120 of 361 frames" — and a cancel checkpoint, so an 

6089 # abandoned build stops within a frame. A no-op outside a card. 

6090 progress.report(k + 1, n_frames, unit="frames") 

6091 traces_in_frame = [] 

6092 traces_idx_in_frame = [] 

6093 for s in specs: 

6094 all_x = s["all_x"] 

6095 all_y = s["all_y"] 

6096 # Fixations whose recorded onset has been reached by time t. 

6097 kk = max(int(np.searchsorted(s["onsets"], t, side="right")), 1) 

6098 

6099 # Trail: full-length, fixations past kk masked to None. The marker 

6100 # (sizes/colours) and `text` are full-length and identical every 

6101 # frame, so only positions change — `redraw=False` then re-renders 

6102 # just this trace, not the whole figure. 

6103 tx, ty = _revealed_xy(all_x, all_y, kk) 

6104 for idx, trace in zip(s["idx_trails"], _trail_traces(s, tx, ty, make=dict)): 

6105 traces_in_frame.append(trace) 

6106 traces_idx_in_frame.append(idx) 

6107 

6108 if show_order: 

6109 # Same full-length positions/text as the base order trace; the 

6110 # reveal is purely the un-masking of x/y for reached fixations, 

6111 # so numbers appear in place (and `redraw=False` shows them). 

6112 ox, oy = _revealed_xy(all_x, all_y, kk) 

6113 traces_in_frame.append( 

6114 dict( 

6115 x=ox, 

6116 y=oy, 

6117 mode="text", 

6118 text=s["order_text"], 

6119 textfont=dict( 

6120 color=s["text_color"], 

6121 size=order_font_size, 

6122 family=font_settings["family"], 

6123 ), 

6124 textposition="top center", 

6125 ) 

6126 ) 

6127 traces_idx_in_frame.append(s["idx_order"]) 

6128 

6129 if show_saccades: 

6130 sac_x, sac_y = _revealed_saccade_xy(all_x, all_y, kk) 

6131 traces_in_frame.append( 

6132 dict( 

6133 x=sac_x, 

6134 y=sac_y, 

6135 mode="lines", 

6136 line=dict( 

6137 color=s["sac_color"], 

6138 width=s["sac_width"], 

6139 dash=s["sac_dash"], 

6140 ), 

6141 ) 

6142 ) 

6143 traces_idx_in_frame.append(s["idx_sac"]) 

6144 

6145 if s["idx_arrow"] is not None: 

6146 # Arrowheads reveal with the saccades above: same constant-length 

6147 # array, only the mask moves (angles are set on the base trace). 

6148 arw_x, arw_y = _revealed_arrow_xy( 

6149 s["arrow_x"], s["arrow_y"], s["arrow_seg"], kk 

6150 ) 

6151 traces_in_frame.append(dict(x=arw_x, y=arw_y, mode="markers")) 

6152 traces_idx_in_frame.append(s["idx_arrow"]) 

6153 

6154 ci = kk - 1 

6155 traces_in_frame.append( 

6156 dict( 

6157 x=[all_x[ci]], 

6158 y=[all_y[ci]], 

6159 mode="markers", 

6160 marker=dict( 

6161 size=[float(s["sizes"][ci]) + 8], 

6162 color=CURRENT_FIX_COLOR, 

6163 line=dict(color=s["curr_outline"], width=s["curr_outline_w"]), 

6164 ), 

6165 ) 

6166 ) 

6167 traces_idx_in_frame.append(s["idx_curr"]) 

6168 

6169 for overlay in s["flag_overlays"]: 

6170 # A flagged fixation's highlight appears with the fixation itself. 

6171 ox_f, oy_f = _revealed_xy(overlay["x"], overlay["y"], kk) 

6172 traces_in_frame.append( 

6173 dict(x=ox_f, y=oy_f, mode="markers", marker=overlay["marker"]) 

6174 ) 

6175 traces_idx_in_frame.append(overlay["idx"]) 

6176 

6177 frames.append( 

6178 dict(data=traces_in_frame, name=str(k), traces=traces_idx_in_frame) 

6179 ) 

6180 fig.frames = frames 

6181 

6182 shapes.append( 

6183 dict( 

6184 type="rect", 

6185 x0=x_range[0], 

6186 y0=y_range[1], 

6187 x1=x_range[1], 

6188 y1=y_range[0], 

6189 line=dict(color="#000000", width=1), 

6190 fillcolor="rgba(0,0,0,0)", 

6191 ) 

6192 ) 

6193 

6194 sliders = ( 

6195 _animation_time_slider(frame_times, reading_span_ms) if frame_times else [] 

6196 ) 

6197 updatemenus = ( 

6198 _animation_play_buttons(_anim_frame_duration_ms(frame_step_ms, playback_speed)) 

6199 if frame_times 

6200 else [] 

6201 ) 

6202 

6203 # fitted_w / fitted_h were computed up front (so the label scale matched). 

6204 # ALL transport controls (play/pause/restart buttons + the time slider with 

6205 # its elapsed-time readout) sit ABOVE the plot in the top margin. Critically, 

6206 # the figure is made tall enough that the plot region stays >= fitted_h after 

6207 # Plotly's automargin reserves space for those controls — otherwise the 

6208 # equal-aspect (`scaleanchor`) plot would shrink to fit the leftover height, 

6209 # making the word boxes smaller than the true-to-scale label font computed 

6210 # for fitted_h (text-too-large bug). _CONTROLS_MARGIN_PX is that reserve. 

6211 # A single-replay numeric colorbar gets the same treatment on the right 

6212 # (the dual-overlay legend overlays the plot, so it needs no reserve). 

6213 # A HORIZONTAL colorbar (VIZ-23) stays below the plot, so it takes bottom 

6214 # reserve rather than right — the same trade `_decoration_margins` makes for 

6215 # the static figure. 

6216 anim_colorbar = bool(specs and specs[0].get("marker_extra", {}).get("showscale")) 

6217 horizontal_colorbar = anim_colorbar and colorbar_orientation == "Horizontal" 

6218 right_reserve = ( 

6219 _COLORBAR_RESERVE_PX if (anim_colorbar and not horizontal_colorbar) else 0 

6220 ) 

6221 top_reserve = _CONTROLS_MARGIN_PX 

6222 bottom_reserve = _COLORBAR_BOTTOM_PX if horizontal_colorbar else 0 

6223 grid_left = _GRID_LEFT_RESERVE_PX if show_coordinate_grid else 0 

6224 grid_bottom = _GRID_BOTTOM_RESERVE_PX if show_coordinate_grid else 0 

6225 # Stimulus-page background image (MultiplEYE) — same layout image as 

6226 # make_scanpath_figure: placed at its (centered) origin, UNDER every trace, 

6227 # and persisting across frames (a layout image, not per-frame data). Lets the 

6228 # animated replay show the rendered page exactly like the static plot. 

6229 bg_spec = _background_image_spec( 

6230 background_image, 

6231 background_image_size, 

6232 background_image_origin, 

6233 background_image_opacity, 

6234 ) 

6235 bg_images = [bg_spec] if bg_spec else [] 

6236 xaxis = dict( 

6237 showticklabels=False, 

6238 showgrid=False, 

6239 zeroline=False, 

6240 title=None, 

6241 range=x_range, 

6242 constrain="domain", 

6243 automargin=False, 

6244 ) 

6245 yaxis = dict( 

6246 showticklabels=False, 

6247 showgrid=False, 

6248 zeroline=False, 

6249 title=None, 

6250 range=y_range, 

6251 constrain="domain", 

6252 scaleanchor="x", 

6253 scaleratio=1, 

6254 automargin=False, 

6255 ) 

6256 _apply_coordinate_grid_axes( 

6257 xaxis, 

6258 yaxis, 

6259 show=show_coordinate_grid, 

6260 spacing=coordinate_grid_spacing, 

6261 x_range=x_range, 

6262 y_range=y_range, 

6263 rendered_width=fitted_w, 

6264 rendered_height=fitted_h, 

6265 ) 

6266 layout = dict( 

6267 height=( 

6268 fitted_h + top_reserve + bottom_reserve + grid_bottom + _CONTROLS_SAFETY_PX 

6269 ), 

6270 width=fitted_w + grid_left + right_reserve, 

6271 autosize=False, 

6272 images=bg_images, 

6273 margin=dict( 

6274 l=grid_left, 

6275 r=right_reserve, 

6276 t=top_reserve, 

6277 b=bottom_reserve + grid_bottom, 

6278 ), 

6279 xaxis=xaxis, 

6280 yaxis=yaxis, 

6281 template="plotly_white", 

6282 plot_bgcolor=background_color, 

6283 paper_bgcolor=background_color, 

6284 font=font_settings, 

6285 shapes=shapes, 

6286 sliders=sliders, 

6287 updatemenus=updatemenus, 

6288 ) 

6289 # The PRE-2 highlight overlays carry legend entries too, so they get the same 

6290 # floating key as the A/B and colour-category legends. 

6291 flag_legend = any(s.get("flag_overlays") for s in specs) 

6292 if dual or category_legend or flag_legend: 

6293 layout["legend"] = dict( 

6294 orientation="h", 

6295 yanchor="top", 

6296 y=0.99, 

6297 xanchor="right", 

6298 x=0.99, 

6299 bgcolor="rgba(255,255,255,0.7)", 

6300 bordercolor="#cccccc", 

6301 borderwidth=1, 

6302 ) 

6303 if dual: 

6304 layout["legend"]["font"] = _compare_legend_font(base_font_size, font_family) 

6305 fig.update_layout(**layout) 

6306 # BUG-93: the replay's clock — each frame's reading time and the speed — for 

6307 # the wall-clock player every HTML surface embeds, plus VIZ-10's autoplay 

6308 # intent, which that player reads on load (Plotly's own `auto_play` ignores 

6309 # the frame duration). No frames → no times, so no player and no autoplay. 

6310 fig.layout.meta = _replay_clock_meta( 

6311 [round(float(t), 3) for t in frame_times], playback_speed, autoplay 

6312 ) 

6313 return fig, frame_step_ms 

6314 

6315 

6316def _resolve_trial_display_name( 

6317 participant: str, 

6318 trial_id: str, 

6319 trial_words: pd.DataFrame, 

6320 trial_labels: tuple[str, str] | None, 

6321 idx: int, 

6322) -> str: 

6323 """Scanpath ``idx``'s name as written; the builders make it literal 

6324 (:func:`_plotly_literal`) where they draw it.""" 

6325 if trial_labels is not None and len(trial_labels) > idx: 

6326 return trial_labels[idx] 

6327 text_id = None 

6328 if "text_id" in trial_words.columns and not trial_words.empty: 

6329 text_id = trial_words["text_id"].iloc[0] 

6330 text_str = str(text_id) if text_id is not None else "" 

6331 trial_str = str(trial_id) 

6332 contains_text = text_str and text_str.lower() in trial_str.lower() 

6333 if text_str: 

6334 return ( 

6335 f"{text_str} · {participant}" 

6336 if contains_text 

6337 else f"{text_str} · {participant} (trial {trial_str})" 

6338 ) 

6339 return f"{trial_str} · {participant}" 

6340 

6341 

6342def _comparison_scanpath_style( 

6343 idx: int, 

6344 override: dict | None = None, 

6345 *, 

6346 default_marker_size_range: tuple[int, int] = DEFAULT_MARKER_SIZE_RANGE, 

6347) -> dict: 

6348 """Resolve the per-scanpath style for a comparison trace. 

6349 

6350 Defaults reproduce the classic two-flat-colour look (``COMPARISON_PALETTE``); 

6351 ``override`` (from the rail's per-scanpath styling panel) wins per key. 

6352 """ 

6353 base = { 

6354 "fix_color": compare_palette_color(idx), 

6355 "saccade_color": compare_palette_color(idx), 

6356 "saccade_style": "solid", 

6357 "saccade_width": DEFAULT_SACCADE_WIDTH, 

6358 "marker_size_range": default_marker_size_range, 

6359 "hollow": False, 

6360 "opacity": COMPARE_FIXATION_OPACITY, 

6361 } 

6362 if override: 

6363 # Drop falsy values (None / "") so a blank colour can't override the 

6364 # palette default and reach Plotly as a dark/None marker colour. The 

6365 # per-scanpath filters (CMP-24) are the exception: an empty one is a 

6366 # real answer — "no filter on this scanpath" — not a missing colour. 

6367 base.update( 

6368 { 

6369 k: v 

6370 for k, v in override.items() 

6371 if v or (k in COMPARE_FILTER_STYLE_KEYS and v is not None) 

6372 } 

6373 ) 

6374 return base 

6375 

6376 

6377#: CMP-24: the style keys that carry a scanpath's own *filters* rather than its 

6378#: look. A comparison draws each reading under its own — the figure-level 

6379#: ``fixation_flags`` / ``saccade_classes`` apply to a scanpath whose style names 

6380#: none, so one setting filters both and a style entry overrides it per side. 

6381COMPARE_FILTER_STYLE_KEYS = frozenset({"fixation_flags", "saccade_classes"}) 

6382 

6383 

6384def _visible_saccade_classes(classes: Iterable[str] | None) -> set[str] | None: 

6385 """The saccade classes to draw, ``None`` meaning all (VIZ-31's rule: an empty 

6386 or complete list is no filter).""" 

6387 if not classes or set(classes) >= set(SACCADE_CLASS_ORDER): 

6388 return None 

6389 return set(classes) 

6390 

6391 

6392def _comparison_filters(style: dict, settings: FigureSettings) -> dict: 

6393 """One scanpath's filters (CMP-24): its style's own, else the figure's.""" 

6394 return dict( 

6395 fixation_flags=style.get("fixation_flags", settings.fixation_flags), 

6396 saccade_classes=style.get("saccade_classes", settings.saccade_classes), 

6397 ) 

6398 

6399 

6400def _arrow_class_mask( 

6401 fixations: pd.DataFrame, saccade_classes: pd.Series, keep: set, aseg: list 

6402) -> list[bool]: 

6403 """Which arrowheads belong to a visible saccade class (VIZ-31). 

6404 

6405 An arrowhead belongs to the saccade leaving fixation ``aseg[j]`` in time 

6406 order, which is exactly how ``_saccade_segments_by_class`` keys a segment — 

6407 so the filter drops arrows for hidden saccades instead of leaving them 

6408 floating over nothing.""" 

6409 ordered_cls = saccade_classes.reindex( 

6410 fixations.sort_values("timestamp_ms").index 

6411 ).tolist() 

6412 return [ 

6413 ( 

6414 "other" 

6415 if i >= len(ordered_cls) or pd.isna(ordered_cls[i]) 

6416 else ordered_cls[i] 

6417 ) 

6418 in keep 

6419 for i in aseg 

6420 ] 

6421 

6422 

6423def _comparison_raw_gaze( 

6424 raw_gaze: pd.DataFrame | None, trial: tuple[str, str], *, show: bool 

6425) -> pd.DataFrame | None: 

6426 """One reading's raw-gaze samples out of the comparison's frame, or ``None``. 

6427 

6428 ``raw_gaze`` carries both readings keyed exactly as the words / fixations 

6429 frames are — B's ids already namespaced or renamed apart by the caller — so 

6430 it is sliced by the same ``(participant, trial)`` pair. 

6431 """ 

6432 if not show or raw_gaze is None or raw_gaze.empty: 

6433 return None 

6434 rows = raw_gaze[ 

6435 (raw_gaze["participant_id"] == trial[0]) & (raw_gaze["trial_id"] == trial[1]) 

6436 ] 

6437 return rows if not rows.empty else None 

6438 

6439 

6440def _add_comparison_raw_gaze_trace( 

6441 fig: go.Figure, 

6442 samples: pd.DataFrame | None, 

6443 display_name: str, 

6444 color: str, 

6445 settings: FigureSettings, 

6446 *, 

6447 row: int | None = None, 

6448 col: int | None = None, 

6449) -> None: 

6450 """One reading's raw-gaze samples in a comparison figure (VIZ-48). 

6451 

6452 Drawn in that scanpath's own colour (its style's ``raw_gaze_color``, else 

6453 its fixation colour) rather than the single-trial figure's time scale: two clouds on one Viridis ramp could not be told apart in an 

6454 overlay, and the A/B colour is the cue every other comparison layer keeps. 

6455 Size and opacity are the 🔵 Raw gaze settings, as on the single figure. The 

6456 trace joins its scanpath's legend group, so toggling A in the legend hides 

6457 A's samples with it. 

6458 """ 

6459 if samples is None or samples.empty: 

6460 return 

6461 if "timestamp_ms" in samples.columns: 

6462 customdata, when = samples["timestamp_ms"], "<br>Timestamp: %{customdata} ms" 

6463 elif SAMPLE_INDEX in samples.columns: # no clock: the sample's number 

6464 customdata, when = samples[SAMPLE_INDEX], "<br>Sample #: %{customdata}" 

6465 else: 

6466 customdata, when = None, "" 

6467 trace = go.Scatter( 

6468 x=samples["x"], 

6469 y=samples["y"], 

6470 mode="markers", 

6471 marker=dict( 

6472 size=settings.raw_gaze_marker_size, 

6473 color=color, 

6474 opacity=settings.raw_gaze_opacity, 

6475 ), 

6476 hovertemplate=( 

6477 f"Raw gaze · {display_name}<br>x: %{{x:.1f}}<br>y: %{{y:.1f}}" 

6478 + when 

6479 + "<extra></extra>" 

6480 ), 

6481 customdata=customdata, 

6482 name=f"{display_name} · raw gaze", 

6483 legendgroup=display_name, 

6484 showlegend=bool(settings.show_legend), 

6485 ) 

6486 if row is not None: 

6487 fig.add_trace(trace, row=row, col=col) 

6488 else: 

6489 fig.add_trace(trace) 

6490 

6491 

6492def _add_comparison_fixation_trace( 

6493 fig: go.Figure, 

6494 trial_fix: pd.DataFrame, 

6495 display_name: str, 

6496 style: dict, 

6497 font_settings: dict, 

6498 *, 

6499 show_fixations: bool = True, 

6500 show_saccades: bool = True, 

6501 show_saccade_arrows: bool = False, 

6502 show_order: bool = True, 

6503 order_font_size: int | None = None, 

6504 show_legend: bool = False, 

6505 color_by: str | None = None, 

6506 colorscale: str = DEFAULT_FIXATION_COLORSCALE, 

6507 color_range: tuple[float, float] | None = None, 

6508 show_colorbar: bool = False, 

6509 colorbar_style: dict | None = None, 

6510 fixation_symbol: str = DEFAULT_FIXATION_SYMBOL, 

6511 fixation_hover_fields: Sequence[str] | None = None, 

6512 row: int | None = None, 

6513 col: int | None = None, 

6514 trial_words: pd.DataFrame | None = None, 

6515 fixation_flags: dict | None = None, 

6516 saccade_classes: Iterable[str] | None = None, 

6517 duration_scale: dict | None = None, 

6518 category_colors: Sequence[str] | None = None, 

6519) -> None: 

6520 """Add one scanpath's saccades + fixation markers to a comparison figure. 

6521 

6522 Saccades and markers are separate traces (mirroring the single-trial figure) 

6523 so the per-scanpath saccade colour/line-style/line-width and hollow markers 

6524 all apply, and the shared ``show_saccades`` / ``show_saccade_arrows`` / 

6525 ``show_order`` toggles take effect. 

6526 

6527 Fixation colour: by default each scanpath uses its flat per-scanpath colour 

6528 (the A/B cue). When ``color_by`` names a numeric column, the marker **fill** is 

6529 coloured by that metric (shared ``colorscale`` / ``color_range`` across both 

6530 scanpaths) and the per-scanpath flat colour becomes the marker **outline**, so 

6531 the readings stay distinguishable while still showing the metric. Order numbers 

6532 are tinted to the per-scanpath colour either way. 

6533 

6534 ``category_colors`` is the discrete counterpart — one literal colour per row 

6535 of ``trial_fix``, from the figure's shared category→colour mapping 

6536 (:func:`_shared_category_colors`, for a categorical column or colour-by-line). 

6537 It is drawn the same way: category fill, per-scanpath outline. 

6538 

6539 ``fixation_symbol`` (VIZ-15/23) sets the marker shape — shape is the channel 

6540 that survives a greyscale print, which is exactly what comparison figures get 

6541 used for. A glyph shape (♥) is drawn as text, as on the static figure 

6542 (:func:`_glyph_scatter_traces`), its A/B outline a larger glyph beneath. 

6543 

6544 ``show_fixations=False`` (CMP-7) drops the marker trace, and with it the 

6545 fixation-index labels that ride on it as marker text — the same thing the 

6546 toggle does on the static figure. The saccade and arrow layers are 

6547 independent and keep their own toggles, so a lines-only comparison is still 

6548 reachable. This is what makes a comparison *heatmap* readable: the whole 

6549 point of the split word boxes is lost under two full sets of markers. 

6550 

6551 CMP-24: ``fixation_flags`` and ``saccade_classes`` are *this* scanpath's 

6552 filters, applied as the static figure applies them — *Discard* drops markers 

6553 and their index labels (the saccades still bridge across them), *Highlight* 

6554 overlays its marker, and hidden saccade classes lose their line and arrow. 

6555 Both need ``trial_words`` for the geometry they classify against. 

6556 

6557 ``duration_scale`` is the figure's (``_settings_size_scale``): both 

6558 scanpaths share it, so under a fixed scale one duration draws at one size 

6559 on either side; only the size *range* is per scanpath. 

6560 """ 

6561 if trial_fix.empty: 

6562 return 

6563 if category_colors is not None: 

6564 # Positional, before *Discard* drops rows, so the two stay aligned. 

6565 trial_fix = trial_fix.assign(_category_color=list(category_colors)) 

6566 words_for_flags = trial_words if trial_words is not None else pd.DataFrame() 

6567 keep = _visible_saccade_classes(saccade_classes) 

6568 class_series = None 

6569 if keep is not None and (show_saccades or show_saccade_arrows): 

6570 existing = trial_fix.get("saccade_class") 

6571 if existing is not None: 

6572 class_series = existing 

6573 else: 

6574 from .measures import classify_saccades 

6575 

6576 class_series = classify_saccades(trial_fix, words_for_flags) 

6577 fix_color = style["fix_color"] 

6578 saccade_color = style["saccade_color"] 

6579 saccade_style = style.get("saccade_style", "solid") 

6580 saccade_width = style.get("saccade_width", DEFAULT_SACCADE_WIDTH) 

6581 

6582 def _add(trace): 

6583 if row is not None and col is not None: 

6584 fig.add_trace(trace, row=row, col=col) 

6585 else: 

6586 fig.add_trace(trace) 

6587 

6588 # Comparison figures always draw straight connectors (no Arc mode here); bind 

6589 # it once so the segments and the arrowheads can never disagree (BUG-9). 

6590 arch_frac: float | None = None 

6591 if show_saccades and len(trial_fix) > 1: 

6592 if class_series is not None: 

6593 segs = _saccade_segments_by_class( 

6594 trial_fix, "x", "y", class_series, arch_frac 

6595 ) 

6596 sx, sy = [], [] 

6597 for cls_name, (cx, cy) in segs.items(): 

6598 if cls_name in keep: 

6599 sx.extend(cx) 

6600 sy.extend(cy) 

6601 else: 

6602 sx, sy = _saccade_segments(trial_fix, "x", "y", arch_frac) 

6603 if sx: 

6604 _add( 

6605 go.Scatter( 

6606 x=sx, 

6607 y=sy, 

6608 mode="lines", 

6609 line=dict( 

6610 color=saccade_color, width=saccade_width, dash=saccade_style 

6611 ), 

6612 name=display_name, 

6613 legendgroup=display_name, 

6614 showlegend=False, 

6615 hoverinfo="skip", 

6616 ) 

6617 ) 

6618 if show_saccade_arrows and len(trial_fix) > 1: 

6619 amx, amy, aang, aseg = _saccade_arrow_rows(trial_fix, "x", "y", arch_frac) 

6620 if amx and class_series is not None: 

6621 mask = _arrow_class_mask(trial_fix, class_series, keep, aseg) 

6622 amx = [v for v, m in zip(amx, mask) if m] 

6623 amy = [v for v, m in zip(amy, mask) if m] 

6624 aang = [v for v, m in zip(aang, mask) if m] 

6625 if amx: 

6626 _add( 

6627 go.Scatter( 

6628 x=amx, 

6629 y=amy, 

6630 mode="markers", 

6631 marker=dict( 

6632 symbol="arrow", 

6633 size=12, 

6634 angle=aang, 

6635 angleref="up", 

6636 color=saccade_color, 

6637 line=dict(width=0), 

6638 ), 

6639 legendgroup=display_name, 

6640 showlegend=False, 

6641 hoverinfo="skip", 

6642 ) 

6643 ) 

6644 

6645 if not show_fixations: 

6646 return 

6647 

6648 # Only the categories doing something: the rail always sends all four, so 

6649 # an untouched set must cost nothing (the classification scans the boxes). 

6650 flags = { 

6651 cat: spec 

6652 for cat, spec in (fixation_flags or {}).items() 

6653 if isinstance(spec, dict) and str(spec.get("mode") or "Off") != "Off" 

6654 } 

6655 if flags: 

6656 trial_fix = _discard_flagged_fixations(trial_fix, words_for_flags, flags) 

6657 if trial_fix.empty: 

6658 return 

6659 sizes = _compute_marker_sizes( 

6660 trial_fix["duration_ms"], 

6661 style["marker_size_range"], 

6662 **(duration_scale or {}), 

6663 ) 

6664 # Metric colouring ("Color fixations by") when a numeric column is chosen: 

6665 # colour the FILL by the metric (shared colorscale/range across both 

6666 # scanpaths) and keep the per-scanpath flat colour as the marker OUTLINE so 

6667 # A/B stay distinguishable. Otherwise the fill is the flat per-scanpath colour. 

6668 metric_color = bool( 

6669 color_by 

6670 and color_by != "line" 

6671 and color_by in trial_fix.columns 

6672 and pd.api.types.is_numeric_dtype(trial_fix[color_by]) 

6673 ) 

6674 symbol = _marker_symbol(fixation_symbol) 

6675 if metric_color: 

6676 marker = dict( 

6677 size=sizes, 

6678 symbol=symbol, 

6679 color=trial_fix[color_by], 

6680 colorscale=colorscale, 

6681 cmin=color_range[0] if color_range else None, 

6682 cmax=color_range[1] if color_range else None, 

6683 showscale=bool(show_colorbar), 

6684 # VIZ-23: the same styled colorbar the static figure builds, so the 

6685 # orientation / tick-angle / tick-size controls reach Compare too. 

6686 colorbar=_colorbar_dict(_column_title(color_by), **(colorbar_style or {})) 

6687 if show_colorbar 

6688 else None, 

6689 line=dict(color=fix_color, width=1.4), 

6690 ) 

6691 elif category_colors is not None: 

6692 # The discrete counterpart: the shared category colour fills, the 

6693 # per-scanpath colour outlines — the same A/B cue as a numeric metric. 

6694 marker = dict( 

6695 size=sizes, 

6696 symbol=symbol, 

6697 color=trial_fix["_category_color"].tolist(), 

6698 line=dict(color=fix_color, width=1.4), 

6699 ) 

6700 else: 

6701 marker = dict( 

6702 size=sizes, 

6703 symbol=symbol, 

6704 color=fix_color, 

6705 line=dict(color=FIX_MARKER_OUTLINE, width=0.5), 

6706 ) 

6707 # Per-scanpath marker alpha (VIZ-6): always set it (even 1.0) so the control 

6708 # overrides Plotly's ~0.7 default for variable-size scatter markers. 

6709 opacity = style.get("opacity", 1.0) 

6710 marker["opacity"] = float(opacity if opacity is not None else 1.0) 

6711 if style.get("hollow"): 

6712 marker = _make_hollow(marker) 

6713 order_font = dict(font_settings) 

6714 order_font["color"] = fix_color 

6715 if order_font_size is not None: 

6716 order_font["size"] = order_font_size 

6717 # VIZ-26 hover fields (the "Hover fields" multiselect under 👁️ Fixations) 

6718 # reach the comparison figure too — it used to hard-code Order/Time/Duration 

6719 # regardless of that setting, which is what made the control look inert 

6720 # while comparing. Same fallback list the static + animation builders use 

6721 # when nothing is explicitly chosen, so the three render paths agree. 

6722 hover_fields = ( 

6723 ["order_in_trial", "duration_ms", "word_id"] 

6724 if fixation_hover_fields is None 

6725 else list(fixation_hover_fields) 

6726 ) 

6727 customdata, hovertemplate = _hover_payload( 

6728 trial_fix, hover_fields, fixation=True, words=trial_words 

6729 ) 

6730 glyph = FIXATION_GLYPH_SYMBOLS.get(fixation_symbol or "") 

6731 # The trace's own legend swatch would mislead under category colours (it 

6732 # shows the first fixation's category) and cannot draw a glyph (♥), so in 

6733 # either case a separate entry names the scanpath. 

6734 own_entry = category_colors is not None or bool(glyph) 

6735 if glyph: 

6736 # VIZ-15: ♥ as text, as on the static figure (`_glyph_scatter_traces`); 

6737 # the index labels and a numeric colour bar get traces of their own. 

6738 for trace in _glyph_scatter_traces( 

6739 trial_fix["x"], 

6740 trial_fix["y"], 

6741 marker, 

6742 glyph, 

6743 name=display_name, 

6744 legendgroup=display_name, 

6745 showlegend=False, 

6746 hovertemplate=f"{display_name}<br>{hovertemplate}", 

6747 customdata=customdata, 

6748 ): 

6749 _add(trace) 

6750 if show_order: 

6751 _add( 

6752 go.Scatter( 

6753 x=trial_fix["x"], 

6754 y=trial_fix["y"], 

6755 mode="text", 

6756 text=trial_fix["order_in_trial"], 

6757 textposition="top center", 

6758 textfont=order_font, 

6759 name=f"{display_name} · index", 

6760 legendgroup=display_name, 

6761 showlegend=False, 

6762 hoverinfo="skip", 

6763 ) 

6764 ) 

6765 bar = _glyph_colorbar_trace(marker, trial_fix[color_by] if metric_color else ()) 

6766 if bar is not None: 

6767 _add(bar) 

6768 else: 

6769 _add( 

6770 go.Scatter( 

6771 x=trial_fix["x"], 

6772 y=trial_fix["y"], 

6773 mode="markers+text" if show_order else "markers", 

6774 marker=marker, 

6775 name=display_name, 

6776 legendgroup=display_name, 

6777 showlegend=show_legend and not own_entry, 

6778 text=trial_fix["order_in_trial"] if show_order else None, 

6779 textposition="top center", 

6780 textfont=order_font, 

6781 hovertemplate=f"{display_name}<br>{hovertemplate}", 

6782 customdata=customdata, 

6783 ) 

6784 ) 

6785 if show_legend and own_entry: 

6786 coloured = metric_color or category_colors is not None 

6787 _add( 

6788 go.Scatter( 

6789 x=[None], 

6790 y=[None], 

6791 mode="markers", 

6792 marker=dict( 

6793 size=10, 

6794 symbol=symbol, 

6795 color="#ffffff" if coloured else fix_color, 

6796 line=dict( 

6797 color=fix_color if coloured else FIX_MARKER_OUTLINE, 

6798 width=2 if coloured else 0.5, 

6799 ), 

6800 ), 

6801 name=display_name, 

6802 legendgroup=display_name, 

6803 showlegend=True, 

6804 hoverinfo="skip", 

6805 ) 

6806 ) 

6807 if flags: 

6808 overlay = _fixation_flag_masks(trial_fix, words_for_flags, flags) 

6809 for cat in _FIX_FLAG_CATEGORIES: 

6810 spec = flags.get(cat, {}) 

6811 if spec.get("mode") != "Highlight" or cat not in overlay: 

6812 continue 

6813 hits = trial_fix[overlay[cat]] 

6814 if hits.empty: 

6815 continue 

6816 name = _FIX_FLAG_LABELS[cat] 

6817 _add( 

6818 go.Scatter( 

6819 x=hits["x"], 

6820 y=hits["y"], 

6821 mode="markers", 

6822 marker=dict( 

6823 symbol=spec.get("symbol") or "x", 

6824 size=13, 

6825 color=spec.get("color") or OUT_OF_TEXT_COLOR, 

6826 line=dict(color=fix_color, width=1.5), 

6827 ), 

6828 name=f"{display_name} · {name}", 

6829 legendgroup=display_name, 

6830 showlegend=show_legend, 

6831 hovertemplate=( 

6832 f"{display_name} · {name} fixation<br>" 

6833 "x %{x:.0f}, y %{y:.0f}<extra></extra>" 

6834 ), 

6835 ) 

6836 ) 

6837 

6838 

6839def _comparison_metric_colorbar( 

6840 fixations: pd.DataFrame, color_by: str | None, show_colorbars: bool 

6841) -> bool: 

6842 """Whether a comparison figure will actually draw a metric colorbar. 

6843 

6844 Same test :func:`_add_comparison_fixation_trace` makes per trace, hoisted so 

6845 the layout can reserve room for a horizontal bar below the plot (VIZ-23). 

6846 """ 

6847 return bool( 

6848 show_colorbars 

6849 and color_by 

6850 and color_by != "line" 

6851 and color_by in fixations.columns 

6852 and pd.api.types.is_numeric_dtype(fixations[color_by]) 

6853 ) 

6854 

6855 

6856def _word_id_keys(values: pd.Series) -> pd.Series: 

6857 """Word ids as join keys that mean the same thing on both frames (CMP-7). 

6858 

6859 The words table and the fixations table routinely disagree on dtype: word 

6860 boxes carry an integer ``word_id`` while a fixation's is a float, because it 

6861 is NaN wherever the fixation landed outside every box. A plain ``str()`` then 

6862 yields ``"7"`` on one side and ``"7.0"`` on the other, so a keyed join 

6863 silently matches nothing — which is exactly how the comparison heatmap came 

6864 out empty. Whole numbers lose the decimal tail here; anything non-numeric 

6865 keeps its stripped string, so datasets with string word ids still join. 

6866 """ 

6867 numeric = pd.to_numeric(values, errors="coerce") 

6868 integral = numeric.notna() & (numeric % 1 == 0) 

6869 text = values.astype(str).str.strip() 

6870 if not integral.any(): 

6871 return text 

6872 return text.mask(integral, numeric.where(integral, 0).astype("int64").astype(str)) 

6873 

6874 

6875def _comparison_word_heatmap_data( 

6876 trial_specs: Sequence[dict], 

6877 *, 

6878 metric: str, 

6879 heatmap_range: tuple[float, float] | None, 

6880 heatmap_norm: str, 

6881) -> tuple[list[dict[str, float]], float, float, str]: 

6882 """Per-trial word values and one shared transformed colour range (CMP-7).""" 

6883 value_maps: list[dict[str, float]] = [] 

6884 all_values: list[float] = [] 

6885 duration_weighted = metric == "duration_ms" 

6886 for spec in trial_specs: 

6887 fixations = spec["trial_fix"] 

6888 if fixations.empty or "word_id" not in fixations.columns: 

6889 values: dict[str, float] = {} 

6890 else: 

6891 valid = fixations[fixations["word_id"].notna()].copy() 

6892 keys = _word_id_keys(valid["word_id"]) 

6893 if duration_weighted and "duration_ms" in valid.columns: 

6894 grouped = ( 

6895 pd.to_numeric(valid["duration_ms"], errors="coerce") 

6896 .groupby(keys) 

6897 .sum() 

6898 ) 

6899 else: 

6900 grouped = valid.groupby(keys).size() 

6901 values = {str(key): float(value) for key, value in grouped.items()} 

6902 value_maps.append(values) 

6903 all_values.extend(value for value in values.values() if value > 0) 

6904 if heatmap_range is not None: 

6905 raw_min, raw_max = map(float, heatmap_range) 

6906 elif all_values: 

6907 # Auto starts at 0, as the single-trial word heatmap does. 

6908 raw_min, raw_max = 0.0, max(all_values) 

6909 else: 

6910 raw_min, raw_max = 0.0, 1.0 

6911 z_min = float(_apply_heatmap_norm(raw_min, heatmap_norm)) 

6912 z_max = float(_apply_heatmap_norm(raw_max, heatmap_norm)) 

6913 if z_max <= z_min: 

6914 z_max = z_min + 1.0 

6915 title = _WORD_DWELL_TITLE if duration_weighted else "Fixation count" 

6916 return value_maps, z_min, z_max, title 

6917 

6918 

6919def _comparison_heatmap_shapes( 

6920 words: pd.DataFrame, 

6921 values: dict[str, float], 

6922 *, 

6923 heatmap_colorscale: str, 

6924 heatmap_norm: str, 

6925 z_min: float, 

6926 z_max: float, 

6927 half: str | None = None, 

6928 xref: str | None = None, 

6929 yref: str | None = None, 

6930) -> list[dict]: 

6931 """Tint full word boxes or their left/right half on a shared scale.""" 

6932 if words.empty or not values: 

6933 return [] 

6934 from plotly.colors import sample_colorscale 

6935 

6936 from .measures import word_box_bounds 

6937 

6938 shapes: list[dict] = [] 

6939 z_span = max(z_max - z_min, 1e-9) 

6940 if "word_id" not in words.columns: 

6941 return [] 

6942 keys = _word_id_keys(words["word_id"]) 

6943 for key, (x0, y0, x1, y1) in zip(keys, zip(*word_box_bounds(words))): 

6944 value = values.get(key, 0.0) 

6945 if value <= 0: 

6946 continue 

6947 midpoint = (x0 + x1) / 2.0 

6948 if half == "left": 

6949 x1 = midpoint 

6950 elif half == "right": 

6951 x0 = midpoint 

6952 transformed = float(_apply_heatmap_norm(value, heatmap_norm)) 

6953 position = max(0.0, min(1.0, (transformed - z_min) / z_span)) 

6954 shape = dict( 

6955 type="rect", 

6956 x0=x0, 

6957 y0=y0, 

6958 x1=x1, 

6959 y1=y1, 

6960 line=dict(width=0), 

6961 fillcolor=sample_colorscale(heatmap_colorscale, [position])[0], 

6962 opacity=0.55, 

6963 layer="below", 

6964 name=_shape_layer_tag("heatmap"), 

6965 ) 

6966 if xref is not None: 

6967 shape["xref"] = xref 

6968 if yref is not None: 

6969 shape["yref"] = yref 

6970 shapes.append(shape) 

6971 return shapes 

6972 

6973 

6974def _comparison_heatmap_colorbar_trace( 

6975 *, 

6976 colorscale: str, 

6977 z_min: float, 

6978 z_max: float, 

6979 title: str, 

6980 heatmap_norm: str, 

6981 colorbar_style: dict, 

6982) -> go.Scatter: 

6983 return go.Scatter( 

6984 x=[None], 

6985 y=[None], 

6986 mode="markers", 

6987 marker=dict( 

6988 colorscale=colorscale, 

6989 showscale=True, 

6990 cmin=z_min, 

6991 cmax=z_max, 

6992 colorbar=_colorbar_dict( 

6993 _heatmap_title(title, heatmap_norm), **colorbar_style 

6994 ), 

6995 ), 

6996 showlegend=False, 

6997 hoverinfo="skip", 

6998 name="comparison heatmap colorbar", 

6999 ) 

7000 

7001 

7002def _comparison_heatmap_colorbar_traces( 

7003 trial_specs: Sequence[dict], 

7004 *, 

7005 z_min: float, 

7006 z_max: float, 

7007 title: str, 

7008 heatmap_norm: str, 

7009 colorbar_style: dict, 

7010 overlay: bool, 

7011) -> list[go.Scatter]: 

7012 """The comparison heatmap's colour bar(s): one while A and B share a colour 

7013 scale, else one per scanpath on the same range (`_arrange_colorbars` sets 

7014 them side by side). The overlay's titles say which half is whose.""" 

7015 scales = [spec["heatmap_colorscale"] for spec in trial_specs] 

7016 sides = ("left A", "right B") if overlay else ("A", "B") 

7017 if len(set(scales)) == 1: 

7018 named = [(scales[0], " · A left half, B right" if overlay else "")] 

7019 else: 

7020 named = [(scale, f" · {side}") for scale, side in zip(scales, sides)] 

7021 return [ 

7022 _comparison_heatmap_colorbar_trace( 

7023 colorscale=scale, 

7024 z_min=z_min, 

7025 z_max=z_max, 

7026 title=title + suffix, 

7027 heatmap_norm=heatmap_norm, 

7028 colorbar_style=colorbar_style, 

7029 ) 

7030 for scale, suffix in named 

7031 ] 

7032 

7033 

7034def _make_split_comparison_figure( 

7035 words: pd.DataFrame, 

7036 fixations: pd.DataFrame, 

7037 trial_a: tuple[str, str], 

7038 trial_b: tuple[str, str], 

7039 *, 

7040 settings: FigureSettings, 

7041 orientation: str, 

7042 styles: tuple[dict, dict] | None = None, 

7043 raw_gaze: pd.DataFrame | None = None, 

7044) -> go.Figure: 

7045 """Two-panel comparison, either horizontal (side-by-side) or vertical (stacked). 

7046 

7047 Each panel computes its **own** axis ranges from its trial's words + 

7048 fixations, which is what lets the two panels sit in different coordinate 

7049 spaces at all (CMP-8). 

7050 

7051 The stimulus-image background (VIZ-4) is added to both panels. It used to be 

7052 the *same* image on the grounds that "the two readings are of the same 

7053 text" — no longer true once B may come from another corpus, so B's panel 

7054 takes ``background_image_b`` (and its size/origin) when given and falls back 

7055 to A's otherwise, which is every same-dataset comparison. 

7056 

7057 Likewise ``canvas_b``: B's screen, defaulting to A's. It feeds B's axis 

7058 ranges *and* its label-fitting pair, so B's reading text stays true-to-scale 

7059 on **B's** monitor rather than being sized against A's. 

7060 """ 

7061 from plotly.subplots import make_subplots 

7062 

7063 canvas_width = settings.canvas_width 

7064 canvas_height = settings.canvas_height 

7065 font_family = settings.font_family 

7066 base_font_size = settings.base_font_size 

7067 show_words = settings.show_words 

7068 show_word_labels = settings.show_word_labels 

7069 trial_labels = settings.trial_labels 

7070 marker_size_range = settings.marker_size_range 

7071 show_fixations = settings.show_fixations 

7072 show_saccades = settings.show_saccades 

7073 show_saccade_arrows = settings.show_saccade_arrows 

7074 show_order = settings.show_order 

7075 show_legend = settings.show_legend 

7076 order_font_size = settings.order_font_size 

7077 color_by = settings.color_by 

7078 color_by_line = settings.color_by_line 

7079 fixation_colorscale = settings.fixation_colorscale 

7080 fixation_color_range = settings.fixation_color_range 

7081 fixation_symbol = settings.fixation_symbol 

7082 # The fixations' colour bar and the heatmap's, each with its own style. 

7083 show_colorbars = settings.show_fixation_colorbar 

7084 show_heatmap_colorbar = settings.show_heatmap_colorbar 

7085 show_heatmap = settings.show_heatmap 

7086 heatmap_metric = settings.heatmap_metric 

7087 heatmap_range = settings.heatmap_range 

7088 heatmap_norm = settings.heatmap_norm 

7089 colorbar_orientation = settings.fixation_colorbar_orientation 

7090 colorbar_tickangle = settings.fixation_colorbar_tickangle 

7091 colorbar_tickfont_size = settings.fixation_colorbar_tickfont_size 

7092 heat_cb_style = dict( 

7093 orientation=settings.heatmap_colorbar_orientation, 

7094 tickangle=settings.heatmap_colorbar_tickangle, 

7095 tickfont_size=settings.heatmap_colorbar_tickfont_size, 

7096 ) 

7097 text_color = settings.text_color 

7098 highlight_column = settings.highlight_column 

7099 highlight_text_color = settings.highlight_text_color 

7100 word_hover_measure = settings.word_hover_measure 

7101 word_hover_fields = settings.word_hover_fields 

7102 fixation_hover_fields = settings.fixation_hover_fields 

7103 background_color = settings.background_color 

7104 line_spacing = settings.line_spacing 

7105 scale_text_to_boxes = settings.scale_text_to_boxes 

7106 background_image = settings.background_image 

7107 background_image_size = settings.background_image_size 

7108 background_image_origin = settings.background_image_origin 

7109 background_image_opacity = settings.background_image_opacity 

7110 fit_to_monitor = settings.fit_to_monitor 

7111 show_coordinate_grid = settings.show_coordinate_grid 

7112 coordinate_grid_spacing = settings.coordinate_grid_spacing 

7113 

7114 font_settings = dict(family=font_family or FONT_FAMILY, size=base_font_size) 

7115 cb_style = dict( 

7116 orientation=colorbar_orientation, 

7117 tickangle=colorbar_tickangle, 

7118 tickfont_size=colorbar_tickfont_size, 

7119 ) 

7120 is_stacked = orientation == "stacked" 

7121 # Per-panel canvas: B falls back to A's, so a same-dataset comparison is 

7122 # byte-identical to the pre-CMP-8 figure. 

7123 canvas_b = settings.canvas_b or (canvas_width, canvas_height) 

7124 panel_canvas = [(canvas_width, canvas_height), (int(canvas_b[0]), int(canvas_b[1]))] 

7125 # Each panel's share of its screen, for the preliminary fit that picks the 

7126 # figure's size. The word labels are *not* sized from it: they are sized 

7127 # from the subplot space each panel finally gets (round-8 review, finding 5). 

7128 panel_widths = [w if is_stacked else w // 2 for w, _ in panel_canvas] 

7129 # B's stimulus page. Inheriting A's is right for a same-dataset pair (two 

7130 # readings of the same text) and *wrong* across datasets — a PoTeC panel with 

7131 # a OneStop page under it is a picture of two different things. So when B has 

7132 # its own screen (`canvas_b`, i.e. a cross-dataset pair) B's panel draws only 

7133 # an image explicitly given for it, and otherwise draws none. 

7134 if settings.background_image_b is not None: 

7135 image_b = ( 

7136 settings.background_image_b, 

7137 settings.background_image_size_b, 

7138 settings.background_image_origin_b, 

7139 ) 

7140 elif settings.canvas_b is None: 

7141 image_b = (background_image, background_image_size, background_image_origin) 

7142 else: 

7143 image_b = (None, None, None) 

7144 panel_images = [ 

7145 (background_image, background_image_size, background_image_origin), 

7146 image_b, 

7147 ] 

7148 

7149 # Shared metric colour range across both panels (when colouring by a metric 

7150 # and no explicit range), so the two scanpaths use one comparable scale. 

7151 metric_range = fixation_color_range 

7152 if ( 

7153 color_by 

7154 and color_by != "line" 

7155 and metric_range is None 

7156 and color_by in fixations.columns 

7157 and pd.api.types.is_numeric_dtype(fixations[color_by]) 

7158 ): 

7159 both = fixations[ 

7160 ( 

7161 (fixations["participant_id"] == trial_a[0]) 

7162 & (fixations["trial_id"] == trial_a[1]) 

7163 ) 

7164 | ( 

7165 (fixations["participant_id"] == trial_b[0]) 

7166 & (fixations["trial_id"] == trial_b[1]) 

7167 ) 

7168 ][color_by] 

7169 if len(both) and pd.notna(both.min()) and pd.notna(both.max()): 

7170 metric_range = (float(both.min()), float(both.max())) 

7171 

7172 trial_specs = [] 

7173 for idx, trial in enumerate([trial_a, trial_b]): 

7174 participant, trial_id = trial 

7175 trial_words = words[ 

7176 (words["participant_id"] == participant) & (words["trial_id"] == trial_id) 

7177 ] 

7178 trial_fix = fixations[ 

7179 (fixations["participant_id"] == participant) 

7180 & (fixations["trial_id"] == trial_id) 

7181 ].sort_values("timestamp_ms") 

7182 display_name = _plotly_literal( 

7183 _resolve_trial_display_name( 

7184 participant, trial_id, trial_words, trial_labels, idx 

7185 ) 

7186 ) 

7187 style = _comparison_scanpath_style( 

7188 idx, 

7189 styles[idx] if styles else None, 

7190 default_marker_size_range=marker_size_range, 

7191 ) 

7192 trial_specs.append( 

7193 dict( 

7194 trial_words=trial_words, 

7195 trial_fix=trial_fix, 

7196 raw_gaze=_comparison_raw_gaze( 

7197 raw_gaze, trial, show=settings.show_raw_gaze 

7198 ), 

7199 display_name=display_name, 

7200 style=style, 

7201 color=style["fix_color"], 

7202 # The word-box outline: the scanpath's own colour unless its 

7203 # style names one (`box_color`). 

7204 box_color=style.get("box_color") or style["fix_color"], 

7205 # Its fill: the figure's unless the style names one. 

7206 box_fill_color=( 

7207 style.get("box_fill_color") or settings.word_box_fill_color 

7208 ), 

7209 # Its raw-gaze samples: the scanpath's own colour unless its 

7210 # style names one (`raw_gaze_color`). 

7211 raw_gaze_color=style.get("raw_gaze_color") or style["fix_color"], 

7212 # Its heatmap's colour scale: the figure's unless the style 

7213 # names one (`heatmap_colorscale`). The range stays shared. 

7214 heatmap_colorscale=( 

7215 style.get("heatmap_colorscale") or settings.heatmap_colorscale 

7216 ), 

7217 ) 

7218 ) 

7219 

7220 heatmap_maps, heatmap_min, heatmap_max, heatmap_title = ( 

7221 _comparison_word_heatmap_data( 

7222 trial_specs, 

7223 metric=heatmap_metric, 

7224 heatmap_range=heatmap_range, 

7225 heatmap_norm=heatmap_norm, 

7226 ) 

7227 if show_heatmap 

7228 else ([], 0.0, 1.0, "") 

7229 ) 

7230 

7231 # A categorical column or colour-by-line: one category→colour mapping for 

7232 # both panels, each reading's lines against its own word boxes. 

7233 category_colors, category_legend = _shared_category_colors( 

7234 [ 

7235 _fixation_category_labels( 

7236 spec["trial_fix"], spec["trial_words"], color_by, color_by_line 

7237 ) 

7238 for spec in trial_specs 

7239 ], 

7240 avoid=[spec["color"] for spec in trial_specs], 

7241 ) 

7242 category_label = "line" if (color_by_line or color_by == "line") else color_by 

7243 

7244 # #374 F26: each panel always says which scanpath it is, "A · …" / "B · …", 

7245 # legend or not — so the top band is reserved for the titles too (BUG-90: 

7246 # without it the upper title was clipped off the canvas). 

7247 subplot_titles = [ 

7248 f"{side} · {spec['display_name']}" for side, spec in zip("AB", trial_specs) 

7249 ] 

7250 

7251 # Each panel's own axis ranges, from its trial's words + fixations (CMP-8). 

7252 panel_ranges = [] 

7253 for idx, spec in enumerate(trial_specs): 

7254 panel_cw, panel_ch = panel_canvas[idx] 

7255 x_range, y_range, *_ = _compute_axis_ranges( 

7256 panel_cw, 

7257 panel_ch, 

7258 (spec["trial_fix"], "x", "y"), 

7259 (spec["raw_gaze"], "x", "y"), 

7260 word_frames=[spec["trial_words"]] if not spec["trial_words"].empty else [], 

7261 fit_to_monitor=fit_to_monitor, 

7262 ) 

7263 panel_ranges.append((x_range, y_range)) 

7264 panel_fits = [ 

7265 _fit_display_size( 

7266 panel_widths[idx], panel_canvas[idx][1], x_range, y_range, spatial_axes=True 

7267 ) 

7268 for idx, (x_range, y_range) in enumerate(panel_ranges) 

7269 ] 

7270 

7271 # The figure's size, chosen before anything is sized against it. Two panels 

7272 # that share a screen reuse the last panel's fit — which keeps every 

7273 # same-dataset figure the size it always was. Two *different* screens can't 

7274 # be reconciled that way: the panels then get the widest / tallest fit of the 

7275 # pair, so neither is clipped. (Which is also why the caption in 

7276 # `tabs._render_comparison_figure` says sizes are not comparable across 

7277 # panels — each panel is true-to-scale on its own monitor.) 

7278 if settings.canvas_b is None: 

7279 panel_w, panel_h = panel_fits[-1] 

7280 else: 

7281 panel_w = max(fit[0] for fit in panel_fits) 

7282 panel_h = max(fit[1] for fit in panel_fits) 

7283 if is_stacked: 

7284 total_width = panel_w 

7285 total_height = panel_h * 2 + 40 

7286 else: # side-by-side 

7287 total_width = panel_w * 2 

7288 total_height = panel_h 

7289 # A colour bar gets its own reserved band — below the panels when horizontal 

7290 # (VIZ-23), to their right when vertical — so the figure grows by it rather 

7291 # than Plotly's automargin shrinking the panels under text already sized for 

7292 # them (the single-trial figure's `_decoration_margins` rule). 

7293 reserves = _colorbar_reserves( 

7294 ( 

7295 _comparison_metric_colorbar(fixations, color_by, show_colorbars), 

7296 colorbar_orientation, 

7297 ), 

7298 ( 

7299 bool(show_heatmap and show_heatmap_colorbar and any(heatmap_maps)), 

7300 heat_cb_style["orientation"], 

7301 ), 

7302 ) 

7303 bottom_px = _COLORBAR_BOTTOM_PX if reserves["colorbar_below"] else 0 

7304 right_px = _COLORBAR_RESERVE_PX if reserves["colorbar_right"] else 0 

7305 grid_left = _GRID_LEFT_RESERVE_PX if show_coordinate_grid else 0 

7306 grid_bottom = _GRID_BOTTOM_RESERVE_PX if show_coordinate_grid else 0 

7307 # The t band was the (now-removed) title; keep a slim band only for the 

7308 # optional legend. 

7309 top_px = _compare_legend_font(base_font_size)["size"] + 14 

7310 figure_width = total_width + grid_left + right_px 

7311 figure_height = total_height + bottom_px + grid_bottom 

7312 plot_area = ( 

7313 max(figure_width - grid_left - right_px, 1), 

7314 max(figure_height - top_px - bottom_px - grid_bottom, 1), 

7315 ) 

7316 if is_stacked: 

7317 fig = make_subplots( 

7318 rows=2, 

7319 cols=1, 

7320 vertical_spacing=0.08, 

7321 subplot_titles=subplot_titles, 

7322 ) 

7323 else: 

7324 fig = make_subplots( 

7325 rows=1, 

7326 cols=2, 

7327 horizontal_spacing=0.04, 

7328 subplot_titles=subplot_titles, 

7329 ) 

7330 

7331 # Each panel's data→screen scale, from the subplot space it actually gets 

7332 # in the final figure: its domain's share of the plot area, then the 

7333 # equal-aspect constraint, which shrinks whichever side has room to spare. 

7334 panel_displays = [] 

7335 for idx, (x_range, y_range) in enumerate(panel_ranges): 

7336 suffix = "" if idx == 0 else str(idx + 1) 

7337 x_domain = fig.layout[f"xaxis{suffix}"].domain 

7338 y_domain = fig.layout[f"yaxis{suffix}"].domain 

7339 scale = _display_scale( 

7340 x_range, 

7341 y_range, 

7342 (x_domain[1] - x_domain[0]) * plot_area[0], 

7343 (y_domain[1] - y_domain[0]) * plot_area[1], 

7344 ) 

7345 panel_displays.append( 

7346 ( 

7347 scale, 

7348 round((x_range[1] - x_range[0]) * scale), 

7349 round((y_range[0] - y_range[1]) * scale), 

7350 ) 

7351 ) 

7352 

7353 all_shapes: list = [] 

7354 for idx, spec in enumerate(trial_specs): 

7355 if is_stacked: 

7356 row, col = idx + 1, 1 

7357 axis_suffix = "" if idx == 0 else str(idx + 1) 

7358 else: 

7359 row, col = 1, idx + 1 

7360 axis_suffix = "" if idx == 0 else str(idx + 1) 

7361 xref = f"x{axis_suffix}" 

7362 yref = f"y{axis_suffix}" 

7363 trial_words = spec["trial_words"] 

7364 trial_fix = spec["trial_fix"] 

7365 x_range, y_range = panel_ranges[idx] 

7366 panel_scale, panel_display_w, panel_display_h = panel_displays[idx] 

7367 

7368 # Stimulus-page background image (VIZ-4/23), one per panel, UNDER every 

7369 # trace — `row`/`col` bind it to this panel's axes. B may carry its own 

7370 # (CMP-8); it falls back to A's when it doesn't. 

7371 panel_image, panel_image_size, panel_image_origin = panel_images[idx] 

7372 _add_background_image( 

7373 fig, 

7374 panel_image, 

7375 panel_image_size, 

7376 panel_image_origin, 

7377 background_image_opacity, 

7378 row=row, 

7379 col=col, 

7380 ) 

7381 

7382 if show_heatmap: 

7383 all_shapes.extend( 

7384 _comparison_heatmap_shapes( 

7385 trial_words, 

7386 heatmap_maps[idx], 

7387 heatmap_colorscale=spec["heatmap_colorscale"], 

7388 heatmap_norm=heatmap_norm, 

7389 z_min=heatmap_min, 

7390 z_max=heatmap_max, 

7391 xref=xref, 

7392 yref=yref, 

7393 ) 

7394 ) 

7395 

7396 if show_words and not trial_words.empty: 

7397 for box in build_word_boxes( 

7398 trial_words, 

7399 color=spec["box_color"], 

7400 fill_color=spec["box_fill_color"], 

7401 fill_opacity=settings.word_box_fill_opacity, 

7402 line_opacity=settings.word_box_line_opacity, 

7403 ): 

7404 box = dict(box) 

7405 box["xref"] = xref 

7406 box["yref"] = yref 

7407 all_shapes.append(box) 

7408 

7409 all_shapes.append( 

7410 dict( 

7411 type="rect", 

7412 xref=xref, 

7413 yref=yref, 

7414 x0=x_range[0], 

7415 y0=y_range[1], 

7416 x1=x_range[1], 

7417 y1=y_range[0], 

7418 line=dict(color="#000000", width=1), 

7419 fillcolor="rgba(0,0,0,0)", 

7420 ) 

7421 ) 

7422 

7423 # Under the scanpath, as on the single-trial figure. 

7424 _add_comparison_raw_gaze_trace( 

7425 fig, 

7426 spec["raw_gaze"], 

7427 spec["display_name"], 

7428 spec["raw_gaze_color"], 

7429 settings, 

7430 row=row, 

7431 col=col, 

7432 ) 

7433 _add_comparison_fixation_trace( 

7434 fig, 

7435 trial_fix, 

7436 spec["display_name"], 

7437 spec["style"], 

7438 font_settings, 

7439 show_fixations=show_fixations, 

7440 show_saccades=show_saccades, 

7441 show_saccade_arrows=show_saccade_arrows, 

7442 show_order=show_order, 

7443 order_font_size=order_font_size, 

7444 show_legend=show_legend, 

7445 color_by=color_by, 

7446 colorscale=fixation_colorscale, 

7447 color_range=metric_range, 

7448 show_colorbar=show_colorbars and idx == 0, 

7449 colorbar_style=cb_style, 

7450 fixation_symbol=fixation_symbol, 

7451 fixation_hover_fields=fixation_hover_fields, 

7452 row=row, 

7453 col=col, 

7454 trial_words=spec["trial_words"], 

7455 **_comparison_filters(spec["style"], settings), 

7456 duration_scale=_settings_size_scale(settings), 

7457 category_colors=category_colors[idx], 

7458 ) 

7459 

7460 if show_word_labels: 

7461 _add_word_label_trace( 

7462 fig, 

7463 trial_words, 

7464 _word_label_font_px( 

7465 trial_words, 

7466 scale=panel_scale, 

7467 line_spacing=line_spacing, 

7468 manual_font_px=base_font_size, 

7469 scale_text_to_boxes=scale_text_to_boxes, 

7470 ), 

7471 font_settings["family"], 

7472 row=row, 

7473 col=col, 

7474 highlight_column=highlight_column, 

7475 text_color=text_color, 

7476 highlight_text_color=highlight_text_color, 

7477 word_hover_measure=word_hover_measure, 

7478 word_hover_fields=word_hover_fields, 

7479 ) 

7480 

7481 xaxis_key = "xaxis" if idx == 0 else f"xaxis{idx + 1}" 

7482 yaxis_key = "yaxis" if idx == 0 else f"yaxis{idx + 1}" 

7483 xaxis = dict( 

7484 showticklabels=False, 

7485 showgrid=False, 

7486 zeroline=False, 

7487 title=None, 

7488 range=x_range, 

7489 constrain="domain", 

7490 ) 

7491 yaxis = dict( 

7492 showticklabels=False, 

7493 showgrid=False, 

7494 zeroline=False, 

7495 title=None, 

7496 range=y_range, 

7497 constrain="domain", 

7498 scaleanchor=xref, 

7499 scaleratio=1, 

7500 ) 

7501 _apply_coordinate_grid_axes( 

7502 xaxis, 

7503 yaxis, 

7504 show=show_coordinate_grid, 

7505 spacing=coordinate_grid_spacing, 

7506 x_range=x_range, 

7507 y_range=y_range, 

7508 rendered_width=panel_display_w, 

7509 rendered_height=panel_display_h, 

7510 ) 

7511 fig.update_layout(**{xaxis_key: xaxis, yaxis_key: yaxis}) 

7512 

7513 if show_heatmap and show_heatmap_colorbar and any(heatmap_maps): 

7514 for trace in _comparison_heatmap_colorbar_traces( 

7515 trial_specs, 

7516 z_min=heatmap_min, 

7517 z_max=heatmap_max, 

7518 title=heatmap_title, 

7519 heatmap_norm=heatmap_norm, 

7520 colorbar_style=heat_cb_style, 

7521 overlay=False, 

7522 ): 

7523 fig.add_trace(trace) 

7524 _add_category_legend(fig, category_legend, category_label or "") 

7525 

7526 fig.update_layout( 

7527 height=figure_height, 

7528 width=figure_width, 

7529 autosize=False, 

7530 margin=dict(l=grid_left, r=right_px, t=top_px, b=bottom_px + grid_bottom), 

7531 legend=dict( 

7532 orientation="h", 

7533 yanchor="bottom", 

7534 y=1.05, 

7535 xanchor="right", 

7536 x=1, 

7537 font=_compare_legend_font(base_font_size, font_family), 

7538 ), 

7539 template="plotly_white", 

7540 plot_bgcolor=background_color, 

7541 paper_bgcolor=background_color, 

7542 font=font_settings, 

7543 shapes=all_shapes, 

7544 ) 

7545 return fig 

7546 

7547 

7548def _compare_stimulus_sides(value: str | None) -> tuple[bool, bool]: 

7549 """``compare_stimulus`` → ``(draw A's stimulus, draw B's)`` (CMP-11). 

7550 

7551 Tolerant of an unrecognised value on purpose: this reads a share-link param 

7552 and a saved config, and drawing both sets of boxes is the honest fallback — 

7553 it shows what is there rather than silently hiding one reading's AOIs. 

7554 """ 

7555 normalized = str(value or "both").strip().lower() 

7556 if normalized == "a": 

7557 return True, False 

7558 if normalized == "b": 

7559 return False, True 

7560 return True, True 

7561 

7562 

7563def _render_comparison_figure( 

7564 words: pd.DataFrame, 

7565 fixations: pd.DataFrame, 

7566 trial_a: tuple[str, str], 

7567 trial_b: tuple[str, str], 

7568 *, 

7569 settings: FigureSettings, 

7570 raw_gaze: pd.DataFrame | None = None, 

7571) -> go.Figure: 

7572 """Two scanpaths on one canvas — overlaid, side by side, or stacked. 

7573 

7574 The shared settings contract keeps marker shape, highlighted text, stimulus 

7575 image, and colorbar styling consistent across all three layouts (VIZ-23). 

7576 """ 

7577 canvas_width = settings.canvas_width 

7578 canvas_height = settings.canvas_height 

7579 font_family = settings.font_family 

7580 base_font_size = settings.base_font_size 

7581 show_words = settings.show_words 

7582 show_word_labels = settings.show_word_labels 

7583 trial_labels = settings.trial_labels 

7584 layout = settings.layout 

7585 marker_size_range = settings.marker_size_range 

7586 style_a = settings.style_a 

7587 style_b = settings.style_b 

7588 show_fixations = settings.show_fixations 

7589 show_saccades = settings.show_saccades 

7590 show_saccade_arrows = settings.show_saccade_arrows 

7591 show_order = settings.show_order 

7592 show_legend = settings.show_legend 

7593 order_font_size = settings.order_font_size 

7594 color_by = settings.color_by 

7595 color_by_line = settings.color_by_line 

7596 fixation_colorscale = settings.fixation_colorscale 

7597 fixation_color_range = settings.fixation_color_range 

7598 fixation_symbol = settings.fixation_symbol 

7599 # The fixations' colour bar and the heatmap's, each with its own style. 

7600 show_colorbars = settings.show_fixation_colorbar 

7601 show_heatmap_colorbar = settings.show_heatmap_colorbar 

7602 show_heatmap = settings.show_heatmap 

7603 heatmap_metric = settings.heatmap_metric 

7604 heatmap_range = settings.heatmap_range 

7605 heatmap_norm = settings.heatmap_norm 

7606 colorbar_orientation = settings.fixation_colorbar_orientation 

7607 colorbar_tickangle = settings.fixation_colorbar_tickangle 

7608 colorbar_tickfont_size = settings.fixation_colorbar_tickfont_size 

7609 heat_cb_style = dict( 

7610 orientation=settings.heatmap_colorbar_orientation, 

7611 tickangle=settings.heatmap_colorbar_tickangle, 

7612 tickfont_size=settings.heatmap_colorbar_tickfont_size, 

7613 ) 

7614 text_color = settings.text_color 

7615 highlight_column = settings.highlight_column 

7616 highlight_text_color = settings.highlight_text_color 

7617 word_hover_measure = settings.word_hover_measure 

7618 word_hover_fields = settings.word_hover_fields 

7619 fixation_hover_fields = settings.fixation_hover_fields 

7620 background_color = settings.background_color 

7621 line_spacing = settings.line_spacing 

7622 scale_text_to_boxes = settings.scale_text_to_boxes 

7623 background_image = settings.background_image 

7624 background_image_size = settings.background_image_size 

7625 background_image_origin = settings.background_image_origin 

7626 background_image_opacity = settings.background_image_opacity 

7627 fit_to_monitor = settings.fit_to_monitor 

7628 show_coordinate_grid = settings.show_coordinate_grid 

7629 coordinate_grid_spacing = settings.coordinate_grid_spacing 

7630 if layout in {"side_by_side", "stacked"}: 

7631 return _make_split_comparison_figure( 

7632 words, 

7633 fixations, 

7634 trial_a, 

7635 trial_b, 

7636 settings=settings, 

7637 orientation=layout, 

7638 styles=(style_a, style_b), 

7639 raw_gaze=raw_gaze, 

7640 ) 

7641 

7642 # Shared metric colour range across BOTH trials (when colouring by a numeric 

7643 # metric and the user didn't pin a range), so the two scanpaths use one 

7644 # comparable scale. 

7645 metric_range = fixation_color_range 

7646 if ( 

7647 color_by 

7648 and color_by != "line" 

7649 and metric_range is None 

7650 and color_by in fixations.columns 

7651 and pd.api.types.is_numeric_dtype(fixations[color_by]) 

7652 ): 

7653 both = fixations[ 

7654 ( 

7655 (fixations["participant_id"] == trial_a[0]) 

7656 & (fixations["trial_id"] == trial_a[1]) 

7657 ) 

7658 | ( 

7659 (fixations["participant_id"] == trial_b[0]) 

7660 & (fixations["trial_id"] == trial_b[1]) 

7661 ) 

7662 ][color_by] 

7663 if len(both) and pd.notna(both.min()) and pd.notna(both.max()): 

7664 metric_range = (float(both.min()), float(both.max())) 

7665 

7666 fig = go.Figure() 

7667 font_settings = dict(family=font_family or FONT_FAMILY, size=base_font_size) 

7668 cb_style = dict( 

7669 orientation=colorbar_orientation, 

7670 tickangle=colorbar_tickangle, 

7671 tickfont_size=colorbar_tickfont_size, 

7672 ) 

7673 overrides = (style_a, style_b) 

7674 # Stimulus-page background image (VIZ-4/23), UNDER both scanpaths. 

7675 _add_background_image( 

7676 fig, 

7677 background_image, 

7678 background_image_size, 

7679 background_image_origin, 

7680 background_image_opacity, 

7681 ) 

7682 

7683 trial_specs = [] 

7684 for idx, trial in enumerate([trial_a, trial_b]): 

7685 participant, trial_id = trial 

7686 trial_words = words[ 

7687 (words["participant_id"] == participant) & (words["trial_id"] == trial_id) 

7688 ] 

7689 trial_fix = fixations[ 

7690 (fixations["participant_id"] == participant) 

7691 & (fixations["trial_id"] == trial_id) 

7692 ].sort_values("timestamp_ms") 

7693 display_name = _plotly_literal( 

7694 _resolve_trial_display_name( 

7695 participant, trial_id, trial_words, trial_labels, idx 

7696 ) 

7697 ) 

7698 style = _comparison_scanpath_style( 

7699 idx, overrides[idx], default_marker_size_range=marker_size_range 

7700 ) 

7701 trial_specs.append( 

7702 dict( 

7703 trial_words=trial_words, 

7704 trial_fix=trial_fix, 

7705 raw_gaze=_comparison_raw_gaze( 

7706 raw_gaze, trial, show=settings.show_raw_gaze 

7707 ), 

7708 display_name=display_name, 

7709 style=style, 

7710 color=style["fix_color"], 

7711 # The word-box outline: the scanpath's own colour unless its 

7712 # style names one (`box_color`). 

7713 box_color=style.get("box_color") or style["fix_color"], 

7714 # Its fill: the figure's unless the style names one. 

7715 box_fill_color=( 

7716 style.get("box_fill_color") or settings.word_box_fill_color 

7717 ), 

7718 # Its raw-gaze samples: the scanpath's own colour unless its 

7719 # style names one (`raw_gaze_color`). 

7720 raw_gaze_color=style.get("raw_gaze_color") or style["fix_color"], 

7721 # Its heatmap's colour scale: the figure's unless the style 

7722 # names one (`heatmap_colorscale`). The range stays shared. 

7723 heatmap_colorscale=( 

7724 style.get("heatmap_colorscale") or settings.heatmap_colorscale 

7725 ), 

7726 ) 

7727 ) 

7728 

7729 heatmap_maps, heatmap_min, heatmap_max, heatmap_title = ( 

7730 _comparison_word_heatmap_data( 

7731 trial_specs, 

7732 metric=heatmap_metric, 

7733 heatmap_range=heatmap_range, 

7734 heatmap_norm=heatmap_norm, 

7735 ) 

7736 if show_heatmap 

7737 else ([], 0.0, 1.0, "") 

7738 ) 

7739 

7740 if show_heatmap: 

7741 reference_words = next( 

7742 ( 

7743 spec["trial_words"] 

7744 for spec in trial_specs 

7745 if not spec["trial_words"].empty 

7746 ), 

7747 pd.DataFrame(), 

7748 ) 

7749 existing = list(fig.layout.shapes) if fig.layout.shapes else [] 

7750 for index, half in enumerate(("left", "right")): 

7751 existing.extend( 

7752 _comparison_heatmap_shapes( 

7753 reference_words, 

7754 heatmap_maps[index], 

7755 heatmap_colorscale=trial_specs[index]["heatmap_colorscale"], 

7756 heatmap_norm=heatmap_norm, 

7757 z_min=heatmap_min, 

7758 z_max=heatmap_max, 

7759 half=half, 

7760 ) 

7761 ) 

7762 fig.update_layout(shapes=existing) 

7763 if show_heatmap_colorbar and any(heatmap_maps): 

7764 for trace in _comparison_heatmap_colorbar_traces( 

7765 trial_specs, 

7766 z_min=heatmap_min, 

7767 z_max=heatmap_max, 

7768 title=heatmap_title, 

7769 heatmap_norm=heatmap_norm, 

7770 colorbar_style=heat_cb_style, 

7771 overlay=True, 

7772 ): 

7773 fig.add_trace(trace) 

7774 

7775 x_range, y_range, *_ = _compute_axis_ranges( 

7776 canvas_width, 

7777 canvas_height, 

7778 *((spec["trial_fix"], "x", "y") for spec in trial_specs), 

7779 *((spec["raw_gaze"], "x", "y") for spec in trial_specs), 

7780 word_frames=[ 

7781 spec["trial_words"] for spec in trial_specs if not spec["trial_words"].empty 

7782 ], 

7783 fit_to_monitor=fit_to_monitor, 

7784 ) 

7785 

7786 # Both trials are overlaid on one shared canvas, so one display scale sizes 

7787 # every word label true-to-scale (geometry is identical across the readings). 

7788 fitted_w, fitted_h = _fit_display_size( 

7789 canvas_width, canvas_height, x_range, y_range, spatial_axes=True 

7790 ) 

7791 overlay_scale = _display_scale(x_range, y_range, fitted_w, fitted_h) 

7792 

7793 # A categorical column or colour-by-line: one category→colour mapping for 

7794 # both readings (see the split layouts). 

7795 category_colors, category_legend = _shared_category_colors( 

7796 [ 

7797 _fixation_category_labels( 

7798 spec["trial_fix"], spec["trial_words"], color_by, color_by_line 

7799 ) 

7800 for spec in trial_specs 

7801 ], 

7802 avoid=[spec["color"] for spec in trial_specs], 

7803 ) 

7804 category_label = "line" if (color_by_line or color_by == "line") else color_by 

7805 legend_on = show_legend or bool(category_legend) 

7806 

7807 draws_stimulus = _compare_stimulus_sides(settings.compare_stimulus) 

7808 # Both readings' samples before either scanpath, so neither cloud covers the 

7809 # other reading's fixations. 

7810 for spec in trial_specs: 

7811 _add_comparison_raw_gaze_trace( 

7812 fig, 

7813 spec["raw_gaze"], 

7814 spec["display_name"], 

7815 spec["raw_gaze_color"], 

7816 settings, 

7817 ) 

7818 for _idx, spec in enumerate(trial_specs): 

7819 _add_comparison_fixation_trace( 

7820 fig, 

7821 spec["trial_fix"], 

7822 spec["display_name"], 

7823 spec["style"], 

7824 font_settings, 

7825 show_fixations=show_fixations, 

7826 show_saccades=show_saccades, 

7827 show_saccade_arrows=show_saccade_arrows, 

7828 show_order=show_order, 

7829 order_font_size=order_font_size, 

7830 show_legend=show_legend, 

7831 color_by=color_by, 

7832 colorscale=fixation_colorscale, 

7833 color_range=metric_range, 

7834 # One shared colorbar (on the first scanpath only) for the metric. 

7835 show_colorbar=show_colorbars and _idx == 0, 

7836 colorbar_style=cb_style, 

7837 fixation_symbol=fixation_symbol, 

7838 fixation_hover_fields=fixation_hover_fields, 

7839 trial_words=spec["trial_words"], 

7840 **_comparison_filters(spec["style"], settings), 

7841 duration_scale=_settings_size_scale(settings), 

7842 category_colors=category_colors[_idx], 

7843 ) 

7844 if show_words and draws_stimulus[_idx]: 

7845 existing = list(fig.layout.shapes) if fig.layout.shapes else [] 

7846 fig.update_layout( 

7847 shapes=existing 

7848 + build_word_boxes( 

7849 spec["trial_words"], 

7850 color=spec["box_color"], 

7851 fill_color=spec["box_fill_color"], 

7852 fill_opacity=settings.word_box_fill_opacity, 

7853 line_opacity=settings.word_box_line_opacity, 

7854 ) 

7855 ) 

7856 if show_word_labels and draws_stimulus[_idx]: 

7857 _add_word_label_trace( 

7858 fig, 

7859 spec["trial_words"], 

7860 _word_label_font_px( 

7861 spec["trial_words"], 

7862 scale=overlay_scale, 

7863 line_spacing=line_spacing, 

7864 manual_font_px=base_font_size, 

7865 scale_text_to_boxes=scale_text_to_boxes, 

7866 ), 

7867 font_settings["family"], 

7868 highlight_column=highlight_column, 

7869 text_color=text_color, 

7870 highlight_text_color=highlight_text_color, 

7871 word_hover_measure=word_hover_measure, 

7872 word_hover_fields=word_hover_fields, 

7873 ) 

7874 

7875 _add_category_legend(fig, category_legend, category_label or "") 

7876 

7877 shapes = list(fig.layout.shapes) if fig.layout.shapes else [] 

7878 shapes.append( 

7879 dict( 

7880 type="rect", 

7881 x0=x_range[0], 

7882 y0=y_range[1], 

7883 x1=x_range[1], 

7884 y1=y_range[0], 

7885 line=dict(color="#000000", width=1), 

7886 fillcolor="rgba(0,0,0,0)", 

7887 ) 

7888 ) 

7889 

7890 # fitted_w / fitted_h were computed up front (so the label scale matched). 

7891 # The title + top A/B legend get reserved space above the plot so they don't 

7892 # shrink the equal-aspect plot region (same fix as make_scanpath_figure). With 

7893 # the legend hidden (CMP-2 default) a slimmer band still fits the title. 

7894 # The top band is now only needed for the optional A/B legend (the "Overlay 

7895 # comparison" title was removed); reclaim it fully when the legend is hidden. 

7896 top_px = _OVERLAY_TOP_PX if legend_on else 0 

7897 # A horizontal colorbar (VIZ-23) sits below the plot, so reserve a band for 

7898 # it; a vertical one keeps today's layout (it hangs off the right edge). 

7899 bottom_px = ( 

7900 _COLORBAR_BOTTOM_PX 

7901 if _colorbar_reserves( 

7902 ( 

7903 _comparison_metric_colorbar(fixations, color_by, show_colorbars), 

7904 colorbar_orientation, 

7905 ), 

7906 ( 

7907 bool(show_heatmap and show_heatmap_colorbar and any(heatmap_maps)), 

7908 heat_cb_style["orientation"], 

7909 ), 

7910 )["colorbar_below"] 

7911 else 0 

7912 ) 

7913 grid_left = _GRID_LEFT_RESERVE_PX if show_coordinate_grid else 0 

7914 grid_bottom = _GRID_BOTTOM_RESERVE_PX if show_coordinate_grid else 0 

7915 xaxis = dict( 

7916 showticklabels=False, 

7917 showgrid=False, 

7918 zeroline=False, 

7919 title=None, 

7920 range=x_range, 

7921 constrain="domain", 

7922 automargin=False, 

7923 ) 

7924 yaxis = dict( 

7925 showticklabels=False, 

7926 showgrid=False, 

7927 zeroline=False, 

7928 title=None, 

7929 range=y_range, 

7930 constrain="domain", 

7931 scaleanchor="x", 

7932 scaleratio=1, 

7933 automargin=False, 

7934 ) 

7935 _apply_coordinate_grid_axes( 

7936 xaxis, 

7937 yaxis, 

7938 show=show_coordinate_grid, 

7939 spacing=coordinate_grid_spacing, 

7940 x_range=x_range, 

7941 y_range=y_range, 

7942 rendered_width=fitted_w, 

7943 rendered_height=fitted_h, 

7944 ) 

7945 fig.update_layout( 

7946 height=fitted_h + top_px + bottom_px + grid_bottom, 

7947 width=fitted_w + grid_left, 

7948 autosize=False, 

7949 showlegend=legend_on, 

7950 margin=dict(l=grid_left, r=0, t=top_px, b=bottom_px + grid_bottom), 

7951 xaxis=xaxis, 

7952 yaxis=yaxis, 

7953 legend=dict( 

7954 orientation="h", 

7955 yanchor="bottom", 

7956 y=1.02, 

7957 xanchor="right", 

7958 x=1, 

7959 font=_compare_legend_font(base_font_size, font_family), 

7960 ), 

7961 template="plotly_white", 

7962 plot_bgcolor=background_color, 

7963 paper_bgcolor=background_color, 

7964 font=font_settings, 

7965 shapes=shapes, 

7966 ) 

7967 return fig 

7968 

7969 

7970# ============================================================================= 

7971# Line figures: metric convergence, trial-index trend 

7972# ============================================================================= 

7973 

7974 

7975def make_metric_convergence_figure( 

7976 series: dict, 

7977 *, 

7978 x_title: str, 

7979 y_title: str, 

7980 title: str, 

7981 canvas_width: int, 

7982 base_font_size: int, 

7983 font_family: str, 

7984 height: int = 340, 

7985 y_range: tuple[float, float] = (0.0, 1.02), 

7986 highlight_x_range: tuple[float, float] | None = None, 

7987) -> go.Figure: 

7988 """Line chart of a metric (one line per model) over a cumulative x axis. 

7989 

7990 ``series`` maps a model name to ``(xs, ys)``. Used by the Multiple Comparison 

7991 tab to show how each model's NLD vs. the real scanpath evolves as more of the 

7992 reading is included — either by cumulative fixation index or by elapsed time. 

7993 Optionally shades ``highlight_x_range`` (e.g. the selected fixation window). 

7994 """ 

7995 fig = go.Figure() 

7996 font_settings = dict(family=font_family or FONT_FAMILY, size=base_font_size) 

7997 has_data = False 

7998 for i, (name, xy) in enumerate(series.items()): 

7999 xs, ys = xy 

8000 if not len(xs): 

8001 continue 

8002 has_data = True 

8003 color = _QUALITATIVE_PALETTE[i % len(_QUALITATIVE_PALETTE)] 

8004 fig.add_trace( 

8005 go.Scatter( 

8006 x=list(xs), 

8007 y=list(ys), 

8008 mode="lines+markers", 

8009 name=str(name), 

8010 line=dict(color=color, width=2), 

8011 marker=dict(size=4, color=color), 

8012 hovertemplate=( 

8013 f"{name}<br>{x_title}: %{{x}}<br>{y_title}: %{{y:.3f}}" 

8014 "<extra></extra>" 

8015 ), 

8016 ) 

8017 ) 

8018 if has_data and highlight_x_range is not None: 

8019 lo, hi = highlight_x_range 

8020 if hi > lo: 

8021 fig.add_vrect(x0=lo, x1=hi, fillcolor="#6c757d", opacity=0.10, line_width=0) 

8022 fig.update_layout( 

8023 height=height, 

8024 width=canvas_width, 

8025 autosize=False, 

8026 margin=dict(l=55, r=10, t=40, b=45), 

8027 template="plotly_white", 

8028 font=font_settings, 

8029 xaxis=dict(title=x_title), 

8030 yaxis=dict(title=y_title, range=list(y_range)), 

8031 legend=dict(orientation="h", yanchor="bottom", y=1.02, xanchor="right", x=1), 

8032 title=title, 

8033 ) 

8034 if not has_data: 

8035 fig.add_annotation( 

8036 text="No data", showarrow=False, x=0.5, y=0.5, xref="paper", yref="paper" 

8037 ) 

8038 return fig 

8039 

8040 

8041def gap_runs(xs) -> list[slice]: 

8042 """Split ``xs`` (sorted) into runs with no gap: a new run starts where two 

8043 whole-number ``xs`` are more than 1 apart — a trial the filters left out 

8044 (#374 F35). Non-integer ``xs`` are one run.""" 

8045 xs = list(xs) 

8046 try: 

8047 whole = all(float(x).is_integer() for x in xs) 

8048 except (TypeError, ValueError): 

8049 whole = False 

8050 if not whole: 

8051 return [slice(0, len(xs))] 

8052 cuts = [i for i in range(1, len(xs)) if float(xs[i]) - float(xs[i - 1]) > 1] 

8053 bounds = [0, *cuts, len(xs)] 

8054 return [slice(a, b) for a, b in itertools.pairwise(bounds)] 

8055 

8056 

8057def break_at_gaps(xs, ys) -> tuple[list, list]: 

8058 """``xs``/``ys`` with a ``None`` at every gap (see :func:`gap_runs`), so a 

8059 Plotly line stops there instead of joining across it.""" 

8060 xs, ys = list(xs), list(ys) 

8061 out_x: list = [] 

8062 out_y: list = [] 

8063 for i, run in enumerate(gap_runs(xs)): 

8064 if i: 

8065 out_x.append(None) 

8066 out_y.append(None) 

8067 out_x += xs[run] 

8068 out_y += ys[run] 

8069 return out_x, out_y 

8070 

8071 

8072def make_trend_figure( 

8073 df: pd.DataFrame, 

8074 *, 

8075 x_col: str, 

8076 y_label: str, 

8077 title: str, 

8078 canvas_width: int, 

8079 base_font_size: int, 

8080 font_family: str, 

8081 height: int = 340, 

8082 x_label: str | None = None, 

8083 break_gaps: bool = False, 

8084) -> go.Figure: 

8085 """Line+marker trend of ``value`` vs ``x_col`` with a ±SEM shaded band. 

8086 

8087 ``df`` has columns ``[x_col, "value", "sem"]`` (see 

8088 ``aggregation.metric_by_trial_index``). Used by the Per reader and Groups 

8089 subtabs for the trial-index trend. ``x_label`` titles the x axis — the 

8090 caller's name for what ``x_col`` holds (AN-9's frame calls it ``x``, which 

8091 is no title); without one, ``x_col`` humanized. ``break_gaps`` stops the 

8092 line and band at a missing whole-number ``x`` (a filtered-out trial). 

8093 """ 

8094 if x_label is None: 

8095 x_label = x_col.replace("_", " ").capitalize() 

8096 fig = go.Figure() 

8097 font_settings = dict(family=font_family or FONT_FAMILY, size=base_font_size) 

8098 if df is None or df.empty: 

8099 fig.update_layout( 

8100 template="plotly_white", 

8101 font=font_settings, 

8102 title=f"{title} (no data)", 

8103 height=height, 

8104 ) 

8105 return fig 

8106 xs = df[x_col].to_numpy() 

8107 ys = df["value"].to_numpy() 

8108 sem = df["sem"].to_numpy() if "sem" in df.columns else np.zeros(len(xs)) 

8109 runs = gap_runs(xs) if break_gaps else [slice(0, len(xs))] 

8110 # One closed band per unbroken run, `None`-separated. 

8111 band_x: list = [] 

8112 band_y: list = [] 

8113 for i, run in enumerate(runs): 

8114 if i: 

8115 band_x.append(None) 

8116 band_y.append(None) 

8117 band_x += [*xs[run], *xs[run][::-1]] 

8118 band_y += [*(ys[run] + sem[run]), *(ys[run] - sem[run])[::-1]] 

8119 line_x, line_y = break_at_gaps(xs, ys) if break_gaps else (xs, ys) 

8120 # ±SEM band (drawn first so the line sits on top). 

8121 fig.add_trace( 

8122 go.Scatter( 

8123 x=band_x, 

8124 y=band_y, 

8125 fill="toself", 

8126 fillcolor="rgba(31,119,180,0.15)", 

8127 line=dict(width=0), 

8128 hoverinfo="skip", 

8129 showlegend=False, 

8130 name="±SEM", 

8131 ) 

8132 ) 

8133 fig.add_trace( 

8134 go.Scatter( 

8135 x=line_x, 

8136 y=line_y, 

8137 mode="lines+markers", 

8138 line=dict(color=COMPARISON_PALETTE[0], width=2), 

8139 marker=dict(size=5, color=COMPARISON_PALETTE[0]), 

8140 name=y_label, 

8141 hovertemplate=f"{x_label}: %{{x}}<br>{y_label}: %{{y:.1f}}<extra></extra>", 

8142 ) 

8143 ) 

8144 fig.update_layout( 

8145 height=height, 

8146 width=canvas_width, 

8147 autosize=False, 

8148 margin=dict(l=60, r=10, t=40, b=45), 

8149 template="plotly_white", 

8150 font=font_settings, 

8151 xaxis=dict(title=x_label), 

8152 yaxis=dict(title=y_label), 

8153 title=title, 

8154 showlegend=False, 

8155 ) 

8156 return fig 

8157 

8158 

8159# ============================================================================= 

8160# Analysis section figures (AN-1 … AN-22) 

8161# ============================================================================= 

8162# 

8163# Builders for the question-oriented Corpus Analysis subtabs. Each takes a tidy 

8164# frame from ``aggregation.py`` plus the usual ``canvas_width`` / ``base_font_size`` 

8165# / ``font_family`` and returns a ``go.Figure``. Empty input → a "(no data)" 

8166# placeholder, matching ``make_trend_figure``. 

8167 

8168_DIVERGING_COLORSCALE = "RdBu" 

8169 

8170 

8171def _hex_to_rgba(color: str, alpha: float) -> str: 

8172 """``#rrggbb`` → ``rgba(r,g,b,alpha)`` for translucent spread bands. Passes 

8173 through non-hex colours (already ``rgb(...)`` / named) by wrapping opacity in 

8174 is impossible, so it returns a sensible grey fallback for those.""" 

8175 c = str(color).lstrip("#") 

8176 if len(c) == 6: 

8177 try: 

8178 r, g, b = (int(c[i : i + 2], 16) for i in (0, 2, 4)) 

8179 return f"rgba({r},{g},{b},{alpha})" 

8180 except ValueError: 

8181 pass 

8182 return f"rgba(120,120,120,{alpha})" 

8183 

8184 

8185def _no_data_figure(title: str, *, font_family: str, base_font_size: int, height=340): 

8186 fig = go.Figure() 

8187 fig.update_layout( 

8188 template="plotly_white", 

8189 font=dict(family=font_family or FONT_FAMILY, size=base_font_size), 

8190 title=f"{title} (no data)", 

8191 height=height, 

8192 ) 

8193 return fig 

8194 

8195 

8196def make_small_multiples_figure( 

8197 per_reader: pd.DataFrame, 

8198 *, 

8199 measure_label: str, 

8200 canvas_width: int, 

8201 base_font_size: int, 

8202 font_family: str, 

8203 cohort: pd.DataFrame | None = None, 

8204 aggregate: str = "mean", 

8205 max_panels: int = 12, 

8206 panel_height: int = 110, 

8207) -> go.Figure: 

8208 """Stacked per-reader word profiles — one panel per participant (AN-1). 

8209 

8210 ``per_reader`` is tidy ``[participant_id, word_id, value]`` (see 

8211 ``aggregation.per_reader_word_measure``); panels share the X (reading order). 

8212 ``cohort`` (``[word_id, value]``) draws a faint cohort-mean overlay in each 

8213 panel. Caps at ``max_panels`` readers and titles the overflow (no silent cut). 

8214 """ 

8215 from plotly.subplots import make_subplots 

8216 

8217 if per_reader is None or per_reader.empty: 

8218 return _no_data_figure( 

8219 f"{measure_label} per participant", 

8220 font_family=font_family, 

8221 base_font_size=base_font_size, 

8222 ) 

8223 readers = list(pd.unique(per_reader["participant_id"])) 

8224 n_total = len(readers) 

8225 readers = readers[:max_panels] 

8226 n = len(readers) 

8227 # Each panel's title (the reader id) sits in the gap above it, so the gap 

8228 # is sized in pixels from the font rather than as a fixed fraction of the 

8229 # figure: a fraction shrinks with few panels and the title lands on the 

8230 # panel above's lowest tick row. 

8231 title_gap_px = round(base_font_size * 2.2) + 6 

8232 plot_px = panel_height * n + title_gap_px * (n - 1) 

8233 fig = make_subplots( 

8234 rows=n, 

8235 cols=1, 

8236 shared_xaxes=True, 

8237 vertical_spacing=title_gap_px / plot_px if n > 1 else 0.0, 

8238 subplot_titles=[str(r) for r in readers], 

8239 ) 

8240 cohort_xy = None 

8241 if cohort is not None and not cohort.empty: 

8242 c = cohort.sort_values("word_id") 

8243 cohort_xy = (c["word_id"].to_numpy(), c["value"].to_numpy()) 

8244 for i, reader in enumerate(readers, start=1): 

8245 sub = per_reader[per_reader["participant_id"] == reader].sort_values("word_id") 

8246 if cohort_xy is not None: 

8247 fig.add_trace( 

8248 go.Scatter( 

8249 x=cohort_xy[0], 

8250 y=cohort_xy[1], 

8251 mode="lines", 

8252 line=dict(color="rgba(120,120,120,0.45)", width=1.2, dash="dot"), 

8253 name=f"Cohort {aggregate}", 

8254 showlegend=(i == 1), 

8255 hoverinfo="skip", 

8256 ), 

8257 row=i, 

8258 col=1, 

8259 ) 

8260 fig.add_trace( 

8261 go.Scatter( 

8262 x=sub["word_id"].to_numpy(), 

8263 y=sub["value"].to_numpy(), 

8264 mode="lines+markers", 

8265 line=dict(color=COMPARISON_PALETTE[0], width=1.5), 

8266 marker=dict(size=3, color=COMPARISON_PALETTE[0]), 

8267 name=str(reader), 

8268 showlegend=False, 

8269 customdata=sub["word_text"].to_numpy() if "word_text" in sub else None, 

8270 hovertemplate=( 

8271 "word %{x}" 

8272 + (" %{customdata}" if "word_text" in sub else "") 

8273 + f"<br>{measure_label}: %{{y:.3~g}}<extra></extra>" 

8274 ), 

8275 ), 

8276 row=i, 

8277 col=1, 

8278 ) 

8279 title = f"{measure_label} per participant (word profile)" 

8280 if n_total > n: 

8281 title += f" — showing {n} of {n_total} participants" 

8282 margin_top = 50 + title_gap_px 

8283 margin_bottom = 40 

8284 fig.update_layout( 

8285 height=plot_px + margin_top + margin_bottom, 

8286 width=canvas_width, 

8287 autosize=False, 

8288 margin=dict(l=55, r=10, t=margin_top, b=margin_bottom), 

8289 template="plotly_white", 

8290 font=dict(family=font_family or FONT_FAMILY, size=base_font_size), 

8291 title=title, 

8292 legend=dict(orientation="h", yanchor="bottom", y=1.0, xanchor="right", x=1), 

8293 ) 

8294 fig.update_xaxes(title_text="Word (reading order)", row=n, col=1) 

8295 return fig 

8296 

8297 

8298def make_word_matrix_heatmap( 

8299 df: pd.DataFrame, 

8300 *, 

8301 row_col: str, 

8302 measure_label: str, 

8303 canvas_width: int, 

8304 base_font_size: int, 

8305 font_family: str, 

8306 value_col: str = "value", 

8307 colorscale: str = DEFAULT_HEATMAP_COLORSCALE, 

8308 row_order: Iterable | None = None, 

8309 height: int | None = None, 

8310 row_label: str | None = None, 

8311) -> go.Figure: 

8312 """Word × {reader|group} heatmap (AN-2, AN-22). 

8313 

8314 ``df`` is long ``[row_col, word_id, value_col]``; rows become Y, ``word_id`` 

8315 X, ``value_col`` the color. ``row_order`` pins the row order (e.g. Group A 

8316 above Group B). Bright columns = universally hard words; bright rows = a 

8317 uniformly slow reader. ``row_label`` names the rows (the dataset's own name 

8318 for ``row_col``, DATA-66); without one, ``row_col`` humanized. 

8319 """ 

8320 if row_label is None: 

8321 row_label = _humanize_column(row_col) 

8322 if df is None or df.empty: 

8323 return _no_data_figure( 

8324 f"{measure_label} by {row_label} × word", 

8325 font_family=font_family, 

8326 base_font_size=base_font_size, 

8327 ) 

8328 matrix = df.pivot_table( 

8329 index=row_col, columns="word_id", values=value_col, aggfunc="mean" 

8330 ) 

8331 if row_order is not None: 

8332 keep = [r for r in row_order if r in matrix.index] 

8333 matrix = matrix.reindex(keep) 

8334 fig = go.Figure( 

8335 go.Heatmap( 

8336 z=matrix.to_numpy(), 

8337 x=[int(c) if float(c).is_integer() else c for c in matrix.columns], 

8338 y=[str(r) for r in matrix.index], 

8339 colorscale=colorscale, 

8340 colorbar=dict(title=measure_label), 

8341 hovertemplate="word %{x}<br>%{y}<br>" 

8342 + measure_label 

8343 + ": %{z:.1f}<extra></extra>", 

8344 ) 

8345 ) 

8346 n_rows = max(len(matrix.index), 1) 

8347 fig.update_layout( 

8348 height=height or min(900, max(220, 26 * n_rows + 120)), 

8349 width=canvas_width, 

8350 autosize=False, 

8351 margin=dict(l=120, r=10, t=50, b=45), 

8352 template="plotly_white", 

8353 font=dict(family=font_family or FONT_FAMILY, size=base_font_size), 

8354 title=f"{measure_label} — {row_label} × word", 

8355 xaxis=dict(title="Word (reading order)"), 

8356 yaxis=dict(title=row_label, autorange="reversed"), 

8357 ) 

8358 return fig 

8359 

8360 

8361def make_word_profile_figure( 

8362 profiles: dict, 

8363 *, 

8364 measure_label: str, 

8365 canvas_width: int, 

8366 base_font_size: int, 

8367 font_family: str, 

8368 spread_label: str = "SD", 

8369 colors: Sequence[str] | None = None, 

8370 height: int = 380, 

8371 aggregate: str = "mean", 

8372) -> go.Figure: 

8373 """Cohort word profile(s): mean line + shaded spread band (AN-3 / AN-15). 

8374 

8375 ``profiles`` maps a label → ``[word_id, value, lo, hi]`` (see 

8376 ``aggregation.cohort_word_profile``). One entry draws the "average reader of 

8377 this text" with uncertainty; several overlay (e.g. two groups). 

8378 """ 

8379 entries = [ 

8380 (str(k), v) 

8381 for k, v in (profiles or {}).items() 

8382 if v is not None and not v.empty 

8383 ] 

8384 if not entries: 

8385 return _no_data_figure( 

8386 f"{measure_label} word profile", 

8387 font_family=font_family, 

8388 base_font_size=base_font_size, 

8389 height=height, 

8390 ) 

8391 fig = go.Figure() 

8392 single = len(entries) == 1 

8393 for i, (label, prof) in enumerate(entries): 

8394 prof = prof.sort_values("word_id") 

8395 xs = prof["word_id"].to_numpy() 

8396 ys = prof["value"].to_numpy() 

8397 color_choices = tuple(colors or COMPARISON_PALETTE) 

8398 color = color_choices[i % len(color_choices)] 

8399 rgba = _hex_to_rgba(color, 0.15) 

8400 if {"lo", "hi"} <= set(prof.columns): 

8401 lo = prof["lo"].to_numpy() 

8402 hi = prof["hi"].to_numpy() 

8403 fig.add_trace( 

8404 go.Scatter( 

8405 x=np.concatenate([xs, xs[::-1]]), 

8406 y=np.concatenate([hi, lo[::-1]]), 

8407 fill="toself", 

8408 fillcolor=rgba, 

8409 line=dict(width=0), 

8410 hoverinfo="skip", 

8411 showlegend=False, 

8412 name=f"{label} {spread_label}", 

8413 ) 

8414 ) 

8415 fig.add_trace( 

8416 go.Scatter( 

8417 x=xs, 

8418 y=ys, 

8419 mode="lines+markers", 

8420 line=dict(color=color, width=2), 

8421 marker=dict(size=4, color=color), 

8422 name=label, 

8423 showlegend=not single, 

8424 customdata=prof["word_text"].to_numpy() 

8425 if "word_text" in prof 

8426 else None, 

8427 hovertemplate=( 

8428 "word %{x}" 

8429 + (" %{customdata}" if "word_text" in prof else "") 

8430 + f"<br>{measure_label}: %{{y:.3~g}}<extra></extra>" 

8431 ), 

8432 ) 

8433 ) 

8434 fig.update_layout( 

8435 height=height, 

8436 width=canvas_width, 

8437 autosize=False, 

8438 margin=dict(l=60, r=10, t=45, b=45), 

8439 template="plotly_white", 

8440 font=dict(family=font_family or FONT_FAMILY, size=base_font_size), 

8441 title=f"{measure_label} by word — cohort {aggregate}, {spread_label} band", 

8442 xaxis=dict(title="Word (reading order)"), 

8443 yaxis=dict(title=measure_label), 

8444 showlegend=not single, 

8445 legend=dict(orientation="h", yanchor="bottom", y=1.02, xanchor="right", x=1), 

8446 ) 

8447 return fig 

8448 

8449 

8450def make_feature_scatter_figure( 

8451 df: pd.DataFrame, 

8452 *, 

8453 measure_label: str, 

8454 feature_label: str, 

8455 categorical: bool, 

8456 canvas_width: int, 

8457 base_font_size: int, 

8458 font_family: str, 

8459 height: int = 400, 

8460) -> go.Figure: 

8461 """Per-word measure vs a bundled linguistic feature (AN-5). 

8462 

8463 Numeric feature → scatter + OLS trend line (with Pearson r in the title); 

8464 categorical feature (POS) → one box per category. 

8465 """ 

8466 if df is None or df.empty or "feature" not in df.columns: 

8467 return _no_data_figure( 

8468 f"{measure_label} vs {feature_label}", 

8469 font_family=font_family, 

8470 base_font_size=base_font_size, 

8471 height=height, 

8472 ) 

8473 font_settings = dict(family=font_family or FONT_FAMILY, size=base_font_size) 

8474 fig = go.Figure() 

8475 if categorical: 

8476 cats = sorted(df["feature"].dropna().astype(str).unique()) 

8477 for i, cat in enumerate(cats): 

8478 vals = df.loc[df["feature"].astype(str) == cat, "value"].dropna().to_numpy() 

8479 if vals.size: 

8480 fig.add_trace( 

8481 go.Box( 

8482 y=vals, 

8483 name=cat, 

8484 boxpoints="outliers", 

8485 marker_color=_QUALITATIVE_PALETTE[ 

8486 i % len(_QUALITATIVE_PALETTE) 

8487 ], 

8488 ) 

8489 ) 

8490 fig.update_layout( 

8491 xaxis=dict(title=feature_label), 

8492 yaxis=dict(title=measure_label), 

8493 showlegend=False, 

8494 ) 

8495 title = f"{measure_label} by {feature_label}" 

8496 else: 

8497 x = pd.to_numeric(df["feature"], errors="coerce").to_numpy() 

8498 y = pd.to_numeric(df["value"], errors="coerce").to_numpy() 

8499 ok = ~(np.isnan(x) | np.isnan(y)) 

8500 x, y = x[ok], y[ok] 

8501 fig.add_trace( 

8502 go.Scatter( 

8503 x=x, 

8504 y=y, 

8505 mode="markers", 

8506 marker=dict(size=6, color=COMPARISON_PALETTE[0], opacity=0.6), 

8507 name="words", 

8508 customdata=df.loc[ok, "word_text"].to_numpy() 

8509 if "word_text" in df 

8510 else None, 

8511 hovertemplate=( 

8512 f"{feature_label}: %{{x:.2f}}<br>{measure_label}: %{{y:.1f}}" 

8513 + ("<br>%{customdata}" if "word_text" in df else "") 

8514 + "<extra></extra>" 

8515 ), 

8516 ) 

8517 ) 

8518 r_txt = "" 

8519 if x.size >= 2 and np.std(x) > 0: 

8520 slope, intercept = np.polyfit(x, y, 1) 

8521 xs = np.array([x.min(), x.max()]) 

8522 fig.add_trace( 

8523 go.Scatter( 

8524 x=xs, 

8525 y=slope * xs + intercept, 

8526 mode="lines", 

8527 line=dict(color=TRENDLINE_COLOR, width=2, dash="dash"), 

8528 name="trend", 

8529 hoverinfo="skip", 

8530 ) 

8531 ) 

8532 r = float(np.corrcoef(x, y)[0, 1]) 

8533 r_txt = f" (r = {r:.2f}, n = {x.size})" 

8534 fig.update_layout( 

8535 xaxis=dict(title=feature_label), 

8536 yaxis=dict(title=measure_label), 

8537 showlegend=False, 

8538 ) 

8539 title = f"{measure_label} vs {feature_label}{r_txt}" 

8540 fig.update_layout( 

8541 height=height, 

8542 width=canvas_width, 

8543 autosize=False, 

8544 margin=dict(l=60, r=10, t=45, b=50), 

8545 template="plotly_white", 

8546 font=font_settings, 

8547 title=title, 

8548 ) 

8549 return fig 

8550 

8551 

8552def make_word_rate_figure( 

8553 df: pd.DataFrame, 

8554 *, 

8555 canvas_width: int, 

8556 base_font_size: int, 

8557 font_family: str, 

8558 height: int = 360, 

8559) -> go.Figure: 

8560 """Skip / regression-in rate per word — lollipop bars (AN-6).""" 

8561 if df is None or df.empty: 

8562 return _no_data_figure( 

8563 "Skip / regression-in rate per word", 

8564 font_family=font_family, 

8565 base_font_size=base_font_size, 

8566 height=height, 

8567 ) 

8568 df = df.sort_values("word_id") 

8569 xs = df["word_id"].to_numpy() 

8570 fig = go.Figure() 

8571 series = [ 

8572 ("Skip rate", "skip_rate", COMPARISON_PALETTE[0]), 

8573 ("Regression-in rate", "regression_in_rate", COMPARISON_PALETTE[1]), 

8574 ] 

8575 for name, col, color in series: 

8576 if col not in df.columns: 

8577 continue 

8578 ys = pd.to_numeric(df[col], errors="coerce").to_numpy() 

8579 fig.add_trace( 

8580 go.Bar( 

8581 x=xs, 

8582 y=ys, 

8583 name=name, 

8584 marker_color=color, 

8585 opacity=0.8, 

8586 hovertemplate="word %{x}<br>" + name + ": %{y:.0%}<extra></extra>", 

8587 ) 

8588 ) 

8589 fig.update_layout( 

8590 height=height, 

8591 width=canvas_width, 

8592 autosize=False, 

8593 margin=dict(l=55, r=10, t=45, b=45), 

8594 template="plotly_white", 

8595 font=dict(family=font_family or FONT_FAMILY, size=base_font_size), 

8596 title="Skip / regression-in rate per word", 

8597 xaxis=dict(title="Word (reading order)"), 

8598 yaxis=dict(title="Rate", tickformat=".0%"), 

8599 barmode="group", 

8600 legend=dict(orientation="h", yanchor="bottom", y=1.02, xanchor="right", x=1), 

8601 ) 

8602 return fig 

8603 

8604 

8605def make_distribution_figure( 

8606 groups: dict, 

8607 *, 

8608 metric_label: str, 

8609 canvas_width: int, 

8610 base_font_size: int, 

8611 font_family: str, 

8612 kind: str = "violin", 

8613 colors: Sequence[str] | None = None, 

8614 height: int = 380, 

8615) -> go.Figure: 

8616 """Overlaid metric distributions — one violin/box per group (AN-7/14/18).""" 

8617 arrays = [ 

8618 (str(name), np.asarray(arr, dtype="float64")) 

8619 for name, arr in (groups or {}).items() 

8620 if arr is not None and len(arr) 

8621 ] 

8622 if not arrays: 

8623 return _no_data_figure( 

8624 f"{metric_label} distribution", 

8625 font_family=font_family, 

8626 base_font_size=base_font_size, 

8627 height=height, 

8628 ) 

8629 fig = go.Figure() 

8630 for i, (name, arr) in enumerate(arrays): 

8631 color_choices = tuple(colors or _QUALITATIVE_PALETTE) 

8632 color = color_choices[i % len(color_choices)] 

8633 if kind == "box": 

8634 fig.add_trace( 

8635 go.Box( 

8636 y=arr, 

8637 name=name, 

8638 marker_color=color, 

8639 boxmean=True, 

8640 boxpoints="outliers", 

8641 ) 

8642 ) 

8643 else: 

8644 fig.add_trace( 

8645 go.Violin( 

8646 y=arr, 

8647 name=name, 

8648 line_color=color, 

8649 opacity=0.7, 

8650 box_visible=True, 

8651 meanline_visible=True, 

8652 points=False, 

8653 ) 

8654 ) 

8655 fig.update_layout( 

8656 height=height, 

8657 width=canvas_width, 

8658 autosize=False, 

8659 margin=dict(l=60, r=10, t=45, b=40), 

8660 template="plotly_white", 

8661 font=dict(family=font_family or FONT_FAMILY, size=base_font_size), 

8662 title=f"{metric_label} distribution", 

8663 yaxis=dict(title=metric_label), 

8664 showlegend=False, 

8665 ) 

8666 return fig 

8667 

8668 

8669def make_density_scatter_figure( 

8670 df: pd.DataFrame, 

8671 *, 

8672 x_col: str, 

8673 y_col: str, 

8674 x_label: str, 

8675 y_label: str, 

8676 canvas_width: int, 

8677 base_font_size: int, 

8678 font_family: str, 

8679 height: int = 420, 

8680) -> go.Figure: 

8681 """2D density of two per-fixation measures — the oculomotor scatter (AN-10).""" 

8682 if df is None or df.empty or not {x_col, y_col} <= set(df.columns): 

8683 return _no_data_figure( 

8684 f"{y_label} vs {x_label}", 

8685 font_family=font_family, 

8686 base_font_size=base_font_size, 

8687 height=height, 

8688 ) 

8689 x = pd.to_numeric(df[x_col], errors="coerce").to_numpy() 

8690 y = pd.to_numeric(df[y_col], errors="coerce").to_numpy() 

8691 ok = ~(np.isnan(x) | np.isnan(y)) 

8692 x, y = x[ok], y[ok] 

8693 fig = go.Figure( 

8694 go.Histogram2d( 

8695 x=x, 

8696 y=y, 

8697 colorscale=DEFAULT_HEATMAP_COLORSCALE, 

8698 nbinsx=40, 

8699 nbinsy=40, 

8700 colorbar=dict(title="Fixations"), 

8701 hovertemplate=f"{x_label}: %{{x}}<br>{y_label}: %{{y}}<br>count: %{{z}}<extra></extra>", 

8702 ) 

8703 ) 

8704 fig.update_layout( 

8705 height=height, 

8706 width=canvas_width, 

8707 autosize=False, 

8708 margin=dict(l=60, r=10, t=45, b=50), 

8709 template="plotly_white", 

8710 font=dict(family=font_family or FONT_FAMILY, size=base_font_size), 

8711 title=f"{y_label} vs {x_label} (n = {x.size})", 

8712 xaxis=dict(title=x_label), 

8713 yaxis=dict(title=y_label), 

8714 ) 

8715 return fig 

8716 

8717 

8718def make_progression_figure( 

8719 df: pd.DataFrame, 

8720 *, 

8721 canvas_width: int, 

8722 base_font_size: int, 

8723 font_family: str, 

8724 height: int = 380, 

8725) -> go.Figure: 

8726 """Progressive vs regressive saccade counts per trial + regression share (AN-11).""" 

8727 from plotly.subplots import make_subplots 

8728 

8729 if df is None or df.empty: 

8730 return _no_data_figure( 

8731 "Progressive vs regressive saccades", 

8732 font_family=font_family, 

8733 base_font_size=base_font_size, 

8734 height=height, 

8735 ) 

8736 df = df.copy() 

8737 labels = [str(t) for t in df["trial_id"].to_numpy()] 

8738 fig = make_subplots(specs=[[{"secondary_y": True}]]) 

8739 fig.add_trace( 

8740 go.Bar( 

8741 x=labels, 

8742 y=df["progressive"].to_numpy(), 

8743 name="Progressive", 

8744 marker_color=COMPARISON_PALETTE[0], 

8745 ), 

8746 secondary_y=False, 

8747 ) 

8748 fig.add_trace( 

8749 go.Bar( 

8750 x=labels, 

8751 y=df["regressive"].to_numpy(), 

8752 name="Regressive", 

8753 marker_color=COMPARISON_PALETTE[1], 

8754 ), 

8755 secondary_y=False, 

8756 ) 

8757 if "regression_share" in df.columns: 

8758 fig.add_trace( 

8759 go.Scatter( 

8760 x=labels, 

8761 y=df["regression_share"].to_numpy(), 

8762 name="Regression share", 

8763 mode="lines+markers", 

8764 line=dict(color="#555", width=2), 

8765 marker=dict(size=5), 

8766 ), 

8767 secondary_y=True, 

8768 ) 

8769 fig.update_layout( 

8770 height=height, 

8771 width=canvas_width, 

8772 autosize=False, 

8773 margin=dict(l=55, r=55, t=45, b=80), 

8774 template="plotly_white", 

8775 font=dict(family=font_family or FONT_FAMILY, size=base_font_size), 

8776 title="Progressive vs regressive saccades per trial", 

8777 barmode="stack", 

8778 legend=dict(orientation="h", yanchor="bottom", y=1.02, xanchor="right", x=1), 

8779 ) 

8780 fig.update_xaxes(title_text="Trial", tickangle=-40) 

8781 fig.update_yaxes(title_text="Saccade count", secondary_y=False) 

8782 fig.update_yaxes( 

8783 title_text="Regression share", tickformat=".0%", secondary_y=True, range=[0, 1] 

8784 ) 

8785 return fig 

8786 

8787 

8788def make_paired_bars_figure( 

8789 df: pd.DataFrame, 

8790 *, 

8791 canvas_width: int, 

8792 base_font_size: int, 

8793 font_family: str, 

8794 height: int = 380, 

8795 aggregate: str = "mean", 

8796) -> go.Figure: 

8797 """Side-by-side group bars per measure (AN-20), no error bars (#374). 

8798 

8799 ``df`` is ``[measure, group, value, …]`` (see 

8800 ``aggregation.paired_group_summary``); ``aggregate`` names what ``value`` 

8801 is, for the title. One subplot per measure so differing units keep their 

8802 own scale. 

8803 """ 

8804 from plotly.subplots import make_subplots 

8805 

8806 if df is None or df.empty: 

8807 return _no_data_figure( 

8808 f"Group {aggregate}s", 

8809 font_family=font_family, 

8810 base_font_size=base_font_size, 

8811 height=height, 

8812 ) 

8813 measures = list(dict.fromkeys(df["measure"])) 

8814 groups = list(dict.fromkeys(df["group"])) 

8815 fig = make_subplots( 

8816 rows=1, cols=len(measures), subplot_titles=measures, horizontal_spacing=0.08 

8817 ) 

8818 for gi, group in enumerate(groups): 

8819 color = COMPARISON_PALETTE[gi % len(COMPARISON_PALETTE)] 

8820 for mi, measure in enumerate(measures, start=1): 

8821 sub = df[(df["measure"] == measure) & (df["group"] == group)] 

8822 if sub.empty: 

8823 continue 

8824 row = sub.iloc[0] 

8825 fig.add_trace( 

8826 go.Bar( 

8827 x=[group], 

8828 y=[row["value"]], 

8829 name=group, 

8830 marker_color=color, 

8831 legendgroup=group, 

8832 showlegend=(mi == 1), 

8833 hovertemplate=f"{group}<br>{measure}: %{{y:.2f}}<extra></extra>", 

8834 ), 

8835 row=1, 

8836 col=mi, 

8837 ) 

8838 fig.update_layout( 

8839 height=height, 

8840 width=canvas_width, 

8841 autosize=False, 

8842 margin=dict(l=55, r=10, t=55, b=40), 

8843 template="plotly_white", 

8844 font=dict(family=font_family or FONT_FAMILY, size=base_font_size), 

8845 title=f"Group {aggregate}s per measure", 

8846 legend=dict(orientation="h", yanchor="bottom", y=1.04, xanchor="right", x=1), 

8847 barmode="group", 

8848 ) 

8849 return fig 

8850 

8851 

8852def make_landing_curve_figure( 

8853 values: np.ndarray, 

8854 *, 

8855 canvas_width: int, 

8856 base_font_size: int, 

8857 font_family: str, 

8858 as_fraction: bool = True, 

8859 height: int = 360, 

8860) -> go.Figure: 

8861 """Preferred-viewing-location curve — landing-position histogram (AN-12). 

8862 

8863 ``values`` come from ``aggregation.landing_positions``: fractions of the 

8864 experiment's word box, unclipped (BUG-83), so a landing assigned from beside 

8865 the box shows as a bar outside 0–1 instead of a spike on the edge. 

8866 """ 

8867 arr = np.asarray(values, dtype="float64") 

8868 arr = arr[~np.isnan(arr)] 

8869 if arr.size == 0: 

8870 return _no_data_figure( 

8871 "Landing position within words", 

8872 font_family=font_family, 

8873 base_font_size=base_font_size, 

8874 height=height, 

8875 ) 

8876 nbins = 20 if as_fraction else 30 

8877 fig = go.Figure( 

8878 go.Histogram( 

8879 x=arr, 

8880 nbinsx=nbins, 

8881 marker_color=COMPARISON_PALETTE[0], 

8882 marker_line=dict(color="white", width=0.4), 

8883 hovertemplate="landing %{x}<br>count: %{y}<extra></extra>", 

8884 ) 

8885 ) 

8886 x_title = ( 

8887 "Landing position within the word's interest area (0 = start, 1 = end)" 

8888 if as_fraction 

8889 else "Landing distance from word start (px)" 

8890 ) 

8891 fig.update_layout( 

8892 height=height, 

8893 width=canvas_width, 

8894 autosize=False, 

8895 margin=dict(l=55, r=10, t=45, b=50), 

8896 template="plotly_white", 

8897 font=dict(family=font_family or FONT_FAMILY, size=base_font_size), 

8898 title=f"Landing-position curve (n = {arr.size})", 

8899 xaxis=dict(title=x_title), 

8900 yaxis=dict(title="Count"), 

8901 ) 

8902 return fig 

8903 

8904 

8905def make_difference_profile_figure( 

8906 df: pd.DataFrame, 

8907 *, 

8908 measure_label: str, 

8909 label_a: str = "Group A", 

8910 label_b: str = "Group B", 

8911 canvas_width: int, 

8912 base_font_size: int, 

8913 font_family: str, 

8914 colors: Sequence[str] | None = None, 

8915 height: int = 380, 

8916) -> go.Figure: 

8917 """Per-word A−B difference profile, diverging color + zero line (AN-19).""" 

8918 if df is None or df.empty or "diff" not in df.columns: 

8919 return _no_data_figure( 

8920 f"{measure_label} difference by word", 

8921 font_family=font_family, 

8922 base_font_size=base_font_size, 

8923 height=height, 

8924 ) 

8925 df = df.sort_values("word_id") 

8926 xs = df["word_id"].to_numpy() 

8927 diffs = pd.to_numeric(df["diff"], errors="coerce").to_numpy() 

8928 vmax = np.nanmax(np.abs(diffs)) if np.isfinite(diffs).any() else 1.0 

8929 vmax = vmax if vmax > 0 else 1.0 

8930 difference_colors = tuple(colors or (COMPARISON_PALETTE[1], COMPARISON_PALETTE[0])) 

8931 colorscale = [ 

8932 [0.0, difference_colors[1 % len(difference_colors)]], 

8933 [0.5, "#f7f7f7"], 

8934 [1.0, difference_colors[0]], 

8935 ] 

8936 fig = go.Figure( 

8937 go.Bar( 

8938 x=xs, 

8939 y=diffs, 

8940 marker=dict( 

8941 color=diffs, 

8942 colorscale=colorscale, 

8943 cmin=-vmax, 

8944 cmax=vmax, 

8945 colorbar=dict(title=f"{label_a} − {label_b}"), 

8946 ), 

8947 customdata=df["word_text"].to_numpy() if "word_text" in df else None, 

8948 hovertemplate=( 

8949 "word %{x}" 

8950 + (" %{customdata}" if "word_text" in df else "") 

8951 + f"<br>Δ {measure_label}: %{{y:.3~g}}<extra></extra>" 

8952 ), 

8953 ) 

8954 ) 

8955 fig.add_hline(y=0, line=dict(color="#333", width=1)) 

8956 fig.update_layout( 

8957 height=height, 

8958 width=canvas_width, 

8959 autosize=False, 

8960 margin=dict(l=60, r=10, t=50, b=45), 

8961 template="plotly_white", 

8962 font=dict(family=font_family or FONT_FAMILY, size=base_font_size), 

8963 title=f"{measure_label}: {label_a} − {label_b} by word", 

8964 xaxis=dict(title="Word (reading order)"), 

8965 yaxis=dict(title=f"Δ {measure_label}"), 

8966 ) 

8967 return fig 

8968 

8969 

8970def _setting_names(excluded: Iterable[str]) -> tuple[str, ...]: 

8971 excluded = set(excluded) 

8972 return tuple( 

8973 field.name for field in fields(FigureSettings) if field.name not in excluded 

8974 ) 

8975 

8976 

8977def _resolve_figure_settings( 

8978 settings: FigureSettings | Mapping[str, Any] | None, 

8979 overrides: Mapping[str, Any], 

8980 *, 

8981 legacy_defaults: Mapping[str, Any] | None = None, 

8982) -> FigureSettings: 

8983 """Resolve a settings object while preserving old builder-only defaults.""" 

8984 if settings is None and legacy_defaults: 

8985 return FigureSettings.from_mapping({**legacy_defaults, **overrides}) 

8986 return FigureSettings.from_mapping(settings, **dict(overrides)) 

8987 

8988 

8989STATIC_FIGURE_OPTIONS = _setting_names( 

8990 { 

8991 "playback_speed", 

8992 "label_a", 

8993 "label_b", 

8994 "show_legend", 

8995 "autoplay", 

8996 "anim_grid_step_ms", 

8997 "anim_max_frames", 

8998 "trial_labels", 

8999 "layout", 

9000 "style_a", 

9001 "style_b", 

9002 # CMP-8 §4 — B-side geometry, read only by the split comparison layouts. 

9003 "canvas_b", 

9004 "background_image_b", 

9005 "background_image_size_b", 

9006 "background_image_origin_b", 

9007 # CMP-11 — the static builder draws one trial, so it has no "whose 

9008 # stimulus?" question to answer. The *animation* builder does read it (a 

9009 # dual co-animation takes `words_b`), so it is NOT excluded there. 

9010 "compare_stimulus", 

9011 # CMP-24 — B's flags in a co-animation; one trial has no B. 

9012 "fixation_flags_b", 

9013 # DATA-66 — names, not an option: read by `_labelled_columns`. 

9014 "column_labels", 

9015 } 

9016) 

9017#: What `make_comparison_figure` accepts (CMP-9). Only the animation-only fields 

9018#: are excluded — this is a typo-catcher for `api.compare_scanpaths`, not a 

9019#: semantic filter, so it still admits settings the comparison builders ignore. 

9020#: Which settings actually reach a comparison figure is the table in 

9021#: `scanpath_studio/CLAUDE.md` → *Which viz settings apply in which render path*. 

9022COMPARISON_FIGURE_OPTIONS = _setting_names( 

9023 { 

9024 "playback_speed", 

9025 "label_a", 

9026 "label_b", 

9027 "autoplay", 

9028 "anim_grid_step_ms", 

9029 "anim_max_frames", 

9030 # CMP-24 — a comparison reads B's filters off `style_b` instead. 

9031 "fixation_flags_b", 

9032 # DATA-66 — names, not an option: read by `_labelled_columns`. 

9033 "column_labels", 

9034 } 

9035) 

9036ANIMATION_FIGURE_OPTIONS = _setting_names( 

9037 { 

9038 "x_field", 

9039 "y_field", 

9040 "show_fixations", 

9041 "show_heatmap", 

9042 "heatmap_metric", 

9043 "heatmap_style", 

9044 "heatmap_norm", 

9045 "heatmap_sigma_px", 

9046 "heatmap_range", 

9047 "heatmap_colorscale", 

9048 "show_raw_gaze", 

9049 "critical_span_style", 

9050 "saccade_color_mode", 

9051 "saccade_class_colors", 

9052 "saccade_type_legend", 

9053 "saccade_classes", 

9054 "saccade_render_mode", 

9055 "fixation_snap_to_word", 

9056 "span_border_color", 

9057 "word_heatmap_col", 

9058 "word_heatmap_title", 

9059 "show_connectors", 

9060 "connector_y", 

9061 "illustration_reasons", 

9062 "trial_labels", 

9063 "layout", 

9064 # CMP-8 §4 — B-side geometry, read only by the split comparison layouts. 

9065 "canvas_b", 

9066 "background_image_b", 

9067 "background_image_size_b", 

9068 "background_image_origin_b", 

9069 # DATA-66 — names, not an option: read by `_labelled_columns`. 

9070 "column_labels", 

9071 } 

9072) 

9073 

9074 

9075def make_scanpath_figure( 

9076 words: pd.DataFrame, 

9077 fixations: pd.DataFrame, 

9078 *, 

9079 settings: FigureSettings | Mapping[str, Any] | None = None, 

9080 raw_gaze: pd.DataFrame | None = None, 

9081 **overrides: Any, 

9082) -> go.Figure: 

9083 """Build a static scanpath from one shared rendering-settings object. 

9084 

9085 Keyword overrides remain useful for focused programmatic calls and tests; 

9086 application code should pass ``settings=FigureSettings(...)`` so the same 

9087 object can flow unchanged through UI, export, and headless surfaces. 

9088 """ 

9089 resolved = _resolve_figure_settings(settings, overrides) 

9090 fields_xy = (resolved.x_field, resolved.y_field) 

9091 words = _finite_for_plotting(words, drop_on=_WORD_BOX_COLUMNS) 

9092 fixations = _finite_for_plotting(fixations, fields_xy) 

9093 raw_gaze = _finite_for_plotting(raw_gaze) 

9094 with _labelled_columns(resolved.column_labels): 

9095 fig = _render_scanpath_figure( 

9096 words, 

9097 fixations, 

9098 settings=resolved, 

9099 raw_gaze=raw_gaze, 

9100 ) 

9101 _arrange_colorbars(fig) 

9102 apply_legend_layout(fig, resolved.legend_layout) 

9103 if resolved.show_fixations: 

9104 _maybe_add_duration_key(fig, resolved, resolved.marker_size_range, fixations) 

9105 return fig 

9106 

9107 

9108def make_scanpath_animation( 

9109 words: pd.DataFrame, 

9110 fixations: pd.DataFrame, 

9111 *, 

9112 settings: FigureSettings | Mapping[str, Any] | None = None, 

9113 fixations_b: pd.DataFrame | None = None, 

9114 words_b: pd.DataFrame | None = None, 

9115 **overrides: Any, 

9116) -> go.Figure: 

9117 """Build an animated replay from the shared rendering settings.""" 

9118 fig, _frame_step_ms = build_scanpath_replay( 

9119 words, 

9120 fixations, 

9121 settings=settings, 

9122 fixations_b=fixations_b, 

9123 words_b=words_b, 

9124 **overrides, 

9125 ) 

9126 return fig 

9127 

9128 

9129def build_scanpath_replay( 

9130 words: pd.DataFrame, 

9131 fixations: pd.DataFrame, 

9132 *, 

9133 settings: FigureSettings | Mapping[str, Any] | None = None, 

9134 fixations_b: pd.DataFrame | None = None, 

9135 words_b: pd.DataFrame | None = None, 

9136 **overrides: Any, 

9137) -> tuple[go.Figure, float]: 

9138 """:func:`make_scanpath_animation`, returning the grid step with the figure. 

9139 

9140 ``(figure, frame_step_ms)``: pass the step to :func:`set_replay_clock` to 

9141 re-time the replay at another speed or autoplay without rebuilding a frame 

9142 (PERF-15).""" 

9143 resolved = _resolve_figure_settings( 

9144 settings, 

9145 overrides, 

9146 legacy_defaults={ 

9147 "order_font_color": "#000000", 

9148 "color_by": None, 

9149 "fixation_color": None, 

9150 "highlight_column": None, 

9151 "word_hover_measure": None, 

9152 }, 

9153 ) 

9154 words = _finite_for_plotting(words, drop_on=_WORD_BOX_COLUMNS) 

9155 words_b = _finite_for_plotting(words_b, drop_on=_WORD_BOX_COLUMNS) 

9156 fixations = _finite_for_plotting(fixations) 

9157 fixations_b = _finite_for_plotting(fixations_b) 

9158 with _labelled_columns(resolved.column_labels): 

9159 fig, frame_step_ms = _render_scanpath_animation( 

9160 words, 

9161 fixations, 

9162 settings=resolved, 

9163 fixations_b=fixations_b, 

9164 words_b=words_b, 

9165 ) 

9166 apply_legend_layout( 

9167 fig, 

9168 resolved.legend_layout, 

9169 comparing=fixations_b is not None and not fixations_b.empty, 

9170 ) 

9171 size_range = replay_size_key_range(resolved, fixations, fixations_b) 

9172 if size_range is not None: 

9173 _maybe_add_duration_key(fig, resolved, size_range, fixations, fixations_b) 

9174 return fig, frame_step_ms 

9175 

9176 

9177def replay_size_key_range( 

9178 settings: FigureSettings, 

9179 fixations: pd.DataFrame | None, 

9180 fixations_b: pd.DataFrame | None = None, 

9181) -> tuple[int, int] | None: 

9182 """The size range a replay's duration key draws, or ``None`` for no key. 

9183 

9184 A lone replay's is the figure's ``marker_size_range``. A co-animation sizes 

9185 each scanpath in its own style's range, and one key serves both only while 

9186 those agree — as on the static comparison.""" 

9187 dual = all(f is not None and not f.empty for f in (fixations, fixations_b)) 

9188 if not dual: 

9189 return tuple(settings.marker_size_range) 

9190 ranges = { 

9191 tuple( 

9192 _comparison_scanpath_style( 

9193 idx, style, default_marker_size_range=settings.marker_size_range 

9194 )["marker_size_range"] 

9195 ) 

9196 for idx, style in enumerate((settings.style_a, settings.style_b)) 

9197 } 

9198 return ranges.pop() if len(ranges) == 1 else None 

9199 

9200 

9201def _require_one_screen_per_reading( 

9202 words: pd.DataFrame | None, 

9203 fixations: pd.DataFrame | None, 

9204 readings: Sequence[tuple[str, str]], 

9205) -> None: 

9206 """Refuse a comparison reading that spans several screens. 

9207 

9208 Every screen of a multipart trial is its own coordinate space, so one 

9209 scanpath drawn from two of them joins its last fixation on one page to the 

9210 first on the next — a saccade nobody made. Callers cut each reading to one 

9211 screen first (`multipart.extract_part`); this is the guard that keeps a new 

9212 caller from forgetting to. 

9213 """ 

9214 for label, (participant, trial) in zip(("A", "B"), readings, strict=False): 

9215 for frame in (fixations, words): 

9216 if frame is None or frame.empty or SCREEN_ID not in frame.columns: 

9217 continue 

9218 rows = frame[ 

9219 (frame["participant_id"] == participant) & (frame["trial_id"] == trial) 

9220 ] 

9221 screens = rows[SCREEN_ID].dropna().astype(str).unique() 

9222 if len(screens) > 1: 

9223 shown = ", ".join(repr(value) for value in screens[:5]) 

9224 raise ValueError( 

9225 f"Scanpath {label} (participant={participant!r}, " 

9226 f"trial={trial!r}) spans {len(screens)} screens ({shown}). " 

9227 "Each screen is its own coordinate space, so a comparison " 

9228 "draws one screen per scanpath: cut each trial to one " 

9229 "screen first (multipart.extract_part, or " 

9230 "compare_scanpaths' screen= / screen_b=)." 

9231 ) 

9232 

9233 

9234def make_comparison_figure( 

9235 words: pd.DataFrame, 

9236 fixations: pd.DataFrame, 

9237 trial_a: tuple[str, str], 

9238 trial_b: tuple[str, str], 

9239 *, 

9240 settings: FigureSettings | Mapping[str, Any] | None = None, 

9241 raw_gaze: pd.DataFrame | None = None, 

9242 **overrides: Any, 

9243) -> go.Figure: 

9244 """Build a two-scanpath comparison from the shared rendering settings. 

9245 

9246 ``raw_gaze`` (VIZ-48) holds either reading's samples, keyed like ``words`` 

9247 and ``fixations``; with ``show_raw_gaze`` each reading's are drawn under its 

9248 scanpath, in its colour. 

9249 

9250 Each scanpath must be one screen: a frame holding several screens of one 

9251 multipart reading raises ``ValueError`` rather than pooling coordinate 

9252 spaces (and drawing saccades across page boundaries).""" 

9253 _require_one_screen_per_reading(words, fixations, (trial_a, trial_b)) 

9254 resolved = _resolve_figure_settings( 

9255 settings, 

9256 overrides, 

9257 legacy_defaults={ 

9258 "show_word_labels": False, 

9259 "show_order": False, 

9260 "order_font_size": None, 

9261 "color_by": None, 

9262 "highlight_column": None, 

9263 "heatmap_metric": "duration_ms", 

9264 }, 

9265 ) 

9266 words = _finite_for_plotting(words, drop_on=_WORD_BOX_COLUMNS) 

9267 fixations = _finite_for_plotting(fixations) 

9268 raw_gaze = _finite_for_plotting(raw_gaze) 

9269 with _labelled_columns(resolved.column_labels): 

9270 fig = _render_comparison_figure( 

9271 words, 

9272 fixations, 

9273 trial_a, 

9274 trial_b, 

9275 settings=resolved, 

9276 raw_gaze=raw_gaze, 

9277 ) 

9278 _arrange_colorbars(fig) 

9279 apply_legend_layout(fig, resolved.legend_layout, comparing=True) 

9280 # One key serves both scanpaths only while they share a size range; with 

9281 # per-scanpath ranges (Compare's own Size) one duration is two sizes. 

9282 ranges = { 

9283 tuple( 

9284 _comparison_scanpath_style( 

9285 idx, style, default_marker_size_range=resolved.marker_size_range 

9286 )["marker_size_range"] 

9287 ) 

9288 for idx, style in enumerate((resolved.style_a, resolved.style_b)) 

9289 } 

9290 if resolved.show_fixations and len(ranges) == 1: 

9291 _maybe_add_duration_key(fig, resolved, ranges.pop(), fixations) 

9292 return fig 

9293 

9294 

9295def _maybe_add_duration_key( 

9296 fig: go.Figure, 

9297 settings: FigureSettings, 

9298 size_range: tuple[int, int], 

9299 *fixation_frames: pd.DataFrame | None, 

9300) -> None: 

9301 """Add the duration-size key when the figure draws fixation markers on a 

9302 fixed scale and the caller asked for it (``duration_size_legend``).""" 

9303 if not settings.duration_size_legend or settings.marker_size_scale == "relative": 

9304 return 

9305 if not any(f is not None and not f.empty for f in fixation_frames): 

9306 return 

9307 _add_duration_size_key( 

9308 fig, 

9309 size_range, 

9310 settings.marker_size_scale, 

9311 settings.marker_duration_range, 

9312 font_family=settings.font_family, 

9313 legend_layout=settings.legend_layout, 

9314 )