Coverage for scanpath_studio/plots.py: 97%
3079 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Plotly figure builders for scanpath visualization."""
3from __future__ import annotations
5import base64
6import copy
7import html
8import itertools
9import math
10import re
11import struct
12from collections.abc import Callable, Iterable, Iterator, Mapping, Sequence
13from contextlib import contextmanager
14from contextvars import ContextVar
15from dataclasses import MISSING, dataclass, fields, replace
16from pathlib import Path
17from typing import Any
19import numpy as np
20import pandas as pd
21import plotly.graph_objects as go
23from . import progress
24from .constants import (
25 APP_THEME,
26 CANVAS_PAD_FRACTION,
27 CANVAS_PAD_MIN_PX,
28 COMPARE_FIXATION_OPACITY,
29 COMPARISON_PALETTE,
30 CURRENT_FIX_COLOR,
31 CURRENT_FIX_OUTLINE,
32 DEFAULT_FIXATION_COLOR,
33 DEFAULT_FIXATION_COLORSCALE,
34 DEFAULT_FIXATION_SYMBOL,
35 DEFAULT_HEATMAP_COLORSCALE,
36 DEFAULT_LINE_SPACING,
37 DEFAULT_MARKER_DURATION_RANGE,
38 DEFAULT_MARKER_SIZE_RANGE,
39 DEFAULT_MARKER_SIZE_SCALE,
40 DEFAULT_SACCADE_WIDTH,
41 FIX_MARKER_OUTLINE,
42 FIXATION_GLYPH_SIZE_SCALE,
43 FIXATION_GLYPH_SYMBOLS,
44 FIXATION_SYMBOLS,
45 FONT_FAMILY,
46 HIGHLIGHTED_TEXT_COLOR,
47 HOLLOW_OUTLINE_WIDTH,
48 LEGEND_ARRANGEMENTS,
49 LEGEND_KINDS,
50 LEGEND_POSITIONS,
51 MARKER_SIZE_SCALES,
52 OUT_OF_TEXT_COLOR,
53 SACCADE_CLASS_COLORS,
54 SACCADE_CLASS_LABELS,
55 SACCADE_CLASS_ORDER,
56 SACCADE_COLOR,
57 SACCADE_COLOR_MODES,
58 SACCADE_DASH_OPTIONS,
59 SACCADE_DIRECTION_CLASSES,
60 SACCADE_DIRECTION_FOLD,
61 SACCADE_DIRECTION_LABELS,
62 SAMPLE_INDEX,
63 TRENDLINE_COLOR,
64 UNIFORM_COLOR_FIELD,
65 WORD_BOX_COLOR,
66 WORD_BOX_FILL_COLOR,
67 WORD_BOX_FILL_OPACITY,
68 WORD_BOX_LINE_OPACITY,
69 WORD_LABEL_COLOR,
70 compare_palette_color,
71)
72from .illustration import MANUAL_LABEL_REASON
73from .multipart import SCREEN_ID
75COLORBAR_LEN_FRACTION = 0.33
78@dataclass(frozen=True)
79class FigureSettings:
80 """Immutable rendering settings shared by every scanpath figure builder.
82 The app, headless API, CLI, and exporters used to forward parallel keyword
83 lists into three builders with 45–69 parameters each. This object is the
84 single rendering contract instead. Builder-specific fields live together
85 deliberately: switching between static, animated, and comparison views must
86 preserve the common visual choices without another translation layer.
88 ``canvas_width``, ``canvas_height``, and ``base_font_size`` are the only
89 context-dependent required values. Everything else has the builder's
90 behavior-preserving default and may be changed with :meth:`with_overrides`.
91 """
93 canvas_width: int
94 canvas_height: int
95 base_font_size: int
96 font_family: str = FONT_FAMILY
97 x_field: str = "x"
98 y_field: str = "y"
99 show_words: bool = True
100 #: The word boxes' outline colour at ``word_box_line_opacity``, and their
101 #: fill — a colour drawn at ``word_box_fill_opacity``. A comparison outlines
102 #: each reading's boxes in its scanpath's ``box_color`` style (its fixation
103 #: colour by default) instead, so ``word_box_color`` is static/replay only;
104 #: the line opacity applies to every outline.
105 word_box_color: str = WORD_BOX_COLOR
106 word_box_line_opacity: float = WORD_BOX_LINE_OPACITY
107 word_box_fill_color: str = WORD_BOX_FILL_COLOR
108 word_box_fill_opacity: float = WORD_BOX_FILL_OPACITY
109 show_word_labels: bool = True
110 show_fixations: bool = True
111 show_order: bool = True
112 show_saccades: bool = True
113 show_heatmap: bool = False
114 color_by: str | None = UNIFORM_COLOR_FIELD
115 heatmap_metric: str | None = None
116 show_saccade_arrows: bool = False
117 heatmap_style: str = "Word boxes"
118 heatmap_norm: str = "Linear"
119 #: The Interpolated heatmap's Gaussian σ in px; ``None`` picks it from the
120 #: data (`interpolated_sigma_px`).
121 heatmap_sigma_px: float | None = None
122 marker_size_range: tuple[int, int] = DEFAULT_MARKER_SIZE_RANGE
123 #: How duration maps onto ``marker_size_range`` — one of
124 #: ``constants.MARKER_SIZE_SCALES``. The fixed scales ("sqrt", "linear",
125 #: "log") map ``marker_duration_range`` (ms) onto it for every figure, so
126 #: equal durations draw at equal sizes across trials, comparison sides,
127 #: replays and exports; "relative" spans each figure's own durations.
128 marker_size_scale: str = DEFAULT_MARKER_SIZE_SCALE
129 marker_duration_range: tuple[float, float] = DEFAULT_MARKER_DURATION_RANGE
130 #: A few reference circles labelled in ms, drawn under a fixed scale.
131 duration_size_legend: bool = True
132 #: Where each legend sits, as ``{kind: {"position", "arrangement", "size"}}``
133 #: for the kinds in ``LEGEND_KINDS``. A kind left out, or set to "auto"
134 #: throughout, is drawn where it always was; whether it is drawn at all is
135 #: still its own layer's switch. See :func:`normalize_legend_layout`.
136 legend_layout: dict | None = None
137 order_font_size: int | None = 10
138 order_font_color: str = "#111111"
139 #: Each colour scale's bar has its own switch and style: the fixations'
140 #: (a numeric ``color_by``) and the heatmap's.
141 show_fixation_colorbar: bool = True
142 fixation_colorbar_orientation: str = "Vertical"
143 fixation_colorbar_tickangle: int = 0
144 fixation_colorbar_tickfont_size: int = 12
145 show_heatmap_colorbar: bool = True
146 heatmap_colorbar_orientation: str = "Vertical"
147 heatmap_colorbar_tickangle: int = 0
148 heatmap_colorbar_tickfont_size: int = 12
149 fixation_color_range: tuple[float, float] | None = None
150 heatmap_range: tuple[float, float] | None = None
151 fixation_colorscale: str = DEFAULT_FIXATION_COLORSCALE
152 heatmap_colorscale: str = DEFAULT_HEATMAP_COLORSCALE
153 show_raw_gaze: bool = False
154 raw_gaze_color: str = "#888888"
155 raw_gaze_marker_size: float = 4.0
156 raw_gaze_opacity: float = 0.6
157 critical_span_style: str = "Mark text"
158 highlight_column: str | None = "is_in_aspan"
159 saccade_color: str = SACCADE_COLOR
160 saccade_style: str = "solid"
161 saccade_width: float = DEFAULT_SACCADE_WIDTH
162 saccade_color_mode: str = "Uniform"
163 saccade_class_colors: dict | None = None
164 saccade_type_legend: bool = True
165 saccade_classes: Iterable[str] | None = None
166 saccade_render_mode: str = "Straight"
167 fixation_snap_to_word: bool = False
168 hollow_fixations: bool = False
169 fixation_opacity: float = 1.0
170 fixation_color: str | None = DEFAULT_FIXATION_COLOR
171 fixation_symbol: str = DEFAULT_FIXATION_SYMBOL
172 text_color: str = WORD_LABEL_COLOR
173 highlight_text_color: str = HIGHLIGHTED_TEXT_COLOR
174 background_color: str | None = None
175 color_by_line: bool = False
176 fixation_flags: dict | None = None
177 #: CMP-24: scanpath B's own flags in a co-animation (Animate + Compare);
178 #: ``None`` gives B the same ``fixation_flags`` as A.
179 fixation_flags_b: dict | None = None
180 span_border_color: str = "#000000"
181 line_spacing: float = DEFAULT_LINE_SPACING
182 scale_text_to_boxes: bool = True
183 background_image: str | None = None
184 background_image_size: tuple[float, float] | None = None
185 background_image_origin: tuple[float, float] | None = None
186 background_image_opacity: float = 1.0
187 fit_to_monitor: bool = False
188 show_coordinate_grid: bool = False
189 coordinate_grid_spacing: float | None = None
190 word_heatmap_col: str | None = None
191 word_heatmap_title: str | None = None
192 word_hover_measure: str | None = "total_fixation_duration_ms"
193 word_hover_fields: Sequence[str] | None = None
194 fixation_hover_fields: Sequence[str] | None = None
195 show_connectors: bool = False
196 connector_y: Sequence[float] | None = None
197 illustration_reasons: Sequence[str] | None = None
198 #: The Illustration label's text; empty writes "Illustration · <reasons>".
199 illustration_text: str = ""
200 playback_speed: float = 1.0
201 label_a: str = "Scanpath A"
202 label_b: str = "Scanpath B"
203 show_legend: bool = False
204 autoplay: bool = True
205 anim_grid_step_ms: float | None = None
206 anim_max_frames: int | None = None
207 trial_labels: tuple[str, str] | None = None
208 layout: str = "overlay"
209 style_a: dict | None = None
210 style_b: dict | None = None
211 # CMP-8 §4 — scanpath B's own screen, honoured *only* by
212 # `_make_split_comparison_figure` (side-by-side / stacked). `None` means "the
213 # same screen as A", which is every same-dataset comparison and so leaves
214 # every existing figure byte-identical. Overlay never needs these: CMP-11
215 # lets a cross-dataset pair overlay only when both screens are equal, so
216 # there is no second canvas for it to reconcile.
217 canvas_b: tuple[int, int] | None = None
218 background_image_b: str | None = None
219 background_image_size_b: tuple[float, float] | None = None
220 background_image_origin_b: tuple[float, float] | None = None
221 # CMP-11 — which reading supplies the stimulus layer (word boxes + labels)
222 # on an OVERLAY: "both" (the default, and byte-identical to pre-CMP-11),
223 # "a", or "b". Two datasets' AOIs coincide only when the text is identical,
224 # so an overlay across corpora can otherwise stack two offset sets of
225 # rectangles. Split layouts ignore it — each panel owns its own stimulus,
226 # and hiding one panel's boxes would just leave a blank half.
227 compare_stimulus: str = "both"
228 # DATA-66 — the dataset's own names for the columns the figure's text names
229 # (hover rows, colour-bar and legend titles, non-spatial axis titles),
230 # canonical column → label. Built by the caller (`tabs`); not a figure
231 # option, so no option set, link or `render` flag carries it.
232 column_labels: dict | None = None
234 @classmethod
235 def from_mapping(
236 cls,
237 settings: FigureSettings | Mapping[str, Any] | None = None,
238 /,
239 **overrides: Any,
240 ) -> FigureSettings:
241 """Build settings from another instance or a plain option mapping."""
242 valid = {field.name for field in fields(cls)}
243 unknown = sorted(set(overrides) - valid)
244 if isinstance(settings, cls):
245 if unknown:
246 raise TypeError(f"Unknown figure settings: {', '.join(unknown)}")
247 return replace(settings, **overrides) if overrides else settings
248 values = dict(settings or {})
249 unknown = sorted((set(values) | set(overrides)) - valid)
250 if unknown:
251 raise TypeError(f"Unknown figure settings: {', '.join(unknown)}")
252 values.update(overrides)
253 return cls(**values)
255 def with_overrides(self, **overrides: Any) -> FigureSettings:
256 """Return a copy with the named settings replaced."""
257 return self.from_mapping(self, **overrides)
259 def for_builder(self, names: Iterable[str]) -> dict[str, Any]:
260 """Return just the fields consumed by one concrete renderer."""
261 return {name: getattr(self, name) for name in names}
263 @classmethod
264 def defaults(cls, names: Iterable[str]) -> dict[str, Any]:
265 """Return dataclass defaults for the requested non-context fields."""
266 defaults: dict[str, Any] = {}
267 by_name = {field.name: field for field in fields(cls)}
268 for name in names:
269 field = by_name.get(name)
270 if field is None:
271 continue
272 if field.default is not MISSING:
273 defaults[name] = field.default
274 return defaults
277#: The figure options that take one of a fixed set of values → those values,
278#: spelt as the builders compare them. `normalize_option_values` reads any
279#: spelling of one (case, spaces, ``-`` / ``_`` and ``/`` ignored, so the CLI's
280#: ``word-boxes`` and ``mark-border`` work) and refuses anything else, so a
281#: script never gets the default drawn in place of a value it misspelt.
282FIGURE_OPTION_CHOICES: dict[str, tuple[str, ...]] = {
283 "heatmap_style": ("Word boxes", "Interpolated"),
284 "heatmap_norm": ("Linear", "Log"),
285 "critical_span_style": ("Mark text", "Mark border", "None"),
286 "saccade_color_mode": tuple(SACCADE_COLOR_MODES),
287 "saccade_render_mode": ("Straight", "Arc"),
288 "saccade_style": tuple(SACCADE_DASH_OPTIONS.values()),
289 "marker_size_scale": tuple(MARKER_SIZE_SCALES),
290 "fixation_symbol": tuple(FIXATION_SYMBOLS),
291 "fixation_colorbar_orientation": ("Vertical", "Horizontal"),
292 "heatmap_colorbar_orientation": ("Vertical", "Horizontal"),
293 "compare_stimulus": ("both", "a", "b"),
294}
297def _choice_key(value: object) -> str:
298 return "".join(ch for ch in str(value).casefold() if ch.isalnum())
301#: Other spellings a choice is known by: the CLI's flag names and the app's
302#: labels where they differ from the value (``Dashed`` is ``dash``).
303_CHOICE_ALIASES: dict[str, dict[str, str]] = {
304 "saccade_color_mode": {
305 "type": "By type",
306 "direction": "Forward / regression",
307 "bydirection": "Forward / regression",
308 },
309 "saccade_render_mode": {"arcs": "Arc"},
310 "saccade_style": {
311 _choice_key(label): value for label, value in SACCADE_DASH_OPTIONS.items()
312 },
313}
316def normalize_option_value(name: str, value: object) -> object:
317 """``value`` for the enumerated figure option ``name``, spelt as the
318 builders compare it; any other option's value is returned as it is.
320 Raises ``ValueError`` listing the choices for a value that is none of
321 them. ``critical_span_style=None`` is the app's "None" (no marking)."""
322 choices = FIGURE_OPTION_CHOICES.get(name)
323 if choices is None:
324 return value
325 if value is None and "None" in choices:
326 return "None"
327 key = _choice_key(value)
328 for choice in choices:
329 if _choice_key(choice) == key:
330 return choice
331 alias = _CHOICE_ALIASES.get(name, {}).get(key)
332 if alias is not None:
333 return alias
334 raise ValueError(f"Unknown {name} {value!r}; choose one of {', '.join(choices)}.")
337#: The palettes' short names (#374): no spaces or brackets, so a shell needs
338#: no quotes — `render --palette print`. The app's own names work too.
339PALETTE_SLUGS = {
340 "default": "Default (colourblind-safe)",
341 "print": "Print / greyscale",
342 "high-contrast": "High contrast",
343}
344_PALETTE_ALIASES = {
345 "colourblindsafe": "Default (colourblind-safe)",
346 "colorblindsafe": "Default (colourblind-safe)",
347 # #374: the names the app shows (US spelling, `PALETTE_LABELS`).
348 "defaultcolorblindsafe": "Default (colourblind-safe)",
349 "greyscale": "Print / greyscale",
350 "grayscale": "Print / greyscale",
351 "printgrayscale": "Print / greyscale",
352}
355def normalize_palette(value: object) -> str:
356 """The palette ``value`` names — its app name, short name (``print``) or
357 any spelling of either — else ``ValueError`` listing them."""
358 from .constants import PALETTES, palette_label
360 key = _choice_key(value)
361 for name in PALETTES:
362 if _choice_key(name) == key:
363 return name
364 for slug, name in PALETTE_SLUGS.items():
365 if _choice_key(slug) == key:
366 return name
367 if key in _PALETTE_ALIASES:
368 return _PALETTE_ALIASES[key]
369 raise ValueError(
370 f"Unknown palette {value!r}; choose one of {', '.join(PALETTE_SLUGS)} "
371 f"({', '.join(palette_label(name) for name in PALETTES)})."
372 )
375def palette_slug(name: str) -> str:
376 """A palette's short name, for a command line."""
377 return next((slug for slug, full in PALETTE_SLUGS.items() if full == name), name)
380def normalize_option_values(options: Mapping[str, Any]) -> dict[str, Any]:
381 """``options`` with every enumerated value read by
382 `normalize_option_value` (inside ``style_a`` / ``style_b`` too)."""
383 out = {name: normalize_option_value(name, value) for name, value in options.items()}
384 for side in ("style_a", "style_b"):
385 style = out.get(side)
386 if isinstance(style, Mapping):
387 out[side] = {
388 key: normalize_option_value(key, value) for key, value in style.items()
389 }
390 return out
393def _sample_colorscale_colors(
394 values, colorscale: str, cmin: float | None, cmax: float | None
395) -> object:
396 """Map numeric values to concrete CSS colours via a named Plotly colorscale.
398 Used for hollow markers: Plotly can render a colorscale on a marker *fill*
399 but not on its outline, so the gradient is sampled to literal colours that
400 can sit on ``marker.line.color``. Falls back to a single outline colour if
401 sampling is unavailable.
402 """
403 try:
404 from plotly.colors import sample_colorscale
405 except Exception:
406 return FIX_MARKER_OUTLINE
407 vals = pd.to_numeric(pd.Series(list(values)), errors="coerce")
408 lo = float(cmin) if cmin is not None else float(vals.min())
409 hi = float(cmax) if cmax is not None else float(vals.max())
410 if not np.isfinite(lo) or not np.isfinite(hi) or hi <= lo:
411 norm = [0.5] * len(vals)
412 else:
413 norm = ((vals.clip(lo, hi) - lo) / (hi - lo)).fillna(0.0).tolist()
414 try:
415 return sample_colorscale(colorscale, norm)
416 except Exception:
417 return FIX_MARKER_OUTLINE
420def _make_hollow(marker: dict) -> dict:
421 """Return a copy of a fixation marker dict rendered as outline-only.
423 The fill colour is moved onto the outline (so the colour is preserved) and
424 the fill itself is made transparent. Numeric colorscale colours are sampled
425 to concrete CSS colours because Plotly can't map a colorscale onto an
426 outline. The colorbar is dropped in hollow mode (it needs the coloured fill).
427 """
428 m = dict(marker)
429 color = m.get("color")
430 colorscale = m.get("colorscale")
431 if colorscale is not None and color is not None and not isinstance(color, str):
432 outline_color = _sample_colorscale_colors(
433 color, colorscale, m.get("cmin"), m.get("cmax")
434 )
435 else:
436 outline_color = color if color is not None else FIX_MARKER_OUTLINE
437 line = dict(m.get("line") or {})
438 line["color"] = outline_color
439 line["width"] = HOLLOW_OUTLINE_WIDTH
440 m["line"] = line
441 m["color"] = "rgba(0,0,0,0)"
442 m["colorscale"] = None
443 m["showscale"] = False
444 m["colorbar"] = None
445 return m
448#: The numeric columns a figure places, sizes or times things by.
449_PLOTTED_NUMBERS = ("x", "y", "width", "height", "duration_ms", "timestamp_ms")
450#: A word whose box has an infinite edge has no box to draw.
451_WORD_BOX_COLUMNS = ("x", "y", "width", "height")
454def _finite_for_plotting(
455 frame: pd.DataFrame | None,
456 extra: Iterable[str] = (),
457 *,
458 drop_on: Iterable[str] = (),
459) -> pd.DataFrame | None:
460 """``frame`` with ±inf in its plotted numbers read as missing.
462 Data checks reports such rows and keeps them in every table; the figure
463 cannot place, size or time them, so it draws them as it draws a missing
464 value — no marker and no saccade to or from them, the smallest marker for a
465 duration, a replay timed by durations. A row infinite in a ``drop_on``
466 column is left out (a word box). Copies only when there is one."""
467 if frame is None or frame.empty:
468 return frame
469 numbers = {
470 c: pd.to_numeric(frame[c], errors="coerce").astype(float)
471 for c in dict.fromkeys((*_PLOTTED_NUMBERS, *extra))
472 if c in frame.columns
473 }
474 infinite = {c: np.isinf(v) for c, v in numbers.items()}
475 infinite = {c: m for c, m in infinite.items() if m.any()}
476 if not infinite:
477 return frame
478 out = frame.copy()
479 drop = pd.Series(False, index=frame.index)
480 for column, mask in infinite.items():
481 out[column] = numbers[column].mask(mask)
482 if column in drop_on:
483 drop |= mask
484 return out[~drop] if drop.any() else out
487def _compute_axis_ranges(
488 canvas_width: int,
489 canvas_height: int,
490 *frames_with_xy: tuple[pd.DataFrame | None, str, str],
491 word_frames: Iterable[pd.DataFrame] = (),
492 fit_to_monitor: bool = False,
493) -> tuple[list, list, float | None, float | None, float | None, float | None]:
494 """Compute padded x/y ranges from any number of (frame, x_col, y_col) tuples.
496 word_frames contribute box-extent bounds: x, x+width and y, y+height.
497 Falls back to (0..canvas_width, canvas_height..0) when there's no data.
498 Returns: x_range, y_range (y inverted), and the unpadded mins/maxs.
500 With ``fit_to_monitor`` the range always spans the full virtual monitor
501 (0..canvas_width, canvas_height..0) regardless of where the data sits, so the
502 whole presentation screen is shown and the scanpath appears at its true
503 on-monitor position rather than the view cropping to the data extent. The
504 returned data mins/maxs still describe the actual data (they size the
505 interpolated heatmap grid), so only the visible window changes.
506 """
507 x_candidates: list = []
508 y_candidates: list = []
510 for df, x_col, y_col in frames_with_xy:
511 if df is None or df.empty:
512 continue
513 if x_col in df.columns:
514 x_candidates.extend([df[x_col].min(), df[x_col].max()])
515 if y_col in df.columns:
516 y_candidates.extend([df[y_col].min(), df[y_col].max()])
518 for df in word_frames:
519 if df is None or df.empty:
520 continue
521 x_candidates.extend([df["x"].min(), (df["x"] + df["width"]).max()])
522 y_candidates.extend([df["y"].min(), (df["y"] + df["height"]).max()])
524 x_range = [0, canvas_width]
525 y_range = [canvas_height, 0]
526 if not x_candidates or not y_candidates:
527 return x_range, y_range, None, None, None, None
529 x_min = float(np.nanmin(x_candidates))
530 x_max = float(np.nanmax(x_candidates))
531 y_min = float(np.nanmin(y_candidates))
532 y_max = float(np.nanmax(y_candidates))
534 if fit_to_monitor:
535 # Show the whole monitor; the scanpath keeps its true on-screen position.
536 # Real data mins/maxs are still returned (heatmap-grid extent).
537 return [0, canvas_width], [canvas_height, 0], x_min, x_max, y_min, y_max
539 x_span = max(x_max - x_min, 1.0)
540 y_span = max(y_max - y_min, 1.0)
541 pad_x = max(CANVAS_PAD_MIN_PX, CANVAS_PAD_FRACTION * x_span)
542 pad_y = max(CANVAS_PAD_MIN_PX, CANVAS_PAD_FRACTION * y_span)
543 x_range = [x_min - pad_x, x_max + pad_x]
544 y_range = [y_max + pad_y, y_min - pad_y]
545 return x_range, y_range, x_min, x_max, y_min, y_max
548@dataclass(frozen=True)
549class CoordinateGridTicks:
550 """Deterministic, zero-anchored screen-coordinate tick contract."""
552 major_spacing: float
553 minor_spacing: float
554 x_values: tuple[float, ...]
555 x_labels: tuple[str, ...]
556 y_values: tuple[float, ...]
557 y_labels: tuple[str, ...]
560def _nice_grid_spacing(raw: float) -> float:
561 """Round a positive interval up to the 1/2/5×10ⁿ sequence."""
562 exponent = math.floor(math.log10(max(raw, 1e-12)))
563 unit = 10.0**exponent
564 normalized = raw / unit
565 for candidate in (1.0, 2.0, 5.0, 10.0):
566 if normalized <= candidate:
567 return candidate * unit
568 return 10.0 * unit
571def _anchored_grid_values(lo: float, hi: float, spacing: float) -> tuple[float, ...]:
572 """Ticks clipped to ``lo..hi`` and anchored at global screen coordinate 0."""
573 lo, hi = min(lo, hi), max(lo, hi)
574 first = math.ceil((lo - spacing * 1e-9) / spacing)
575 last = math.floor((hi + spacing * 1e-9) / spacing)
576 count = max(0, last - first + 1)
577 if count > 10_000:
578 raise ValueError(
579 "Coordinate-grid spacing creates more than 10,000 ticks; choose a larger interval."
580 )
581 return tuple(round(index * spacing, 10) for index in range(first, last + 1))
584def _grid_label(value: float) -> str:
585 return f"{value:g}"
588def coordinate_grid_ticks(
589 x_range: Sequence[float],
590 y_range: Sequence[float],
591 *,
592 spacing: float | None = None,
593 rendered_width: int = 900,
594 rendered_height: int = 650,
595 monitor_bounds: tuple[float, float, float, float] | None = None,
596) -> CoordinateGridTicks:
597 """Return screen-X/Y grid ticks without changing either visible range.
599 ``spacing=None`` chooses one shared 1/2/5×10ⁿ major interval for both axes.
600 Manual intervals stay exact. Labels are thinned when the rendered display
601 could not fit them, while the underlying tick/grid positions remain stable.
602 ``monitor_bounds`` is accepted to make the coordinate frame explicit; ticks
603 are intentionally clipped to the *visible* ranges and always anchored at
604 screen zero, so cropped/full-monitor transitions cannot shift the grid.
605 """
606 if len(x_range) != 2 or len(y_range) != 2:
607 raise ValueError("Coordinate-grid ranges must each contain two values.")
608 if monitor_bounds is not None and len(monitor_bounds) != 4:
609 raise ValueError("monitor_bounds must be (x_min, x_max, y_min, y_max).")
610 x_span = abs(float(x_range[1]) - float(x_range[0]))
611 y_span = abs(float(y_range[1]) - float(y_range[0]))
612 if spacing is None:
613 target_x = x_span / max(float(rendered_width) / 90.0, 1.0)
614 target_y = y_span / max(float(rendered_height) / 70.0, 1.0)
615 major = _nice_grid_spacing(max(target_x, target_y, 1.0))
616 else:
617 major = float(spacing)
618 if not math.isfinite(major) or major <= 0:
619 raise ValueError(
620 "Coordinate-grid spacing must be a positive finite number."
621 )
622 x_values = _anchored_grid_values(float(x_range[0]), float(x_range[1]), major)
623 y_values = _anchored_grid_values(float(y_range[0]), float(y_range[1]), major)
625 def _labels(
626 values: tuple[float, ...], pixels: int, minimum_px: int
627 ) -> tuple[str, ...]:
628 capacity = max(int(pixels) // minimum_px, 1)
629 stride = max(1, math.ceil(len(values) / capacity))
630 return tuple(
631 _grid_label(value) if index % stride == 0 else ""
632 for index, value in enumerate(values)
633 )
635 return CoordinateGridTicks(
636 major_spacing=major,
637 minor_spacing=major / 5.0,
638 x_values=x_values,
639 x_labels=_labels(x_values, rendered_width, 68),
640 y_values=y_values,
641 y_labels=_labels(y_values, rendered_height, 42),
642 )
645_GRID_LEFT_RESERVE_PX = 52
646_GRID_BOTTOM_RESERVE_PX = 36
647_GRID_TICK_FONT_PX = 18 # remains legible after true-scale responsive downscaling
650def _coordinate_grid_axis_options(
651 ticks: CoordinateGridTicks, *, axis: str
652) -> dict[str, Any]:
653 """Plotly axis options for one restrained major/minor coordinate grid."""
654 values = ticks.x_values if axis == "x" else ticks.y_values
655 labels = ticks.x_labels if axis == "x" else ticks.y_labels
656 return dict(
657 showticklabels=True,
658 showgrid=True,
659 tickmode="array",
660 tickvals=list(values),
661 ticktext=list(labels),
662 ticks="outside",
663 ticklen=4,
664 tickwidth=1,
665 tickcolor="#667085",
666 tickfont=dict(size=_GRID_TICK_FONT_PX, color="#475467"),
667 gridcolor="rgba(71,84,103,0.22)",
668 gridwidth=1,
669 zeroline=True,
670 zerolinecolor="rgba(16,24,40,0.42)",
671 zerolinewidth=1.25,
672 minor=dict(
673 showgrid=True,
674 dtick=ticks.minor_spacing,
675 gridcolor="rgba(71,84,103,0.09)",
676 gridwidth=0.5,
677 ticks="",
678 ),
679 )
682def _apply_coordinate_grid_axes(
683 xaxis: dict,
684 yaxis: dict,
685 *,
686 show: bool,
687 spacing: float | None,
688 x_range: Sequence[float],
689 y_range: Sequence[float],
690 rendered_width: int,
691 rendered_height: int,
692) -> None:
693 """Mutate spatial axis dicts only when the optional grid is enabled."""
694 if not show:
695 return
696 ticks = coordinate_grid_ticks(
697 x_range,
698 y_range,
699 spacing=spacing,
700 rendered_width=rendered_width,
701 rendered_height=rendered_height,
702 )
703 xaxis.update(_coordinate_grid_axis_options(ticks, axis="x"))
704 yaxis.update(_coordinate_grid_axis_options(ticks, axis="y"))
707# Cap the *fixed* render size so the true-to-scale plot (rendered at exactly
708# these pixels via tabs._render_true_scale_chart) fits a typical research display
709# without horizontal scrolling. Aspect ratio is preserved when shrinking — both
710# dims scale together, so boxes/text/fixations keep one true scale. A wider
711# monitor just leaves margin (the plot is "narrower than the column", never
712# stretched); a narrower window scrolls rather than distorting.
713_DISPLAY_MAX_HEIGHT = 690
714_DISPLAY_MAX_WIDTH = 960
717def _fit_display_size(
718 canvas_width: int,
719 canvas_height: int,
720 x_range: list,
721 y_range: list,
722 spatial_axes: bool,
723) -> tuple[int, int]:
724 """Return (width, height) for `fig.update_layout` so the plot fits onscreen.
726 With `scaleanchor="x", scaleratio=1` the plot domain shrinks to the data
727 aspect ratio, leaving large blank vertical strips when the figure box is
728 the full monitor height. We match the figure box to the actual plot
729 domain — and additionally clamp both dims so the whole plot fits in one
730 viewport without scrolling. Falls back to (canvas_w, canvas_h) when axes
731 aren't spatial or the data range is degenerate.
732 """
733 if not spatial_axes:
734 return canvas_width, canvas_height
735 x_span = x_range[1] - x_range[0]
736 y_span = y_range[0] - y_range[1] # y_range is inverted [y_max, y_min]
737 if x_span <= 0 or y_span <= 0:
738 return canvas_width, canvas_height
739 aspect = x_span / y_span
740 w, h = canvas_width, round(canvas_width / aspect)
741 # Shrink (preserving aspect) until both dims fit the viewport caps.
742 if h > _DISPLAY_MAX_HEIGHT:
743 h = _DISPLAY_MAX_HEIGHT
744 w = round(h * aspect)
745 if w > _DISPLAY_MAX_WIDTH:
746 w = _DISPLAY_MAX_WIDTH
747 h = round(w / aspect)
748 return max(w, 100), max(h, 100)
751#: The two colour bars' settings (fixations', heatmap's) and their defaults —
752#: the one list the app's settings dicts and saved config copy them by.
753COLORBAR_DEFAULTS: dict = {
754 f.name: f.default
755 for f in fields(FigureSettings)
756 if f.name.endswith("_colorbar") or "_colorbar_" in f.name
757}
759# Extra figure size (px) reserved OUTSIDE the equal-aspect plot region for a
760# right-side colorbar or a top legend. Without this, Plotly's automargin shrinks
761# the scaleanchor'd plot domain to fit them — and because the word labels are
762# sized for the full fitted_w x fitted_h plot region, a shrunken plot leaves the
763# text overflowing the boxes (the "colorbar / discrete colour legend shrinks the
764# plot and breaks the aspect ratio" bug). Mirroring the _CONTROLS_MARGIN_PX trick
765# the animation uses for its transport controls, we instead grow the figure by
766# the reserve and pin it as an explicit margin, so the plot region stays exactly
767# fitted_w x fitted_h whether or not a colorbar/legend is shown.
768_COLORBAR_RESERVE_PX = 160
769_LEGEND_RESERVE_PX = 60
770# Top reserve for the overlay-comparison figure's title + A/B legend (same idea
771# as _LEGEND_RESERVE_PX, but the title needs a touch more room).
772_OVERLAY_TOP_PX = 64
773#: UX-172: the Compare A/B legend is the only thing naming the two readings, so
774#: it reads larger than the figure's body text (overlay, split and co-animation).
775_COMPARE_LEGEND_FONT_SCALE = 1.3
778def _compare_legend_font(base_font_size, font_family=None) -> dict:
779 """The A/B legend's font in every Compare figure (UX-172)."""
780 font = {"size": round(float(base_font_size or 16) * _COMPARE_LEGEND_FONT_SCALE)}
781 if font_family:
782 font["family"] = font_family
783 return font
786# A horizontal colorbar sits below the plot, so it reserves bottom (not right).
787_COLORBAR_BOTTOM_PX = 96
790def _decoration_margins(
791 fitted_w: int,
792 fitted_h: int,
793 *,
794 legend: bool,
795 colorbar_right: bool = False,
796 colorbar_below: bool = False,
797 bottom: int = 0,
798 coordinate_grid: bool = False,
799) -> dict:
800 """Grow a spatial figure so a right/bottom colorbar + top legend sit in
801 reserved margin instead of stealing space from the equal-aspect plot region.
803 Returns ``{"width", "height", "margin"}`` for ``fig.update_layout``: the plot
804 region stays ``fitted_w x fitted_h`` (so the true-to-scale word labels keep
805 matching the boxes); ``bottom`` reserves additional space below the plot for
806 transport controls (the animation figure); ``colorbar_right`` /
807 ``colorbar_below`` reserve room for a vertical / horizontal colour bar —
808 both, when the fixations' and the heatmap's bars point different ways.
809 """
810 right = _COLORBAR_RESERVE_PX if colorbar_right else 0
811 cb_bottom = _COLORBAR_BOTTOM_PX if colorbar_below else 0
812 top = _LEGEND_RESERVE_PX if legend else 0
813 left = _GRID_LEFT_RESERVE_PX if coordinate_grid else 0
814 grid_bottom = _GRID_BOTTOM_RESERVE_PX if coordinate_grid else 0
815 return {
816 "width": fitted_w + left + right,
817 "height": fitted_h + top + bottom + cb_bottom + grid_bottom,
818 "margin": dict(l=left, r=right, t=top, b=bottom + cb_bottom + grid_bottom),
819 }
822def _colorbar_reserves(*bars: tuple[bool, str]) -> dict:
823 """``colorbar_right`` / ``colorbar_below`` for `_decoration_margins`, from
824 each bar's ``(drawn, orientation)``."""
825 return {
826 "colorbar_right": any(on and o != "Horizontal" for on, o in bars),
827 "colorbar_below": any(on and o == "Horizontal" for on, o in bars),
828 }
831def _colorbar_dict(
832 title: str,
833 *,
834 orientation: str = "Vertical",
835 tickangle: int = 0,
836 tickfont_size: int = 12,
837) -> dict:
838 """A styled Plotly colorbar dict (vertical right / horizontal below), with
839 rotatable, sizable tick labels and a slim bar."""
840 horizontal = orientation == "Horizontal"
841 cb = dict(
842 title=dict(
843 text=title,
844 side="top" if horizontal else "right",
845 font=dict(size=max(10, int(tickfont_size) + 1)),
846 ),
847 thickness=14,
848 tickangle=int(tickangle),
849 tickfont=dict(size=int(tickfont_size)),
850 outlinewidth=0,
851 )
852 if horizontal:
853 cb.update(
854 orientation="h",
855 x=0.5,
856 xanchor="center",
857 y=-0.04,
858 yanchor="top",
859 lenmode="fraction",
860 len=0.6,
861 )
862 else:
863 cb.update(
864 x=1.02,
865 xanchor="left",
866 y=0.5,
867 yanchor="middle",
868 lenmode="fraction",
869 len=COLORBAR_LEN_FRACTION,
870 )
871 return cb
874def _colorbar_owners(fig: go.Figure) -> list:
875 """Every object drawing a colour bar on ``fig``, in trace order: a trace
876 with its own scale (``go.Heatmap``) or a trace's marker."""
877 owners = []
878 for trace in fig.data:
879 if getattr(trace, "showscale", None):
880 owners.append(trace)
881 continue
882 marker = getattr(trace, "marker", None)
883 if marker is not None and getattr(marker, "showscale", None):
884 owners.append(marker)
885 return owners
888def _arrange_colorbars(fig: go.Figure) -> None:
889 """Give each of several colour bars its own place (round-7 review, finding 16).
891 `_colorbar_dict` puts every bar of one orientation at one spot, so a
892 heatmap's scale and the fixations' were drawn over each other. With two or
893 more pointing the same way, vertical bars stand side by side to the right
894 of the plot and horizontal ones stack below it, the figure growing by the
895 room they take — the plot region, and so the true-to-scale text, keep
896 their size. A bar alone in its orientation keeps its geometry exactly (a
897 vertical and a horizontal bar each already have their reserved margin).
898 Each bar is its own mapping (variable, units, palette, range): none is
899 merged into another, so every scale stays readable.
900 """
901 owners = _colorbar_owners(fig)
902 below = [o for o in owners if o.colorbar.orientation == "h"]
903 right = [o for o in owners if o.colorbar.orientation != "h"]
904 layout = fig.layout
905 if len(below) > 1:
906 first = below[0].colorbar
907 tick_px = float(first.tickfont.size or 12)
908 rotated = abs(float(first.tickangle or 0)) > 30
909 # Title above the bar, the bar, its tick labels below.
910 row_px = 56.0 + 2.0 * tick_px + (2.0 * tick_px if rotated else 0.0)
911 height = float(layout.height or 450)
912 top, bottom = float(layout.margin.t or 0), float(layout.margin.b or 0)
913 plot_h = max(height - top - bottom, 1.0)
914 base_y = float(first.y if first.y is not None else -0.04)
915 for i, owner in enumerate(below):
916 owner.colorbar.y = base_y - i * row_px / plot_h
917 grow = (len(below) - 1) * row_px
918 fig.update_layout(height=height + grow, margin=dict(b=bottom + grow))
919 if len(right) > 1:
920 first = right[0].colorbar
921 tick_px = float(first.tickfont.size or 12)
922 # The bar, its tick labels, then its title read sideways.
923 step_px = 70.0 + 3.0 * tick_px
924 # Every spatial builder sizes its figure; Plotly's own default otherwise.
925 width = float(layout.width or 700)
926 left, margin_r = float(layout.margin.l or 0), float(layout.margin.r or 0)
927 plot_w = max(width - left - margin_r, 1.0)
928 base_x = float(first.x if first.x is not None else 1.02)
929 for i, owner in enumerate(right):
930 owner.colorbar.x = base_x + i * step_px / plot_w
931 new_right = (
932 max(margin_r, float(_COLORBAR_RESERVE_PX)) + (len(right) - 1) * step_px
933 )
934 fig.update_layout(
935 width=width + (new_right - margin_r), margin=dict(r=new_right)
936 )
939# Text in Plotly is sized in screen pixels with no native "data unit" mode, so
940# to keep word labels true-to-scale we convert a real (monitor-pixel) font size
941# into the figure's screen pixels using the same scale the boxes/fixations use.
942_MIN_LABEL_PX = 1.0
944# Advance-width / em of a monospaced glyph, used to back the box *width* cap that
945# stops long words from colliding when the on-screen font is a touch wider than
946# the one the experiment was rendered in. Latin monospace (DejaVu Sans Mono,
947# Courier, …) ≈ 0.6 em; in a full-width CJK monospace (Noto Sans Mono CJK) the CJK
948# glyphs are a full square (1.0 em) while Latin glyphs are half-width (0.5 em).
949# Reading stimuli are monospaced (OneStop, MultiplEYE), so summing per-character
950# advances recovers a per-word em (≈ the font size) even for mixed CJK+Latin runs
951# (a Chinese paragraph with an English URL), which a single global aspect can't.
952_MONO_ASPECT = 0.6
953_CJK_LATIN_ASPECT = 0.5
954_FULLWIDTH_ASPECT = 1.0
955_WIDTH_FIT_MARGIN = 0.92 # leave a sliver of horizontal padding inside each box
958def _is_fullwidth(ch: str) -> bool:
959 """Whether ``ch`` is an East-Asian wide / full-width glyph (≈ 1 em advance)."""
960 o = ord(ch)
961 return (
962 0x1100 <= o <= 0x115F # Hangul Jamo
963 or 0x2E80 <= o <= 0x303E # CJK radicals / Kangxi / CJK symbols & punct
964 or 0x3041 <= o <= 0x33FF # Hiragana, Katakana, CJK symbols
965 or 0x3400 <= o <= 0x4DBF # CJK Unified Ext A
966 or 0x4E00 <= o <= 0x9FFF # CJK Unified
967 or 0xA000 <= o <= 0xA4CF # Yi
968 or 0xAC00 <= o <= 0xD7A3 # Hangul syllables
969 or 0xF900 <= o <= 0xFAFF # CJK compatibility
970 or 0xFF00 <= o <= 0xFF60 # full-width forms
971 or 0xFFE0 <= o <= 0xFFE6 # full-width signs
972 )
975def _latin_advance(words: pd.DataFrame) -> float:
976 """Per-em advance of *Latin* glyphs for this corpus' font.
978 In a full-width CJK monospace (Noto Sans Mono CJK — MultiplEYE) Latin glyphs
979 are half-width (0.5 em); in a plain Latin monospace (Courier/DejaVu — OneStop)
980 they're ≈ 0.6 em. Detected from whether the labels are CJK-heavy, so a Chinese
981 corpus' embedded English isn't measured with the wrong cell width.
982 """
983 if "text" not in words.columns:
984 return _MONO_ASPECT
985 # dropna first: with the Arrow `str` dtype, `.astype(str)` leaves a NaN as a
986 # float (it doesn't stringify it), which would break the join + char scan.
987 text = "".join(words["text"].dropna().astype(str).tolist())
988 if not text:
989 return _MONO_ASPECT
990 wide = sum(_is_fullwidth(ch) for ch in text)
991 return _CJK_LATIN_ASPECT if wide >= 0.3 * len(text) else _MONO_ASPECT
994def _line_pitch(words: pd.DataFrame) -> float | None:
995 """Median line-to-line distance (data px) of the word boxes.
997 The true-to-scale font budget is a fraction of the *line pitch* (the gap
998 between consecutive baselines), not the box height: some corpora (MultiplEYE)
999 draw AOI boxes tight around the glyph (height ≈ font), while the line slot is
1000 much taller. OneStop's boxes tile the lines (height == pitch), so this returns
1001 the same value there and leaves OneStop sizing unchanged. Falls back to the
1002 median box height when there's only one line or geometry is missing.
1003 """
1004 if words.empty or "y" not in words.columns or "height" not in words.columns:
1005 return None
1006 from .measures import cluster_word_lines
1008 heights = pd.to_numeric(words["height"], errors="coerce")
1009 y_center = pd.to_numeric(words["y"], errors="coerce") + heights.fillna(0) / 2.0
1010 centers = y_center.groupby(cluster_word_lines(words)).median().sort_values()
1011 if len(centers) >= 2:
1012 pitch = float(centers.diff().dropna().median())
1013 if pitch > 0:
1014 return pitch
1015 box_h = float(heights.median()) if heights.notna().any() else None
1016 return box_h if box_h and box_h > 0 else None
1019def _width_fit_font(words: pd.DataFrame) -> float | None:
1020 """Largest font (data px) at which every word still fits its box width.
1022 Word boxes hug the rendered text, so a word's box width equals the sum of its
1023 glyph advances. Each glyph advances 1 em (full-width CJK) or ``_latin_advance``
1024 em (Latin), so ``box_width / Σ advances`` recovers the em ≈ the font size — per
1025 word, which is correct even for a CJK word boxed beside a half-width Latin URL
1026 (a single global aspect would size the line from the narrowest run). The
1027 tightest words bind, so we take a low quantile (robust to one odd box). For an
1028 all-Latin corpus this reduces exactly to the old ``(box_width / n) / aspect``.
1029 Returns None when there's no text/width to measure.
1030 """
1031 if "width" not in words.columns or "text" not in words.columns:
1032 return None
1033 latin_adv = _latin_advance(words)
1034 widths = pd.to_numeric(words["width"], errors="coerce")
1035 ems = []
1036 for text, width in zip(words["text"], widths):
1037 # Skip a NaN label (Arrow `str` keeps it a float, not "nan") or NaN width —
1038 # matches the old vectorized path, where both dropped out before the quantile.
1039 if pd.isna(text) or not np.isfinite(width):
1040 continue
1041 units = sum(
1042 _FULLWIDTH_ASPECT if _is_fullwidth(c) else latin_adv for c in str(text)
1043 )
1044 if units > 0:
1045 ems.append(width / units)
1046 if not ems:
1047 return None
1048 tight = float(pd.Series(ems).quantile(0.05))
1049 return tight * _WIDTH_FIT_MARGIN if tight > 0 else None
1052# A monospace word box is its glyphs plus the same padding on every word — half
1053# the gap to each neighbour (a space, plus any extra word spacing). Line-start
1054# words carry only the right half, and a fixation cross's box none, so a box
1055# agrees when it carries the full padding, half of it or none, and most of a
1056# trial's boxes must agree before the font is read off them.
1057_PADDED_MIN_AGREEMENT = 0.8
1058_PADDED_MIN_WORDS = 5
1059_PADDED_TOL = 0.02 # of one cell, or 1.5 data px, whichever is larger
1062def _padded_monospace_font(words: pd.DataFrame) -> float | None:
1063 """The font (data px) of a monospace layout whose boxes are padded alike:
1064 the character cell (:func:`_padded_monospace_layout`) over the font's
1065 advance. ``None`` when the layout is not one."""
1066 layout = _padded_monospace_layout(words)
1067 return None if layout is None else layout[0] / _latin_advance(words)
1070def _padded_monospace_layout(words: pd.DataFrame) -> tuple[float, float] | None:
1071 """``(cell, padding)`` in data px of a monospace layout padded alike.
1073 Each box is ``n_chars`` cells plus a padding shared by every word, so the
1074 cell is the slope of box width against word length and the font is that
1075 cell over the font's advance — exact, with no margin to guess: the padding
1076 *is* the margin. The slope is read from the *regular* boxes only, leaving
1077 out each line's first box (it carries only the right half of the padding)
1078 and any box starting where another does (a fixation cross over the first
1079 word), since on a short screen those few would tip a median. ``None`` when
1080 the boxes do not agree (proportional fonts, too few words or lengths,
1081 full-width text), so the caller falls back to :func:`_width_fit_font`.
1082 """
1083 if not {"x", "width", "text"} <= set(words.columns):
1084 return None
1085 frame = words.dropna(subset=["text"])
1086 text = frame["text"].astype(str)
1087 if any(_is_fullwidth(ch) for ch in "".join(text.tolist())):
1088 return None
1089 chars = text.str.len().to_numpy(dtype=float)
1090 x = pd.to_numeric(frame["x"], errors="coerce").to_numpy(dtype=float)
1091 width = pd.to_numeric(frame["width"], errors="coerce").to_numpy(dtype=float)
1092 ok = (chars > 0) & np.isfinite(width) & (width > 0) & np.isfinite(x)
1093 if ok.sum() < _PADDED_MIN_WORDS:
1094 return None
1095 regular = ok.copy()
1096 if {"y", "height"} <= set(frame.columns):
1097 from .measures import cluster_word_lines
1099 lines = np.asarray(cluster_word_lines(frame))
1100 for line in pd.unique(lines[ok]):
1101 on_line = np.flatnonzero(ok & (lines == line))
1102 regular[on_line[x[on_line] <= x[on_line].min()]] = False
1103 shared_start = pd.Series(x).duplicated(keep=False).to_numpy()
1104 regular &= ~shared_start
1105 lengths = np.unique(chars[regular])
1106 if regular.sum() < _PADDED_MIN_WORDS - 1 or len(lengths) < 2:
1107 return None
1108 typical = np.array([np.median(width[regular & (chars == n)]) for n in lengths])
1109 i, j = np.triu_indices(len(lengths), k=1)
1110 cell = float(np.median((typical[j] - typical[i]) / (lengths[j] - lengths[i])))
1111 if not np.isfinite(cell) or cell <= 0:
1112 return None
1113 pad = float(np.median(width[regular] - chars[regular] * cell))
1114 if pad < -0.5 * cell:
1115 return None
1116 tol = max(1.5, _PADDED_TOL * cell)
1117 glyphs = chars[ok] * cell
1118 agree = (
1119 (np.abs(width[ok] - glyphs - pad) <= tol)
1120 | (np.abs(width[ok] - glyphs - pad / 2) <= tol)
1121 | (np.abs(width[ok] - glyphs) <= tol)
1122 )
1123 if agree.mean() < _PADDED_MIN_AGREEMENT:
1124 return None
1125 return cell, pad
1128def _word_label_x(words: pd.DataFrame) -> np.ndarray:
1129 """Where each word label is centred: its box's middle (BUG-97) — except a
1130 line's first word in a padded monospace layout.
1132 There the box carries only the right half of the padding (the line starts
1133 at the text), so the word sat flush with the box's left edge in the
1134 experiment; centring it in the box shifted it right by a quarter of the gap.
1135 Such a word is centred on its own letters instead, starting at ``x``.
1136 Right-to-left words and every other layout keep the box's middle.
1137 """
1138 from .measures import cluster_word_lines, word_box_bounds
1140 x0, _, x1, _ = word_box_bounds(words)
1141 label_x = (x0 + x1) / 2.0
1142 layout = _padded_monospace_layout(words) if len(words) else None
1143 if layout is None or not {"y", "height"} <= set(words.columns):
1144 return label_x
1145 cell, pad = layout
1146 chars = words["text"].astype(str).str.len().to_numpy(dtype=float)
1147 width = x1 - x0
1148 half_padded = np.abs(width - chars * cell - pad / 2) <= max(1.5, _PADDED_TOL * cell)
1149 rtl = words.get("right_to_left")
1150 ltr = (
1151 np.ones(len(words), dtype=bool)
1152 if rtl is None
1153 else ~rtl.fillna(False).astype(bool).to_numpy()
1154 )
1155 lines = np.asarray(cluster_word_lines(words))
1156 first = np.zeros(len(words), dtype=bool)
1157 for line in pd.unique(lines):
1158 on_line = np.flatnonzero(lines == line)
1159 first[on_line[x0[on_line] <= np.nanmin(x0[on_line])]] = True
1160 flush = first & half_padded & ltr & (chars > 0)
1161 label_x = label_x.copy()
1162 label_x[flush] = x0[flush] + chars[flush] * cell / 2.0
1163 return label_x
1166def _display_scale(x_range: list, y_range: list, fitted_w: int, fitted_h: int) -> float:
1167 """Screen px per data unit for a fixed-size, equal-aspect spatial plot.
1169 With ``scaleratio=1`` the x and y mappings are identical; we take the min so
1170 rounding can never make text/markers sized through this overflow the boxes.
1171 Returns 1.0 for degenerate ranges.
1172 """
1173 x_span = x_range[1] - x_range[0]
1174 y_span = y_range[0] - y_range[1] # y_range is inverted [y_max, y_min]
1175 if x_span <= 0 or y_span <= 0:
1176 return 1.0
1177 return min(fitted_w / x_span, fitted_h / y_span)
1180def _word_label_font_px(
1181 words: pd.DataFrame,
1182 *,
1183 scale: float,
1184 line_spacing: float,
1185 manual_font_px: float,
1186 scale_text_to_boxes: bool,
1187) -> float:
1188 """Word-label font size in *screen* px so the text stays true-to-scale.
1190 The experiment's font lives in monitor pixels. To keep the rendered glyphs
1191 the same physical fraction of the (data-space) word boxes at any display
1192 size, the font is expressed in data px and multiplied by ``scale`` (screen
1193 px per data unit):
1195 - ``scale_text_to_boxes`` (default): one line of text fills
1196 ``1 / line_spacing`` of the **line pitch** (the median line-to-line distance
1197 from the data, see :func:`_line_pitch`) — *not* the raw box height, which is
1198 only equal to the pitch when the boxes tile the lines. For OneStop the boxes
1199 tile the lines (pitch == height) and ``line_spacing == 3`` (one blank line
1200 above + below), so the budget is height / 3 as before; for corpora whose AOI
1201 boxes hug the glyph (MultiplEYE), the pitch is the right, larger budget. The
1202 size is *also* capped so the longest words still fit their box width (see
1203 :func:`_width_fit_font`), which keeps the font from colliding; the smaller of
1204 the two wins.
1205 **Except** when the boxes hold their word plus the same padding, half the
1206 gap to each neighbour (:func:`_padded_monospace_font`): then the font is
1207 read off the boxes exactly,
1208 and neither the line-spacing guess nor the fit's safety margin applies —
1209 each shrank such text by its own few percent.
1210 - otherwise / no usable boxes: ``manual_font_px`` is treated as the real
1211 monitor font size and scaled the same way.
1212 """
1213 font_data_px = float(manual_font_px)
1214 exact = _padded_monospace_font(words) if scale_text_to_boxes else None
1215 if exact:
1216 font_data_px = exact
1217 elif scale_text_to_boxes and not words.empty and "height" in words.columns:
1218 pitch = _line_pitch(words)
1219 height_fit = pitch / line_spacing if (pitch and line_spacing > 0) else None
1220 width_fit = _width_fit_font(words)
1221 candidates = [c for c in (height_fit, width_fit) if c and c > 0]
1222 if candidates:
1223 font_data_px = min(candidates)
1224 return max(font_data_px * scale, _MIN_LABEL_PX)
1227_QUALITATIVE_PALETTE = [
1228 "#1f77b4",
1229 "#ff7f0e",
1230 "#2ca02c",
1231 "#d62728",
1232 "#9467bd",
1233 "#8c564b",
1234 "#e377c2",
1235 "#7f7f7f",
1236 "#bcbd22",
1237 "#17becf",
1238]
1241def _resolve_marker_colors(
1242 color_data: pd.Series | None,
1243 is_numeric_color: bool,
1244 uniform_color: str = DEFAULT_FIXATION_COLOR,
1245) -> tuple[object, list]:
1246 """Return (marker_color, category_legend) for the fixation scatter trace.
1248 - Numeric color_data is passed straight through (Plotly maps it via colorscale).
1249 - Categorical color_data is mapped to a discrete palette so the picker has
1250 visible effect; the returned legend is a list of (category, hex) pairs the
1251 caller can render as legend-only scatter traces.
1252 - No color_data (VIZ-17's uniform default, or a `color_by` column that isn't
1253 in the frame) paints every marker ``uniform_color``.
1254 """
1255 if color_data is None:
1256 return uniform_color, []
1257 if is_numeric_color:
1258 return color_data, []
1259 series = color_data.fillna("(missing)").astype(str)
1260 unique_vals = list(pd.unique(series))
1261 cat_to_color = {
1262 val: _QUALITATIVE_PALETTE[i % len(_QUALITATIVE_PALETTE)]
1263 for i, val in enumerate(unique_vals)
1264 }
1265 marker_color = [cat_to_color[val] for val in series]
1266 legend = [(val, cat_to_color[val]) for val in unique_vals]
1267 return marker_color, legend
1270def _fixation_category_labels(
1271 fixations: pd.DataFrame,
1272 words: pd.DataFrame | None,
1273 color_by: str | None,
1274 color_by_line: bool = False,
1275) -> pd.Series | None:
1276 """The discrete label each fixation is coloured by, or ``None`` when the
1277 colouring is not discrete (uniform, numeric, or a column the frame lacks).
1279 The same labels the static figure draws: ``"Line N"`` / ``"Out of bounds"``
1280 for colour-by-line, against ``words``' own geometry; a categorical column's
1281 values as strings, ``"(missing)"`` for a gap. Aligned to ``fixations``.
1282 """
1283 if fixations.empty:
1284 return None
1285 if color_by_line or color_by == "line":
1286 if words is None or words.empty:
1287 return None
1288 from .measures import assign_fixation_lines
1290 line_ids = assign_fixation_lines(fixations, words)
1291 return line_ids.map(
1292 lambda v: f"Line {int(v) + 1}" if pd.notna(v) else "Out of bounds"
1293 )
1294 if (
1295 not color_by
1296 or color_by == UNIFORM_COLOR_FIELD
1297 or color_by not in fixations.columns
1298 or pd.api.types.is_numeric_dtype(fixations[color_by])
1299 ):
1300 return None
1301 return fixations[color_by].fillna("(missing)").astype(str)
1304def _shared_category_colors(
1305 labels: Sequence[pd.Series | None],
1306 avoid: Iterable[str] = (),
1307) -> tuple[list[list[str] | None], list[tuple[str, str]]]:
1308 """One category→colour mapping across several scanpaths (Compare, the dual
1309 replay), so a category wears the same colour on A and on B.
1311 Categories are numbered in order of first appearance, A's before B's.
1312 ``avoid`` names the scanpaths' own colours, which outline their markers: a
1313 category never fills in one, or that scanpath's outline would vanish into
1314 its fill. Returns each scanpath's per-row colours (``None`` where it had no
1315 labels) and the shared legend.
1316 """
1317 present = [series for series in labels if series is not None]
1318 if not present:
1319 return [None] * len(labels), []
1320 taken = {str(color).lower() for color in avoid if color}
1321 palette = [c for c in _QUALITATIVE_PALETTE if c.lower() not in taken]
1322 palette = palette or list(_QUALITATIVE_PALETTE)
1323 order = list(pd.unique(pd.concat(present, ignore_index=True)))
1324 cat_to_color = {val: palette[i % len(palette)] for i, val in enumerate(order)}
1325 colors = [
1326 None if series is None else [cat_to_color[val] for val in series]
1327 for series in labels
1328 ]
1329 return colors, [(val, cat_to_color[val]) for val in order]
1332def _add_category_legend(
1333 fig: go.Figure, legend: Sequence[tuple[str, str]], color_label: str
1334) -> None:
1335 """Legend-only entries naming each colour category (``label: value``), with
1336 a "… +N more" entry past the qualitative palette's length."""
1337 limit = len(_QUALITATIVE_PALETTE)
1338 for category, color in list(legend)[:limit]:
1339 fig.add_trace(
1340 go.Scatter(
1341 x=[None],
1342 y=[None],
1343 mode="markers",
1344 marker=dict(
1345 size=10,
1346 color=color,
1347 line=dict(color=FIX_MARKER_OUTLINE, width=0.5),
1348 ),
1349 name=category
1350 if color_label == "line"
1351 else f"{_column_name(color_label)}: {category}",
1352 showlegend=True,
1353 hoverinfo="skip",
1354 meta=_COLORS_LEGEND_META,
1355 )
1356 )
1357 if len(legend) > limit:
1358 fig.add_trace(
1359 go.Scatter(
1360 x=[None],
1361 y=[None],
1362 mode="markers",
1363 marker=dict(size=10, color="#cccccc"),
1364 name=f"… +{len(legend) - limit} more",
1365 showlegend=True,
1366 hoverinfo="skip",
1367 meta=_COLORS_LEGEND_META,
1368 )
1369 )
1372#: How much larger (font px) the outline layer behind a glyph marker is: a text
1373#: glyph takes no stroke, so its outline is a second, larger glyph drawn under it.
1374_GLYPH_OUTLINE_PX = 4.0
1375#: The outline-only form of each glyph shape, for hollow markers.
1376_HOLLOW_GLYPHS = {"♥": "♡"}
1379def _glyph_scatter_traces(
1380 x, y, marker: dict, glyph: str, **top: Any
1381) -> list[go.Scatter]:
1382 """Draw a fixation ``marker`` dict as text glyphs (VIZ-15's ♥), bottom first.
1384 Plotly's marker-symbol enum has no heart, so every builder draws one as
1385 text, from the marker dict it would otherwise have used, keeping what that
1386 dict says: duration→size (an array ``textfont.size``, scaled by
1387 ``FIXATION_GLYPH_SIZE_SCALE``), the colour (a colorscale is sampled to
1388 literal colours — ``textfont.color`` takes none), the opacity, and a
1389 non-default outline (Compare's A/B cue) as a larger glyph underneath in the
1390 outline colour. A hollow marker becomes the outline glyph (♡) in its
1391 outline colour. ``top`` goes onto the glyph layer (name, hover, legend);
1392 the outline layer under it takes no hover and no legend entry.
1394 The two halves are separate so a replay can state the layers once
1395 (:func:`_glyph_layers`) and draw them at each frame's positions
1396 (:func:`_glyph_layer_traces`) without re-sampling the colours.
1397 """
1398 return _glyph_layer_traces(x, y, _glyph_layers(marker, glyph, len(x)), **top)
1401def _glyph_layers(marker: dict, glyph: str, n: int) -> list[dict]:
1402 """What :func:`_glyph_scatter_traces` draws, bottom first, minus positions.
1404 One dict per layer — ``text`` / ``textfont`` / ``opacity`` — with the
1405 colours sampled and the sizes scaled, for ``n`` fixations."""
1406 sizes = np.asarray(marker.get("size"), dtype=float) * FIXATION_GLYPH_SIZE_SCALE
1407 sizes = np.broadcast_to(sizes, (n,)) if sizes.ndim == 0 else sizes
1408 color = marker.get("color")
1409 line = marker.get("line") or {}
1410 layers: list[tuple[str, object, np.ndarray]] = []
1411 if isinstance(color, str) and color == "rgba(0,0,0,0)":
1412 layers.append(
1413 (
1414 _HOLLOW_GLYPHS.get(glyph, glyph),
1415 line.get("color") or FIX_MARKER_OUTLINE,
1416 sizes,
1417 )
1418 )
1419 else:
1420 if (
1421 marker.get("colorscale") is not None
1422 and color is not None
1423 and not isinstance(color, str)
1424 ):
1425 color = _sample_colorscale_colors(
1426 color, marker["colorscale"], marker.get("cmin"), marker.get("cmax")
1427 )
1428 elif color is not None and not isinstance(color, str):
1429 color = list(color)
1430 outline = line.get("color")
1431 if isinstance(outline, str) and outline != FIX_MARKER_OUTLINE:
1432 layers.append((glyph, outline, sizes + _GLYPH_OUTLINE_PX))
1433 layers.append((glyph, color, sizes))
1434 opacity = float(marker.get("opacity", 1.0))
1435 return [
1436 dict(
1437 text=[char] * n,
1438 textfont=dict(color=layer_color, size=list(layer_sizes)),
1439 opacity=opacity,
1440 )
1441 for char, layer_color, layer_sizes in layers
1442 ]
1445def _glyph_layer_traces(
1446 x, y, layers: list[dict], *, make: Callable = go.Scatter, **top: Any
1447) -> list:
1448 """:func:`_glyph_layers`' layers drawn at ``x``/``y``; ``top`` on the last.
1450 ``make=dict`` returns the traces unvalidated, for a replay frame (see the
1451 frame loop in :func:`_render_scanpath_animation`)."""
1452 traces = []
1453 for i, layer in enumerate(layers):
1454 is_top = i == len(layers) - 1
1455 traces.append(
1456 make(
1457 x=x,
1458 y=y,
1459 mode="text",
1460 text=layer["text"],
1461 textfont=layer["textfont"],
1462 textposition="middle center",
1463 opacity=layer["opacity"],
1464 **(
1465 top
1466 if is_top
1467 else dict(
1468 hoverinfo="skip",
1469 showlegend=False,
1470 legendgroup=top.get("legendgroup"),
1471 )
1472 ),
1473 )
1474 )
1475 return traces
1478def _glyph_colorbar_trace(marker: dict, values) -> go.Scatter | None:
1479 """The colour bar a glyph marker's numeric colouring would have drawn.
1481 A text glyph carries no colorscale, so the bar rides on an invisible
1482 two-point marker trace pinned to the same range."""
1483 if not marker.get("showscale") or marker.get("colorscale") is None:
1484 return None
1485 numeric = pd.to_numeric(pd.Series(list(values)), errors="coerce")
1486 lo = marker.get("cmin")
1487 hi = marker.get("cmax")
1488 lo = float(numeric.min()) if lo is None else float(lo)
1489 hi = float(numeric.max()) if hi is None else float(hi)
1490 if not (np.isfinite(lo) and np.isfinite(hi)):
1491 return None
1492 return go.Scatter(
1493 x=[None, None],
1494 y=[None, None],
1495 mode="markers",
1496 marker=dict(
1497 color=[lo, hi],
1498 colorscale=marker["colorscale"],
1499 cmin=lo,
1500 cmax=hi,
1501 showscale=True,
1502 colorbar=marker.get("colorbar"),
1503 size=0,
1504 ),
1505 name="color scale",
1506 showlegend=False,
1507 hoverinfo="skip",
1508 )
1511def _marker_symbol(symbol: str | None) -> str:
1512 """A ``marker.symbol`` Plotly will accept.
1514 The VIZ-15 glyph shapes (♥) aren't in Plotly's symbol enum — the static
1515 figure draws them as text instead — so anywhere that *must* hand Plotly a
1516 marker symbol falls back to the default rather than raising.
1517 """
1518 if not symbol or symbol in FIXATION_GLYPH_SYMBOLS:
1519 return DEFAULT_FIXATION_SYMBOL
1520 return symbol
1523def _duration_bounds(duration_range) -> tuple[float, float]:
1524 """The fixed scale's (lo, hi) in ms, ordered and never empty."""
1525 lo, hi = sorted(float(v) for v in duration_range)
1526 lo = max(lo, 1.0) # log needs a positive floor; no fixation is shorter
1527 return lo, max(hi, lo + 1.0)
1530def _scale_transform(scale: str):
1531 try:
1532 return {"sqrt": np.sqrt, "linear": lambda d: d, "log": np.log}[scale]
1533 except KeyError:
1534 raise ValueError(
1535 f"Unknown marker_size_scale {scale!r}; choose one of "
1536 f"{', '.join(MARKER_SIZE_SCALES)}."
1537 ) from None
1540def _compute_marker_sizes(
1541 durations: pd.Series,
1542 size_range: tuple[int, int] = DEFAULT_MARKER_SIZE_RANGE,
1543 scale: str = DEFAULT_MARKER_SIZE_SCALE,
1544 duration_range: tuple[float, float] = DEFAULT_MARKER_DURATION_RANGE,
1545) -> np.ndarray:
1546 """Map fixation durations to marker sizes (px diameter).
1548 A fixed ``scale`` ("sqrt" / "linear" / "log") maps ``duration_range`` (ms)
1549 onto ``size_range`` through that curve, the same for every figure: a
1550 duration at or below the lower bound gets the smallest marker, at or above
1551 the upper bound the largest, so a fixation's size never depends on the
1552 other fixations drawn beside it. ``"relative"`` is the original scale —
1553 linear between this set's own shortest and longest durations — which is
1554 why it is only comparable within one figure.
1555 """
1556 durations = pd.to_numeric(durations, errors="coerce").fillna(0)
1557 min_size, max_size = size_range
1558 if scale == "relative":
1559 d_min, d_max = float(durations.min()), float(durations.max())
1560 if d_max - d_min > 0:
1561 return np.interp(durations, (d_min, d_max), (min_size, max_size))
1562 return np.full(len(durations), (min_size + max_size) / 2)
1563 transform = _scale_transform(scale)
1564 lo, hi = _duration_bounds(duration_range)
1565 clipped = np.clip(durations.to_numpy(dtype=float), lo, hi)
1566 f_lo, f_hi = float(transform(lo)), float(transform(hi))
1567 frac = (transform(clipped) - f_lo) / (f_hi - f_lo)
1568 return min_size + frac * (max_size - min_size)
1571def _settings_size_scale(settings: FigureSettings) -> dict:
1572 """The duration-scale keywords of :func:`_compute_marker_sizes`."""
1573 return {
1574 "scale": settings.marker_size_scale,
1575 "duration_range": settings.marker_duration_range,
1576 }
1579def _duration_key_references(duration_range) -> list[tuple[float, str]]:
1580 """The durations a size key draws, each with its label.
1582 The two bounds (labelled ``≤`` / ``≥``, because everything beyond them
1583 clamps) plus the round values 100/200/400/800/1600 ms that fall strictly
1584 between them — at most three of those, so the key stays compact."""
1585 lo, hi = _duration_bounds(duration_range)
1586 inner = [d for d in (100, 200, 400, 800, 1600) if lo < d < hi][:3]
1587 if not inner and hi - lo > 2:
1588 inner = [round((lo + hi) / 2)]
1589 refs = [(lo, f"≤{lo:g}")]
1590 refs += [(float(d), f"{d:g}") for d in inner]
1591 refs.append((hi, f"≥{hi:g} ms"))
1592 return refs
1595# Name of the size key's label annotations, so the layer split can find them.
1596_SIZE_KEY_NAME = "duration_size_key"
1597# The Illustration stamp shares the size key's bottom-right corner.
1598_ILLUSTRATION_LABEL_NAME = "illustration_label"
1601def _stack_bottom_right(fig: go.Figure) -> None:
1602 """Lift the Illustration stamp above the duration size key when both are drawn.
1604 Both sit in the plot's bottom-right corner, and either can be added first
1605 (the public builders stamp the label inside ``make_scanpath_figure`` and add
1606 the key after; the app's replay does the reverse), so each calls this once
1607 it is on the figure. The key's height is read off its own circles — the
1608 only pixel-sized circles in paper coordinates — so the stamp clears the
1609 largest one at any size range."""
1610 key_top = [
1611 float(sh.y1)
1612 for sh in fig.layout.shapes or ()
1613 if sh.type == "circle"
1614 and sh.yref == "paper"
1615 and sh.ysizemode == "pixel"
1616 # Only a key inside the bottom-right corner shares the stamp's spot.
1617 and sh.xanchor == 1
1618 and sh.yanchor == 0
1619 and float(sh.x1) <= 0
1620 and float(sh.y0) >= 0
1621 ]
1622 if not key_top:
1623 return
1624 for ann in fig.layout.annotations or ():
1625 if ann.name == _ILLUSTRATION_LABEL_NAME:
1626 ann.yshift = max(key_top) + 4.0
1629# --- Legend layout (where each legend sits) ---------------------------------
1630#
1631# Each legend *kind* can be moved to a spot of its own, sized, and laid out as a
1632# stack or a row, on top of its own layer's show/hide switch. A kind left on
1633# "auto" throughout is drawn exactly where it always was: every figure built
1634# before this setting existed is unchanged. A kind moved anywhere gets its own
1635# Plotly legend (``legend2`` …), and a spot outside the plot grows the figure on
1636# that side, so the equal-aspect plot region — and the true-to-scale text with
1637# it — keeps its size (the same rule as `_decoration_margins`).
1639#: The size key's own spot when left on "auto": inside, bottom-right.
1640_SIZE_KEY_AUTO_POSITION = "bottom-right"
1641_COLORS_LEGEND_META = "legend:colors"
1642_LEGEND_IDS = {"compare": "legend2", "saccades": "legend3", "colors": "legend4"}
1643_LEGEND_GAP_PX = 8
1644_OUTSIDE = ("above", "below", "left", "right")
1645_CORNERS = ("top-left", "top-right", "bottom-left", "bottom-right")
1648def normalize_legend_layout(layout: Mapping | None) -> dict:
1649 """``{kind: {"position", "arrangement", "size"}}`` with every kind present.
1651 Unknown kinds, positions and arrangements raise rather than being dropped,
1652 so a typo in a script or a link can't quietly draw the default. ``size``
1653 is the legend's text size in px, or ``None`` for the figure's own.
1654 """
1655 out = {
1656 kind: {"position": "auto", "arrangement": "auto", "size": None}
1657 for kind in LEGEND_KINDS
1658 }
1659 for kind, spec in dict(layout or {}).items():
1660 if kind not in out:
1661 raise ValueError(
1662 f"Unknown legend {kind!r}; expected one of {', '.join(LEGEND_KINDS)}."
1663 )
1664 spec = dict(spec or {})
1665 unknown = set(spec) - {"position", "arrangement", "size"}
1666 if unknown:
1667 raise ValueError(
1668 f"Unknown legend setting(s) {sorted(unknown)} for {kind!r}; "
1669 "expected position, arrangement, size."
1670 )
1671 position = str(spec.get("position") or "auto")
1672 if position not in LEGEND_POSITIONS:
1673 raise ValueError(
1674 f"Legend position {position!r} for {kind!r}; expected one of "
1675 f"{', '.join(LEGEND_POSITIONS)}."
1676 )
1677 arrangement = str(spec.get("arrangement") or "auto")
1678 if arrangement not in LEGEND_ARRANGEMENTS:
1679 raise ValueError(
1680 f"Legend arrangement {arrangement!r} for {kind!r}; expected one "
1681 f"of {', '.join(LEGEND_ARRANGEMENTS)}."
1682 )
1683 size = spec.get("size")
1684 if size is not None:
1685 size = int(size)
1686 if size <= 0:
1687 raise ValueError(f"Legend size for {kind!r} must be positive.")
1688 out[kind] = {"position": position, "arrangement": arrangement, "size": size}
1689 return out
1692def parse_legend_spec(text: str) -> dict:
1693 """``"right,stacked,14"`` → ``{"position", "arrangement"?, "size"?}``.
1695 The spelling shared by ``render --legend KIND=…``, the ``legend_<kind>``
1696 link parameters and the code snippet: a spot first, then an arrangement
1697 and a text size in either order, each recognised by its value. Raises
1698 ``ValueError`` on anything else.
1699 """
1700 head, *rest = [part.strip().lower() for part in str(text).split(",")]
1701 spec: dict = {"position": head or "auto"}
1702 for part in (p for p in rest if p):
1703 if part in LEGEND_ARRANGEMENTS:
1704 spec["arrangement"] = part
1705 elif part.isdigit():
1706 spec["size"] = int(part)
1707 else:
1708 raise ValueError(
1709 f"{part!r} is neither an arrangement "
1710 f"({', '.join(LEGEND_ARRANGEMENTS[1:])}) nor a text size."
1711 )
1712 normalize_legend_layout({"compare": spec}) # validates the values
1713 return spec
1716def legend_spec_text(spec: Mapping) -> str:
1717 """The inverse of :func:`parse_legend_spec`, for a normalized entry."""
1718 parts = [str(spec["position"])]
1719 if spec.get("arrangement", "auto") != "auto":
1720 parts.append(str(spec["arrangement"]))
1721 if spec.get("size") is not None:
1722 parts.append(str(int(spec["size"])))
1723 return ",".join(parts)
1726def _legend_is_moved(spec: Mapping) -> bool:
1727 return (
1728 spec["position"] != "auto"
1729 or spec["arrangement"] != "auto"
1730 or spec["size"] is not None
1731 )
1734def _trace_legend_kind(trace, comparing: bool) -> str:
1735 if trace.legendgroup == "saccade_type":
1736 return "saccades"
1737 if trace.meta == _COLORS_LEGEND_META or not comparing:
1738 return "colors"
1739 return "compare"
1742def _legend_extent(names: list, font_px: float, horizontal: bool) -> tuple:
1743 """A legend's estimated ``(width, height)`` in px, for reserving room."""
1744 row = max(font_px * 1.3, 20.0) + 4.0
1745 widths = [40.0 + 0.6 * font_px * len(str(n)) for n in names] or [0.0]
1746 if horizontal:
1747 return sum(widths) + 10.0, row + 10.0
1748 return max(widths) + 10.0, row * len(names) + 10.0
1751def _plot_px(fig: go.Figure) -> tuple:
1752 """The plot region's ``(width, height)`` in px, and the margins."""
1753 m = fig.layout.margin
1754 margin = {k: float(getattr(m, k) or 0) for k in ("l", "r", "t", "b")}
1755 width = float(fig.layout.width or 0) - margin["l"] - margin["r"]
1756 height = float(fig.layout.height or 0) - margin["t"] - margin["b"]
1757 return max(width, 1.0), max(height, 1.0), margin
1760def _grow(fig: go.Figure, side: str, px: float) -> None:
1761 """Add ``px`` of margin on one side, growing the figure by as much, so the
1762 plot region keeps its size."""
1763 if not px or not fig.layout.width or not fig.layout.height:
1764 return
1765 m = fig.layout.margin
1766 setattr(m, side, float(getattr(m, side) or 0) + px)
1767 if side in ("l", "r"):
1768 fig.layout.width = float(fig.layout.width) + px
1769 else:
1770 fig.layout.height = float(fig.layout.height) + px
1773def apply_legend_layout(
1774 fig: go.Figure, layout: Mapping | None, *, comparing: bool = False
1775) -> go.Figure:
1776 """Move each legend kind the user placed into a Plotly legend of its own.
1778 Kinds still on "auto" stay in the figure's default ``legend``; the size key
1779 is placed by :func:`_add_duration_size_key`, not here. Legends sharing a
1780 side are laid out one after another along it, and an outside side reserves
1781 room for the widest (or tallest) of them.
1782 """
1783 specs = normalize_legend_layout(layout)
1784 moved = {
1785 kind
1786 for kind in ("compare", "saccades", "colors")
1787 if _legend_is_moved(specs[kind])
1788 }
1789 if not moved:
1790 return fig
1791 names: dict = {kind: [] for kind in moved}
1792 titled: set = set()
1793 for trace in fig.data:
1794 kind = _trace_legend_kind(trace, comparing)
1795 if kind not in moved:
1796 continue
1797 trace.legend = _LEGEND_IDS[kind]
1798 if trace.showlegend is False:
1799 continue
1800 title = trace.legendgrouptitle.text if trace.legendgrouptitle else None
1801 if title and trace.legendgroup not in titled:
1802 titled.add(trace.legendgroup)
1803 names[kind].append(title)
1804 if trace.name:
1805 names[kind].append(trace.name)
1806 base_font = float((fig.layout.font and fig.layout.font.size) or 12)
1807 plot_w, plot_h, margin = _plot_px(fig)
1808 gap = _LEGEND_GAP_PX
1809 along = {side: 0.0 for side in (*_OUTSIDE, *_CORNERS)}
1810 reserve = {side: 0.0 for side in _OUTSIDE}
1811 for kind in ("compare", "saccades", "colors"):
1812 if kind not in moved or not names[kind]:
1813 continue
1814 spec = specs[kind]
1815 position = "above" if spec["position"] == "auto" else spec["position"]
1816 horizontal = (
1817 spec["arrangement"] == "side-by-side"
1818 if spec["arrangement"] != "auto"
1819 else position in ("above", "below")
1820 )
1821 font_px = float(spec["size"] or base_font)
1822 w, h = _legend_extent(names[kind], font_px, horizontal)
1823 cfg: dict = {
1824 "orientation": "h" if horizontal else "v",
1825 "bgcolor": "rgba(255,255,255,0.75)",
1826 }
1827 if spec["size"]:
1828 cfg["font"] = {"size": font_px}
1829 if position == "above":
1830 cfg.update(
1831 xanchor="right",
1832 x=1 - along["above"] / plot_w,
1833 yanchor="bottom",
1834 y=1 + gap / plot_h,
1835 )
1836 along["above"] += w + gap
1837 reserve["above"] = max(reserve["above"], h + gap)
1838 elif position == "below":
1839 cfg.update(
1840 xanchor="left",
1841 x=along["below"] / plot_w,
1842 yanchor="top",
1843 y=-(margin["b"] + gap) / plot_h,
1844 )
1845 along["below"] += w + gap
1846 reserve["below"] = max(reserve["below"], h + gap)
1847 elif position == "left":
1848 cfg.update(
1849 xanchor="right",
1850 x=-(margin["l"] + gap) / plot_w,
1851 yanchor="top",
1852 y=1 - along["left"] / plot_h,
1853 )
1854 along["left"] += h + gap
1855 reserve["left"] = max(reserve["left"], w + gap)
1856 elif position == "right":
1857 cfg.update(
1858 xanchor="left",
1859 x=1 + (margin["r"] + gap) / plot_w,
1860 yanchor="top",
1861 y=1 - along["right"] / plot_h,
1862 )
1863 along["right"] += h + gap
1864 reserve["right"] = max(reserve["right"], w + gap)
1865 else:
1866 top = position.startswith("top")
1867 left = position.endswith("left")
1868 inset = along[position]
1869 cfg.update(
1870 xanchor="left" if left else "right",
1871 x=gap / plot_w if left else 1 - gap / plot_w,
1872 yanchor="top" if top else "bottom",
1873 y=1 - (gap + inset) / plot_h if top else (gap + inset) / plot_h,
1874 )
1875 along[position] += h + gap
1876 fig.update_layout({_LEGEND_IDS[kind]: cfg})
1877 # The default legend's strip above the plot: when every entry has moved out
1878 # of it and the strip is exactly that reserve (the single-trial figures —
1879 # a comparison's top margin also holds its title), it is handed back.
1880 default_left = any(
1881 t.legend in (None, "legend") and t.showlegend is not False and t.name
1882 for t in fig.data
1883 )
1884 if not default_left and margin["t"] == _LEGEND_RESERVE_PX and not comparing:
1885 _grow(fig, "t", -_LEGEND_RESERVE_PX)
1886 margin["t"] = 0.0
1887 # Above: the default legend's own reserve may already cover it.
1888 _grow(fig, "t", max(0.0, reserve["above"] - margin["t"]))
1889 _grow(fig, "b", reserve["below"])
1890 _grow(fig, "l", reserve["left"])
1891 _grow(fig, "r", reserve["right"])
1892 return fig
1895def _size_key_layout(layout: Mapping | None) -> dict:
1896 """The size key's resolved spot, arrangement and label size."""
1897 spec = normalize_legend_layout(layout)["size_key"]
1898 position = spec["position"]
1899 return {
1900 "position": _SIZE_KEY_AUTO_POSITION if position == "auto" else position,
1901 "stacked": spec["arrangement"] == "stacked",
1902 "size": spec["size"],
1903 }
1906def _add_duration_size_key(
1907 fig: go.Figure,
1908 size_range: tuple[int, int],
1909 scale: str,
1910 duration_range,
1911 *,
1912 font_family: str | None = None,
1913 legend_layout: Mapping | None = None,
1914) -> None:
1915 """Draw the fixed duration scale's key: reference circles labelled in ms.
1917 Pixel-sized shapes anchored to a corner of the plot (paper coordinates), so
1918 each circle is exactly the diameter a fixation of that duration gets — at
1919 any canvas size and through every export path, since they are layout
1920 shapes rather than a trace. ``legend_layout``'s ``size_key`` entry picks the
1921 spot (inside bottom-right by default), a row or a column, and the labels'
1922 size; the circles themselves never scale, since their size *is* the key.
1923 An outside spot grows the figure so the plot region keeps its size.
1924 Nothing is drawn for the relative scale: its sizes mean something only
1925 inside one figure."""
1926 if scale == "relative":
1927 return
1928 placement = _size_key_layout(legend_layout)
1929 refs = _duration_key_references(duration_range)
1930 sizes = [
1931 float(v)
1932 for v in _compute_marker_sizes(
1933 pd.Series([d for d, _ in refs]), size_range, scale, duration_range
1934 )
1935 ]
1936 label_font = float(placement["size"] or 10)
1937 label_px = 1.6 * label_font
1938 biggest = float(size_range[1])
1939 pad = 8.0
1940 n = len(refs)
1941 # The key's own box, in px, origin bottom-left and y up: where each circle's
1942 # centre and each label sit inside it.
1943 if placement["stacked"]:
1944 row = max(biggest, label_px) + 6.0
1945 label_w = 0.6 * label_font * max(len(label) for _, label in refs)
1946 w = pad + biggest + 6.0 + label_w + pad
1947 h = 2 * pad + n * row
1948 centres = [(pad + biggest / 2, h - pad - (i + 0.5) * row) for i in range(n)]
1949 labels = [(pad + biggest + 6.0, cy, "left") for _, cy in centres]
1950 else:
1951 slot = max(biggest, 3.0 * label_font) + 8.0
1952 w = 2 * pad + n * slot
1953 h = pad + label_px + biggest + pad
1954 cy = pad + label_px + biggest / 2.0
1955 centres = [(pad + (i + 0.5) * slot, cy) for i in range(n)]
1956 labels = [(cx, pad + label_px / 2.0, "center") for cx, _ in centres]
1957 left, bottom, ax, ay = _size_key_anchor(fig, placement["position"], w, h)
1958 for (_, label), size, (cx, cy), (lx, ly, align) in zip(
1959 refs, sizes, centres, labels
1960 ):
1961 r = size / 2.0
1962 fig.add_shape(
1963 type="circle",
1964 xref="paper",
1965 yref="paper",
1966 xsizemode="pixel",
1967 ysizemode="pixel",
1968 xanchor=ax,
1969 yanchor=ay,
1970 x0=left + cx - r,
1971 x1=left + cx + r,
1972 y0=bottom + cy - r,
1973 y1=bottom + cy + r,
1974 line=dict(color="#555555", width=1),
1975 fillcolor="rgba(120,120,120,0.35)",
1976 layer="above",
1977 # Rides the fixations layer of a separable export (VIZ-5).
1978 name=_shape_layer_tag("fixations"),
1979 )
1980 fig.add_annotation(
1981 x=ax,
1982 y=ay,
1983 xref="paper",
1984 yref="paper",
1985 xshift=left + lx,
1986 yshift=bottom + ly,
1987 text=label,
1988 showarrow=False,
1989 xanchor=align,
1990 yanchor="middle",
1991 font=dict(
1992 size=label_font, color="#444444", family=font_family or FONT_FAMILY
1993 ),
1994 name=_SIZE_KEY_NAME,
1995 )
1996 _stack_bottom_right(fig)
1999def _size_key_anchor(fig: go.Figure, position: str, w: float, h: float) -> tuple:
2000 """``(left, bottom, anchor_x, anchor_y)``: where the size key's box goes.
2002 ``left`` / ``bottom`` are px from the paper anchor to the box's bottom-left
2003 corner. An outside spot sits beyond whatever already occupies that margin
2004 (a colour bar, the transport controls, another legend) and grows it.
2005 """
2006 gap = _LEGEND_GAP_PX
2007 if position in _CORNERS:
2008 top = position.startswith("top")
2009 right = position.endswith("right")
2010 return (
2011 (-w if right else 0.0),
2012 (-h if top else 0.0),
2013 (1 if right else 0),
2014 (1 if top else 0),
2015 )
2016 _plot_w, _plot_h, margin = _plot_px(fig)
2017 if position == "right":
2018 _grow(fig, "r", w + gap)
2019 return margin["r"] + gap, 0.0, 1, 0
2020 if position == "left":
2021 _grow(fig, "l", w + gap)
2022 return -(margin["l"] + gap + w), 0.0, 0, 0
2023 if position == "above":
2024 _grow(fig, "t", max(0.0, h + gap - margin["t"]))
2025 return 0.0, gap, 0, 1
2026 # below
2027 _grow(fig, "b", h + gap)
2028 return -w, -(margin["b"] + gap + h), 1, 0
2031# VIZ-9 "linear reading" mode: draw saccades as upward arcs instead of straight
2032# connectors. The apex rises by _ARCH_FRAC of the saccade's horizontal span; each
2033# arc is sampled into _ARCH_SAMPLES points so it stays smooth in the true-scale
2034# embed. `arch_frac=None` (the default everywhere) keeps the straight connectors.
2035_ARCH_FRAC = 0.28
2036_ARCH_SAMPLES = 20
2039def _arch_control_point(
2040 x0: float, y0: float, x1: float, y1: float, frac: float
2041) -> tuple[float, float]:
2042 """Control point of the quadratic Bézier arch drawn between two fixations.
2044 Single source of truth for the arch geometry: ``_arch_points`` samples the
2045 curve it defines and ``_arch_point_and_tangent`` (the arrowheads, BUG-9)
2046 evaluates the same curve, so the marker can never drift off the drawn line.
2047 Screen y grows downward, so the control point sits *above* the chord."""
2048 return (x0 + x1) / 2.0, min(y0, y1) - frac * abs(x1 - x0)
2051def _arch_points(
2052 x0: float, y0: float, x1: float, y1: float, frac: float, n: int = _ARCH_SAMPLES
2053) -> tuple[list, list]:
2054 """Sample a quadratic Bézier arch from (x0,y0) to (x1,y1), bulging upward.
2056 Screen y grows downward, so the control point is *above* the chord (smaller
2057 y). Returns (xs, ys) of length ``n`` including both endpoints. A NaN endpoint
2058 propagates to NaN samples, which Plotly simply skips."""
2059 cx, cy = _arch_control_point(x0, y0, x1, y1, frac)
2060 ts = np.linspace(0.0, 1.0, n)
2061 xs = ((1 - ts) ** 2) * x0 + 2 * (1 - ts) * ts * cx + (ts**2) * x1
2062 ys = ((1 - ts) ** 2) * y0 + 2 * (1 - ts) * ts * cy + (ts**2) * y1
2063 return xs.tolist(), ys.tolist()
2066def _arch_apex_y(x0: float, y0: float, x1: float, y1: float, frac: float) -> float:
2067 """Topmost (smallest) ``y`` reached by the drawn arch — its true apex (BUG-13).
2069 Minimises ``y(t)`` over the SAME quadratic ``_arch_points`` samples and
2070 ``_arch_point_and_tangent`` evaluates, instead of reading off one sampled
2071 parameter. Writing the curve as ``y(t) = y0 + 2t(cy - y0) + t^2 * d`` with
2072 ``d = y0 - 2*cy + y1`` gives the closed-form extremum ``t* = (y0 - cy)/d`` and
2073 the value ``y0 - (y0 - cy)^2 / d``. The control point is at or above both
2074 endpoints (``cy <= min(y0, y1)``), so ``d >= 0`` and ``t*`` always lands in
2075 ``[0, 1]``.
2077 For a level saccade the peak is at ``t = 0.5`` (the old estimate), but as the
2078 endpoints diverge in ``y`` it slides toward the higher one and rises well
2079 above the ``t = 0.5`` point — which is why a wide, steeply-sloped arc used to
2080 clip against the top of the view.
2081 """
2082 _, cy = _arch_control_point(x0, y0, x1, y1, frac)
2083 denom = y0 - 2.0 * cy + y1
2084 if denom <= 0.0:
2085 # Degenerate: zero rise and level endpoints — the "arch" is a flat line.
2086 return min(y0, y1)
2087 t = (y0 - cy) / denom
2088 if t <= 0.0:
2089 return y0
2090 if t >= 1.0:
2091 return y1
2092 return y0 - (y0 - cy) ** 2 / denom
2095def _extend_segment(
2096 xs: list, ys: list, x0, y0, x1, y1, arch_frac: float | None
2097) -> None:
2098 """Append one saccade segment (straight, or an arch when ``arch_frac``) to the
2099 None-separated ``xs``/``ys`` accumulators."""
2100 if arch_frac is None:
2101 xs.extend([x0, x1, None])
2102 ys.extend([y0, y1, None])
2103 else:
2104 ax, ay = _arch_points(x0, y0, x1, y1, arch_frac)
2105 xs.extend(ax + [None])
2106 ys.extend(ay + [None])
2109def _saccade_segments(
2110 fix_df: pd.DataFrame,
2111 x_col: str,
2112 y_col: str,
2113 arch_frac: float | None = None,
2114) -> tuple[list, list]:
2115 """Return concatenated x/y arrays separated by None for a single saccade trace.
2117 ``arch_frac`` (VIZ-9) draws each segment as an upward arc instead of a straight
2118 line."""
2119 if len(fix_df) < 2:
2120 return [], []
2121 ordered = fix_df.sort_values("timestamp_ms")
2122 xs: list = []
2123 ys: list = []
2124 x_vals = ordered[x_col].tolist()
2125 y_vals = ordered[y_col].tolist()
2126 for i in range(len(ordered) - 1):
2127 _extend_segment(
2128 xs, ys, x_vals[i], y_vals[i], x_vals[i + 1], y_vals[i + 1], arch_frac
2129 )
2130 return xs, ys
2133def _saccade_segments_by_class(
2134 fix_df: pd.DataFrame,
2135 x_col: str,
2136 y_col: str,
2137 classes: pd.Series,
2138 arch_frac: float | None = None,
2139) -> dict:
2140 """Group saccade segments by reading class → ``{class: (xs, ys)}`` (VIZ-8).
2142 Each segment (fixation i → i+1) takes the class of its *departing* fixation
2143 (``classes[i]``), so it matches ``measures.classify_saccades``. Segments with
2144 no class (``None``/NaN — the last fixation, or an unclassifiable one) fall
2145 into ``"other"`` so they still draw. Each class's arrays are None-separated,
2146 ready for one Scatter trace per class. ``arch_frac`` (VIZ-9) arcs each
2147 segment."""
2148 if len(fix_df) < 2:
2149 return {}
2150 ordered = fix_df.sort_values("timestamp_ms")
2151 cls = classes.reindex(ordered.index).tolist()
2152 x_vals = ordered[x_col].tolist()
2153 y_vals = ordered[y_col].tolist()
2154 out: dict = {}
2155 for i in range(len(ordered) - 1):
2156 c = cls[i]
2157 if pd.isna(c):
2158 c = "other"
2159 xs, ys = out.setdefault(c, ([], []))
2160 _extend_segment(
2161 xs, ys, x_vals[i], y_vals[i], x_vals[i + 1], y_vals[i + 1], arch_frac
2162 )
2163 return out
2166def _snap_fixations_to_words(
2167 fixations: pd.DataFrame, words: pd.DataFrame, x_field: str, y_field: str
2168) -> pd.DataFrame:
2169 """Return a copy of ``fixations`` with each fixation moved to the top-centre of
2170 the word it lands on (VIZ-9 "linear reading" mode).
2172 Fixations with no assigned word keep their raw position. Uses a precomputed
2173 ``word_id`` column when present, else assigns via bounding-box containment."""
2174 out = fixations.copy()
2175 if "word_id" not in words.columns:
2176 return out
2177 if (
2178 "word_id" in out.columns
2179 and pd.to_numeric(out["word_id"], errors="coerce").notna().any()
2180 ):
2181 wid = pd.to_numeric(out["word_id"], errors="coerce")
2182 else:
2183 from .measures import assign_fixations_to_words
2185 wid = pd.to_numeric(
2186 assign_fixations_to_words(out, words)["word_id"], errors="coerce"
2187 )
2188 # Snap above the middle of the word's box, where its label is drawn (BUG-97).
2189 # Render-only: which word a fixation belongs to is still
2190 # `assign_fixations_to_words`, against the same boxes.
2191 from .measures import word_box_bounds
2193 x0, _, x1, _ = word_box_bounds(words)
2194 cx_by_id = dict(zip(words["word_id"], (x0 + x1) / 2.0))
2195 top_by_id = dict(zip(words["word_id"], pd.to_numeric(words["y"], errors="coerce")))
2196 snap_x = wid.map(cx_by_id)
2197 snap_y = wid.map(top_by_id)
2198 out[x_field] = snap_x.where(snap_x.notna(), out[x_field])
2199 out[y_field] = snap_y.where(snap_y.notna(), out[y_field])
2200 return out
2203# Saccades shorter than this fraction of the fixation-extent diagonal get no
2204# direction arrow — their heading is sub-pixel noise (refixations on one word).
2205_ARROW_MIN_LEN_FRAC = 0.005
2207# Bézier parameter at which the arrowhead sits on an arched saccade (BUG-9).
2208_ARROW_ARCH_T = 0.5
2211def _arch_point_and_tangent(
2212 x0: float, y0: float, x1: float, y1: float, frac: float, t: float = _ARROW_ARCH_T
2213) -> tuple[float, float, float, float]:
2214 """Point on the drawn arch at Bézier parameter ``t``, plus the curve's tangent.
2216 Evaluates the very curve ``_arch_points`` samples (same
2217 ``_arch_control_point``), so an arrowhead placed here lands exactly on the
2218 rendered line. Returns ``(x, y, dx, dy)`` where ``(dx, dy)`` is the
2219 (unnormalized) tangent B'(t).
2221 Note the quadratic's identity at ``t=0.5``: B'(0.5) == P1 - P0, i.e. the
2222 tangent at the parameter midpoint is parallel to the chord regardless of the
2223 control point. So arcing a saccade moves the arrowhead *up onto* the curve
2224 (the visible defect) without rotating it — which is exactly right, the curve
2225 really is chord-parallel there."""
2226 cx, cy = _arch_control_point(x0, y0, x1, y1, frac)
2227 u = 1.0 - t
2228 px = u * u * x0 + 2 * u * t * cx + t * t * x1
2229 py = u * u * y0 + 2 * u * t * cy + t * t * y1
2230 tx = 2 * u * (cx - x0) + 2 * t * (x1 - cx)
2231 ty = 2 * u * (cy - y0) + 2 * t * (y1 - cy)
2232 return px, py, tx, ty
2235def _saccade_arrow_rows(
2236 fix_df: pd.DataFrame,
2237 x_col: str,
2238 y_col: str,
2239 arch_frac: float | None = None,
2240) -> tuple[list, list, list, list]:
2241 """:func:`_saccade_arrow_markers` plus each arrowhead's saccade index.
2243 Returns ``(mid_x, mid_y, angle_deg, segment_index)``. Arrowheads are dropped
2244 for micro/degenerate saccades, so the arrays are shorter than the saccade
2245 count and the position alone doesn't say which saccade an arrow belongs to —
2246 ``segment_index[j]`` is the index of the departing fixation, which is what
2247 lets the animated replay reveal each arrow with its own saccade (VIZ-23).
2248 """
2249 if len(fix_df) < 2:
2250 return [], [], [], []
2251 ordered = fix_df.sort_values("timestamp_ms")
2252 xv = pd.to_numeric(ordered[x_col], errors="coerce").to_numpy()
2253 yv = pd.to_numeric(ordered[y_col], errors="coerce").to_numpy()
2254 # Suppress arrowheads on micro-saccades: a sub-pixel refixation has a
2255 # well-defined midpoint but its direction is just noise, so a full-size
2256 # arrow would point a random way. Threshold scales with the data extent so
2257 # it's dataset-agnostic.
2258 finite = np.isfinite(xv) & np.isfinite(yv)
2259 if finite.any():
2260 x_ext = float(np.nanmax(xv[finite]) - np.nanmin(xv[finite]))
2261 y_ext = float(np.nanmax(yv[finite]) - np.nanmin(yv[finite]))
2262 min_len = np.hypot(x_ext, y_ext) * _ARROW_MIN_LEN_FRAC
2263 else:
2264 min_len = 0.0
2265 mid_x: list = []
2266 mid_y: list = []
2267 angles: list = []
2268 seg_index: list = []
2269 for i in range(len(ordered) - 1):
2270 x0, y0, x1, y1 = xv[i], yv[i], xv[i + 1], yv[i + 1]
2271 if not np.isfinite((x0, y0, x1, y1)).all():
2272 continue
2273 dx, dy = x1 - x0, y1 - y0
2274 seg_len = float(np.hypot(dx, dy))
2275 if seg_len == 0.0 or seg_len < min_len:
2276 continue
2277 if arch_frac is None:
2278 mx, my, hx, hy = (x0 + x1) / 2.0, (y0 + y1) / 2.0, dx, dy
2279 else:
2280 mx, my, hx, hy = _arch_point_and_tangent(x0, y0, x1, y1, arch_frac)
2281 mid_x.append(mx)
2282 mid_y.append(my)
2283 # marker.angle is clockwise from up; screen-up is decreasing data y
2284 # (the y-axis is drawn reversed), so negate the heading's dy.
2285 angles.append(float(np.degrees(np.arctan2(hx, -hy))))
2286 seg_index.append(i)
2287 return mid_x, mid_y, angles, seg_index
2290def _saccade_arrow_markers(
2291 fix_df: pd.DataFrame,
2292 x_col: str,
2293 y_col: str,
2294 arch_frac: float | None = None,
2295) -> tuple[list, list, list]:
2296 """Arrowhead position + rotation for each saccade, for a marker trace.
2298 Returns (mid_x, mid_y, angle_deg) with one entry per consecutive-fixation
2299 segment: a marker at the segment midpoint, rotated to point along the gaze
2300 direction. Angles follow Plotly's ``marker.angle`` convention (degrees
2301 clockwise from "up") and account for the reversed y-axis — data y grows
2302 downward on screen — so they read correctly on the plot.
2304 ``arch_frac`` (BUG-9) must be the same value the segment builders got: in Arc
2305 mode (VIZ-9) the marker moves to the *arch's* midpoint and takes the arc's
2306 tangent there, instead of floating below the curve at the straight chord's
2307 midpoint. ``None`` (the default everywhere) keeps the straight-chord
2308 placement.
2309 """
2310 mid_x, mid_y, angles, _ = _saccade_arrow_rows(fix_df, x_col, y_col, arch_frac)
2311 return mid_x, mid_y, angles
2314_RGB_FUNCTION = re.compile(
2315 r"rgba?\(\s*(\d{1,3})\s*,\s*(\d{1,3})\s*,\s*(\d{1,3})\s*(?:,[^)]*)?\)"
2316)
2319def color_with_alpha(color: str, alpha: float) -> str:
2320 """``color`` as ``rgba(r,g,b,alpha)`` — a fill drawn at its own opacity.
2322 Takes ``#rrggbb``, ``#rgb`` or ``rgb(…)`` / ``rgba(…)`` (whose own alpha is
2323 replaced). Anything else raises ``ValueError`` naming it, rather than
2324 quietly drawing some other colour.
2325 """
2326 text = str(color).strip()
2327 match = re.fullmatch(r"#([0-9A-Fa-f]{3}|[0-9A-Fa-f]{6})", text)
2328 if match:
2329 digits = match.group(1)
2330 if len(digits) == 3:
2331 digits = "".join(c * 2 for c in digits)
2332 r, g, b = (int(digits[i : i + 2], 16) for i in (0, 2, 4))
2333 return f"rgba({r},{g},{b},{alpha})"
2334 match = _RGB_FUNCTION.fullmatch(text)
2335 if match and all(int(v) <= 255 for v in match.groups()):
2336 r, g, b = match.groups()
2337 return f"rgba({r},{g},{b},{alpha})"
2338 raise ValueError(
2339 f"{color!r} is not a color a fill can take: use #rrggbb, #rgb or rgb(r, g, b)."
2340 )
2343def build_word_boxes(
2344 words: pd.DataFrame,
2345 color: str = WORD_BOX_COLOR,
2346 fill_color: str = WORD_BOX_FILL_COLOR,
2347 fill_opacity: float = WORD_BOX_FILL_OPACITY,
2348 line_opacity: float = WORD_BOX_LINE_OPACITY,
2349) -> list:
2350 """Rectangles for the word interest areas.
2352 Drawn from ``measures.word_box_bounds`` — the experiment's own rectangles
2353 (BUG-83) — so what's on screen is exactly what ``assign_fixations_to_words``
2354 assigns against. On a tiling corpus each outline therefore runs on across
2355 the space after its word, and the word *label* is centred in it (BUG-97).
2356 ``color`` is the outline, drawn at ``line_opacity`` (0 = no outline); the
2357 fill is ``fill_color`` at ``fill_opacity`` (0 = no fill). Each is its own
2358 alpha, so one never fades the other. A fully opaque outline keeps ``color``
2359 as given, so any colour Plotly accepts still works there.
2360 """
2361 from .measures import word_box_bounds
2363 fill = color_with_alpha(fill_color, fill_opacity)
2364 if line_opacity < 1:
2365 color = color_with_alpha(color, line_opacity)
2366 shapes = []
2367 for x0, y0, x1, y1 in zip(*word_box_bounds(words)):
2368 shapes.append(
2369 dict(
2370 type="rect",
2371 x0=x0,
2372 y0=y0,
2373 x1=x1,
2374 y1=y1,
2375 line=dict(color=color, width=1),
2376 fillcolor=fill,
2377 # VIZ-5: tag the layer so split_scanpath_layers can separate the
2378 # word boxes from the (visually similar) heatmap rects.
2379 name=_shape_layer_tag("word_boxes"),
2380 )
2381 )
2382 return shapes
2385# Bold-frame overlay for critical-span words; rendered on top of regular word
2386# boxes only when the trial was shown with a preview question (Hunting condition).
2387_CRITICAL_FRAME_COLOR = "#000000" # black — high-contrast frame, readable over heatmaps
2388_CRITICAL_FRAME_WIDTH = 2
2389_CRITICAL_TEXT_COLOR = (
2390 HIGHLIGHTED_TEXT_COLOR # dark pink — used when critical_span_style="Mark text"
2391)
2394def build_critical_span_overlay(
2395 words: pd.DataFrame,
2396 column: str = "is_in_aspan",
2397 color: str = _CRITICAL_FRAME_COLOR,
2398) -> list:
2399 """Return outline shapes for the highlighted span (``column``, default the
2400 OneStop answer span ``is_in_aspan``), outlined in ``color``.
2402 Each visual line that contains highlighted words gets its own outline
2403 rectangle, going from the *first* to the *last* highlighted word on that
2404 line (not the whole line). Returns [] when the column is missing or no
2405 words match.
2406 """
2407 if not column or column not in words.columns:
2408 return []
2409 mask = words[column].fillna(False).astype(bool)
2410 if not mask.any():
2411 return []
2412 span = words[mask].copy()
2414 # Cluster words into visual lines by y. `line_idx` upstream is often a
2415 # constant (no real per-word line numbers in OneStop IA exports), so we
2416 # group by y with a tolerance of ~half a word-height: rows whose y jumps
2417 # by more than that are on a new line.
2418 typical_h = float(span["height"].median() or 1.0)
2419 y_sorted = span["y"].sort_values()
2420 line_ids = (y_sorted.diff().fillna(0) > typical_h * 0.5).cumsum()
2421 span["_line_id"] = line_ids.reindex(span.index)
2423 from .measures import word_box_bounds
2425 span_x0, _, span_x1, _ = word_box_bounds(span)
2426 span["_box_x0"], span["_box_x1"] = span_x0, span_x1
2428 shapes = []
2429 for _, group in span.groupby("_line_id"):
2430 x0 = float(group["_box_x0"].min())
2431 x1 = float(group["_box_x1"].max())
2432 y0 = float(group["y"].min())
2433 y1 = float((group["y"] + group["height"]).max())
2434 shapes.append(
2435 dict(
2436 type="rect",
2437 x0=x0,
2438 y0=y0,
2439 x1=x1,
2440 y1=y1,
2441 line=dict(color=color, width=_CRITICAL_FRAME_WIDTH),
2442 fillcolor="rgba(0,0,0,0)",
2443 layer="above",
2444 # VIZ-5: the critical-span outline rides the word-boxes layer.
2445 name=_shape_layer_tag("word_boxes"),
2446 )
2447 )
2448 return shapes
2451# --- VIZ-5: separable-layer export ------------------------------------------
2452# The scanpath figure is a single flattened image, but publication workflows want
2453# to restyle each layer in Illustrator / Inkscape. `split_scanpath_layers` returns
2454# one figure per layer, each a copy of the full figure with only that layer's
2455# elements kept and everything else removed — so the layouts (axis ranges, size,
2456# equal-aspect scaleanchor) stay byte-identical and the exported files register
2457# perfectly when stacked. Every element is tagged with its layer: shapes carry a
2458# `_LAYER_SHAPE_TAG`-prefixed `name`, traces are classified by their (stable) name,
2459# and the single `layout.image` is the stimulus.
2460_LAYER_SHAPE_TAG = "__sps_layer:"
2461# Draw order (bottom → top), matching how make_scanpath_figure stacks them.
2462SCANPATH_LAYER_ORDER = (
2463 "stimulus_image",
2464 "heatmap",
2465 "word_boxes",
2466 "saccades",
2467 "fixations",
2468 "raw_gaze",
2469 "labels",
2470 "frame",
2471)
2472_TRANSPARENT = "rgba(0,0,0,0)"
2475def _shape_layer_tag(layer: str) -> str:
2476 """The `name` marker stamped on a shape so its layer survives into the figure."""
2477 return f"{_LAYER_SHAPE_TAG}{layer}"
2480def _shape_layer(shape) -> str | None:
2481 """Layer of a tagged shape, or None for an untagged one."""
2482 name = getattr(shape, "name", None) or ""
2483 if name.startswith(_LAYER_SHAPE_TAG):
2484 return name[len(_LAYER_SHAPE_TAG) :]
2485 return None
2488def _trace_layer(trace) -> str:
2489 """Classify a scanpath trace into its layer by its (stable) name.
2491 Only a handful of names are fixed — ``words`` (labels), the saccade traces,
2492 ``Raw gaze``, and any heatmap trace (``… heatmap …``). Every *other* trace the
2493 scanpath figure draws is a fixation-marker variant with a data-dependent name
2494 (``Fixations``, per-line ``line: …``, categorical colour-legend entries, the
2495 PRE-2 flag overlays, the PRE-3 ``drift`` connectors), so they all fall through
2496 to ``fixations`` — robust to those names changing."""
2497 name = trace.name or ""
2498 low = name.lower()
2499 if name == "words":
2500 return "labels"
2501 if "heatmap" in low:
2502 return "heatmap"
2503 if name == "Raw gaze":
2504 return "raw_gaze"
2505 if name in ("saccades", "saccade direction") or name in set(
2506 SACCADE_CLASS_LABELS.values()
2507 ):
2508 return "saccades"
2509 return "fixations"
2512def split_scanpath_layers(fig: go.Figure) -> dict[str, go.Figure]:
2513 """Split a `make_scanpath_figure` result into one figure per visible layer.
2515 Returns ``{layer_name: figure}`` in bottom-to-top draw order, keeping only the
2516 layers that actually have elements. Each returned figure is a copy of ``fig``
2517 with (a) only that layer's traces/shapes/images kept and (b) a transparent
2518 paper/plot background, so stacking the exported files in a vector editor
2519 reproduces the combined figure exactly (identical axis ranges + size ⇒ perfect
2520 registration). See VIZ-5."""
2521 # Which layers are present, and each trace's layer (computed once).
2522 trace_layers = [_trace_layer(tr) for tr in fig.data]
2523 shape_layers = [_shape_layer(sh) for sh in (fig.layout.shapes or ())]
2524 present = set(trace_layers) | {s for s in shape_layers if s}
2525 if fig.layout.images:
2526 present.add("stimulus_image")
2528 out: dict[str, go.Figure] = {}
2529 for layer in SCANPATH_LAYER_ORDER:
2530 if layer not in present:
2531 continue
2532 g = copy.deepcopy(fig)
2533 g.data = tuple(tr for tr, tl in zip(g.data, trace_layers) if tl == layer)
2534 g.layout.shapes = tuple(
2535 sh for sh, sl in zip(g.layout.shapes or (), shape_layers) if sl == layer
2536 )
2537 g.layout.images = fig.layout.images if layer == "stimulus_image" else ()
2538 # The duration size key's ms labels belong with its circles, which ride
2539 # the fixations layer; every other annotation (title text, the
2540 # Illustration stamp) stays on each layer as before.
2541 if layer != "fixations":
2542 g.layout.annotations = tuple(
2543 a for a in (g.layout.annotations or ()) if a.name != _SIZE_KEY_NAME
2544 )
2545 # Transparent background so the layers overlay cleanly when re-stacked.
2546 g.update_layout(paper_bgcolor=_TRANSPARENT, plot_bgcolor=_TRANSPARENT)
2547 out[layer] = g
2548 return out
2551_HOVER_MEASURE_LABELS: dict[str, str] = {
2552 "total_fixation_duration_ms": "TFD",
2553 "first_fixation_ms": "FFD",
2554 "first_pass_gaze_duration_ms": "FPRT",
2555 "regression_path_duration_ms": "RPD",
2556 "n_fixations": "Fixations",
2557}
2559#: DATA-66: `FigureSettings.column_labels` for the build in progress. The three
2560#: builders set it (`_labelled_columns`) and the helpers that write a column's
2561#: name into the figure read it (`_column_title`, `_hover_label`), so the dozen
2562#: helpers in between keep their signatures. Empty outside a build.
2563_COLUMN_LABELS: ContextVar[Mapping[str, str] | None] = ContextVar(
2564 "scanpath_column_labels", default=None
2565)
2568@contextmanager
2569def _labelled_columns(labels: Mapping[str, str] | None) -> Iterator[None]:
2570 """Make ``labels`` the column names one figure build writes."""
2571 token = _COLUMN_LABELS.set(dict(labels or {}))
2572 try:
2573 yield
2574 finally:
2575 _COLUMN_LABELS.reset(token)
2578def _humanize_column(column: str, *, unit: bool = True) -> str:
2579 """``total_fixation_duration_ms`` → "Total fixation duration (ms)", and
2580 ``participant_id`` → "Participant ID".
2582 ``unit=False`` drops the unit, for a hover row that writes it after the
2583 value — which used to read "… Duration Ms: 200 ms"."""
2584 from .column_names import _CANONICAL_LABELS, canonical_label
2586 text = str(column)
2587 if text in _CANONICAL_LABELS:
2588 # #374: the app's own columns read as the rail names them.
2589 label = canonical_label(text)
2590 return label if unit else label.removesuffix(" (ms)")
2591 in_ms = text.endswith("_ms")
2592 if in_ms:
2593 text = text[: -len("_ms")]
2594 words = text.replace("_", " ").strip()
2595 title = re.sub(r"\bid\b", "ID", words[:1].upper() + words[1:], flags=re.IGNORECASE)
2596 return f"{title} (ms)" if in_ms and unit else title
2599def _column_name(column: str) -> str:
2600 """A column's name in a legend entry: the dataset's own (DATA-66), else the
2601 column's own, as the legend has always written it."""
2602 labels = _COLUMN_LABELS.get() or {}
2603 return labels.get(column, str(column))
2606def _column_title(column: str) -> str:
2607 """A column's name in a figure's titles: the dataset's own (DATA-66), else
2608 the column humanized."""
2609 labels = _COLUMN_LABELS.get() or {}
2610 return labels[column] if column in labels else _humanize_column(column)
2613def _table_label(field: str, table: str | None) -> str | None:
2614 """``field``'s label in ``table`` when the build names it: the
2615 ``"<table>:<field>"`` entry first — a word table and a fixation table can
2616 call one canonical column differently (``word_id``) — then the plain one."""
2617 labels = _COLUMN_LABELS.get() or {}
2618 if table is not None and f"{table}:{field}" in labels:
2619 return labels[f"{table}:{field}"]
2620 return labels.get(field)
2623def _hover_label(field: str, table: str | None = None) -> str:
2624 """Readable label for an arbitrary hover column.
2626 The dataset's own name when the build has one (DATA-66), as ``table``
2627 names it; else a short label for the app's own columns, else the column
2628 humanized without its unit (the row writes the unit after the value)."""
2629 own = _table_label(field, table)
2630 if own is not None:
2631 return own
2632 aliases = {
2633 "text": "Word",
2634 "word_id": "Word #",
2635 "line_idx": "Line #",
2636 "order_in_trial": "Fixation #",
2637 "duration_ms": "Duration",
2638 "timestamp_ms": "Timestamp",
2639 }
2640 return aliases.get(
2641 field,
2642 _HOVER_MEASURE_LABELS.get(field, _humanize_column(field, unit=False)),
2643 )
2646def _plotly_literal(value: str) -> str:
2647 """``value`` as Plotly text that draws its own characters.
2649 Plotly reads a text or hover string as its pseudo-HTML, so a stimulus
2650 token ``<b>bold</b>`` drew bold and ``x<br>y`` broke the line (round-8
2651 review, finding 6). Escaping ``&``, ``<`` and ``>`` — the entities Plotly
2652 decodes back — keeps the dataset's characters on screen, and ``%{`` is
2653 written ``%{`` so a name placed in a hover template is not read as a
2654 template field (round 9). Applied once, to data values and user text only,
2655 at the figure boundary: the tables, exports and the app's own markup (a
2656 hover's ``<br>``) are left as they are."""
2657 return html.escape(value, quote=False).replace("%{", "%{")
2660def _plotly_literal_values(series: pd.Series) -> pd.Series:
2661 """A hover column with its strings made literal (:func:`_plotly_literal`);
2662 numbers, missing values and anything else pass through untouched."""
2663 if pd.api.types.is_numeric_dtype(series) or pd.api.types.is_bool_dtype(series):
2664 return series
2665 return series.map(
2666 lambda value: _plotly_literal(value) if isinstance(value, str) else value,
2667 na_action="ignore",
2668 )
2671def _fixation_order_labels(ordered: pd.DataFrame) -> list[str]:
2672 """The fixation-number labels of a replay trail, one per row of ``ordered``.
2674 The trial's own ``order_in_trial``, as the static figure, the comparison
2675 and every hover show it — so a later screen of a multipart trial, a
2676 fixation window or a *Discard* keeps its gaps (501, 502 …) instead of
2677 renumbering what is left 1..n. A row without an index gets no label, as
2678 on the static figure. Only a frame with no usable index at all (no column,
2679 or nothing numeric in it) falls back to the ordinal 1..n."""
2680 n = len(ordered)
2681 if "order_in_trial" in ordered.columns:
2682 values = pd.to_numeric(ordered["order_in_trial"], errors="coerce")
2683 if values.notna().any():
2684 return [
2685 ""
2686 if pd.isna(value)
2687 else str(int(value))
2688 if float(value).is_integer()
2689 else f"{value:g}"
2690 for value in values.tolist()
2691 ]
2692 return [str(j + 1) for j in range(n)]
2695#: #374 F7: the fixation hover's lead line is written from these, in this
2696#: order, as "Fixation 41 · 336 ms · on “Droppings!” (word 27)".
2697_FIXATION_HEAD_FIELDS = ("order_in_trial", "duration_ms", "word_id")
2700def _hover_number(value) -> str:
2701 """A hover number without a trailing ``.0`` (``27.0`` → ``27``)."""
2702 number = pd.to_numeric(value, errors="coerce")
2703 if pd.isna(number):
2704 return str(value)
2705 return f"{number:.0f}" if float(number).is_integer() else f"{number:g}"
2708def _fixation_hover_head(
2709 frame: pd.DataFrame, head: Sequence[str], words: pd.DataFrame | None
2710) -> pd.Series:
2711 """The fixation hover's lead line, one string per row (#374 F7).
2713 The word a fixation landed on is named by its text when the trial's word
2714 table has it (and its word ids are unique); a fixation on no word reads
2715 "outside the text"."""
2716 word_text: dict[float, str] = {}
2717 if (
2718 "word_id" in head
2719 and words is not None
2720 and not words.empty
2721 and {"word_id", "text"} <= set(words.columns)
2722 ):
2723 ids = pd.to_numeric(words["word_id"], errors="coerce")
2724 if ids.notna().all() and ids.is_unique:
2725 word_text = dict(zip(ids.astype(float), words["text"].astype(str)))
2726 columns = {field: frame[field].tolist() for field in head}
2727 # A fixation with no word id is "outside the text" only when it is outside
2728 # every word box: the data's own assignment can be blank inside one.
2729 outside = [False] * len(frame)
2730 if (
2731 "word_id" in head
2732 and words is not None
2733 and not words.empty
2734 and {"x", "y"} <= set(frame.columns)
2735 ):
2736 from .measures import fixation_in_text_mask
2738 outside = (~fixation_in_text_mask(frame, words)).tolist()
2739 lines = []
2740 for i in range(len(frame)):
2741 parts = []
2742 if "order_in_trial" in columns:
2743 value = columns["order_in_trial"][i]
2744 if pd.notna(value):
2745 parts.append(f"Fixation {_hover_number(value)}")
2746 if "duration_ms" in columns:
2747 value = columns["duration_ms"][i]
2748 if pd.notna(value):
2749 parts.append(f"{_hover_number(value)} ms")
2750 if "word_id" in columns:
2751 value = pd.to_numeric(columns["word_id"][i], errors="coerce")
2752 if pd.isna(value):
2753 if outside[i]:
2754 parts.append("outside the text")
2755 else:
2756 text = word_text.get(float(value))
2757 word = f"word {_hover_number(value)}"
2758 parts.append(
2759 f"on “{_plotly_literal(text)}” ({word})" if text else f"on {word}"
2760 )
2761 lines.append(" · ".join(parts))
2762 return pd.Series(lines, index=frame.index, dtype=object)
2765def _hover_cell(value):
2766 """One hover value as shown: a missing one is "—" (Plotly printed
2767 ``null``), a fraction is cut to four significant digits (#374)."""
2768 if value is None:
2769 return "—"
2770 try:
2771 if pd.isna(value):
2772 return "—"
2773 except (TypeError, ValueError):
2774 return value
2775 if isinstance(value, (float, np.floating)):
2776 return f"{value:.0f}" if float(value).is_integer() else f"{value:.4g}"
2777 return value
2780def _hover_cells(series: pd.Series) -> pd.Series:
2781 """:func:`_hover_cell` over a hover column."""
2782 return series.astype(object).map(_hover_cell)
2785def _hover_payload(
2786 frame: pd.DataFrame,
2787 fields: Sequence[str],
2788 *,
2789 line_display: pd.Series | None = None,
2790 table: str | None = None,
2791 fixation: bool = False,
2792 words: pd.DataFrame | None = None,
2793) -> tuple[np.ndarray | None, str]:
2794 """Plotly customdata + template for a user-selected field list (VIZ-26);
2795 ``table`` says whose names label the rows (DATA-66).
2797 ``fixation=True`` writes the fixation number, duration and word as one
2798 plain lead line (#374 F7), the word by its text from ``words``; any other
2799 chosen field follows as a ``Label: value`` row."""
2800 valid = [
2801 field
2802 for field in fields
2803 if field in frame.columns or (field == "line_idx" and line_display is not None)
2804 ]
2805 if not valid:
2806 return None, "<extra></extra>"
2807 values: list[pd.Series] = []
2808 rows: list[str] = []
2809 if fixation:
2810 head = [field for field in _FIXATION_HEAD_FIELDS if field in valid]
2811 if head:
2812 values.append(_fixation_hover_head(frame, head, words))
2813 rows.append("%{customdata[0]}")
2814 valid = [field for field in valid if field not in head]
2815 for idx, field in enumerate(valid, start=len(values)):
2816 series = (
2817 line_display
2818 if field == "line_idx" and line_display is not None
2819 else frame[field]
2820 )
2821 values.append(_hover_cells(_plotly_literal_values(series)))
2822 suffix = " ms" if field.endswith("_ms") else ""
2823 rows.append(f"{_hover_label(field, table)}: %{{customdata[{idx}]}}{suffix}")
2824 customdata = pd.concat(values, axis=1).to_numpy(dtype=object)
2825 return customdata, "<br>".join(rows) + "<extra></extra>"
2828#: The highlight key's annotation (#374 F6), so it is drawn once a figure.
2829_HIGHLIGHT_KEY_NAME = "highlight_key"
2832def _add_highlight_key(
2833 fig: go.Figure, column: str, color: str, *, border: bool = False
2834) -> None:
2835 """Name the highlighted words on the figure itself (#374 F6): a swatch in
2836 the highlight's colour and "Highlighted: is_in_aspan (answer span)", in
2837 the plot's bottom-left corner. Drawn once however many panels mark words."""
2838 if any(a.name == _HIGHLIGHT_KEY_NAME for a in fig.layout.annotations or ()):
2839 return
2840 from .column_names import HIGHLIGHT_NOTES
2842 note = HIGHLIGHT_NOTES.get(str(column))
2843 swatch = "▢" if border else "■"
2844 text = (
2845 f'<span style="color:{color}">{swatch}</span> Highlighted: '
2846 f"{_plotly_literal(_column_name(column))}" + (f" ({note})" if note else "")
2847 )
2848 fig.add_annotation(
2849 x=0,
2850 y=0,
2851 xref="paper",
2852 yref="paper",
2853 xanchor="left",
2854 yanchor="bottom",
2855 xshift=6,
2856 yshift=6,
2857 text=text,
2858 showarrow=False,
2859 align="left",
2860 font=dict(size=12, color="#444444"),
2861 bgcolor="rgba(255,255,255,0.75)",
2862 name=_HIGHLIGHT_KEY_NAME,
2863 )
2866def _add_word_label_trace(
2867 fig: go.Figure,
2868 words: pd.DataFrame,
2869 base_font_size: int,
2870 font_family: str,
2871 row: int | None = None,
2872 col: int | None = None,
2873 highlight_column: str | None = None,
2874 text_color: str = WORD_LABEL_COLOR,
2875 highlight_text_color: str = _CRITICAL_TEXT_COLOR,
2876 word_hover_measure: str | None = None,
2877 word_hover_fields: Sequence[str] | None = None,
2878) -> None:
2879 if words.empty or "text" not in words.columns:
2880 return
2881 customdata = None
2882 hover = "Word: %{text}<extra></extra>"
2883 if "word_id" in words.columns:
2884 from .measures import cluster_word_lines
2886 # The source ``line_idx`` is often a constant (OneStop IA exports rarely
2887 # carry a real per-word line number), so infer the visual line from
2888 # word-box geometry — same clustering the by-line coloring uses — and
2889 # show it 1-based.
2890 line_display = (cluster_word_lines(words) + 1).rename("line")
2891 if word_hover_fields is not None:
2892 customdata, hover = _hover_payload(
2893 words, word_hover_fields, line_display=line_display, table="words"
2894 )
2895 else:
2896 # Legacy API/deep-link behaviour: the three fixed identity lines plus
2897 # the old single optional measure.
2898 customdata_parts: list[pd.Series] = [
2899 _plotly_literal_values(words["word_id"]),
2900 line_display,
2901 ]
2902 hover = "Word: %{text}<br>Word #%{customdata[0]}<br>Line #%{customdata[1]}"
2903 if word_hover_measure and word_hover_measure in words.columns:
2904 label = _table_label(
2905 word_hover_measure, "words"
2906 ) or _HOVER_MEASURE_LABELS.get(word_hover_measure, word_hover_measure)
2907 suffix = " ms" if word_hover_measure.endswith("_ms") else ""
2908 hover += f"<br>{label}: %{{customdata[2]}}{suffix}"
2909 customdata_parts.append(
2910 _plotly_literal_values(words[word_hover_measure])
2911 )
2912 hover += "<extra></extra>"
2913 customdata = pd.concat(customdata_parts, axis=1)
2914 # Per-word text color: the highlight colour for highlighted words when the
2915 # caller asks for "Mark text" (``highlight_column`` set), the base text
2916 # colour otherwise. Both are configurable from the plot rail.
2917 if highlight_column and highlight_column in words.columns:
2918 critical_mask = words[highlight_column].fillna(False).astype(bool)
2919 label_color = [
2920 highlight_text_color if is_crit else text_color for is_crit in critical_mask
2921 ]
2922 if critical_mask.any():
2923 _add_highlight_key(fig, highlight_column, highlight_text_color)
2924 else:
2925 label_color = text_color
2926 # BUG-97 — the label is centred in its word's box, as the data defines it
2927 # (`_word_label_x`: a padded layout's line-start word sits flush left, as it
2928 # was shown). BUG-30 centred it on the glyph run instead, which on a tiling
2929 # corpus (the box carries the following space) drew every word flush left.
2930 # Centred text needs no LTR/RTL anchor; the Unicode direction isolates stay —
2931 # they are about *shaping* mixed Hebrew/Arabic + punctuation, not placement.
2932 from .preprocessing import detect_right_to_left
2934 rtl = words.get("right_to_left")
2935 if rtl is None:
2936 rtl = words["text"].astype(str).map(detect_right_to_left)
2937 else:
2938 rtl = rtl.fillna(False).astype(bool)
2939 label_x = _word_label_x(words)
2940 # The word drawn as its own characters (finding 6 — not as Plotly markup),
2941 # escaped before the direction isolates wrap it; the hover's `%{text}`
2942 # reads this same string, so it shows the word literally too.
2943 label_text = [
2944 f"\u2067{_plotly_literal(value)}\u2069" if is_rtl else _plotly_literal(value)
2945 for value, is_rtl in zip(words["text"].astype(str), rtl)
2946 ]
2947 trace = go.Scatter(
2948 x=label_x,
2949 y=words["y"] + words["height"] / 2,
2950 text=label_text,
2951 mode="text",
2952 textposition="middle center",
2953 showlegend=False,
2954 textfont=dict(color=label_color, size=base_font_size, family=font_family),
2955 hovertemplate=hover,
2956 customdata=customdata,
2957 name="words",
2958 )
2959 if row is not None and col is not None:
2960 fig.add_trace(trace, row=row, col=col)
2961 else:
2962 fig.add_trace(trace)
2965_IMAGE_MIME = {
2966 ".png": "image/png",
2967 ".jpg": "image/jpeg",
2968 ".jpeg": "image/jpeg",
2969 ".gif": "image/gif",
2970 ".webp": "image/webp",
2971}
2974def _image_to_data_uri(src: str | None) -> str | None:
2975 """A ``data:`` URI for an image path, or pass through an existing one.
2977 Returns None for a missing / unreadable file so the background-image layer
2978 simply doesn't draw (e.g. an uploaded MultiplEYE dataset has no image path)."""
2979 if not src:
2980 return None
2981 text = str(src)
2982 if text.startswith("data:"):
2983 return text
2984 path = Path(text)
2985 try:
2986 raw = path.read_bytes()
2987 except OSError:
2988 return None
2989 mime = _IMAGE_MIME.get(path.suffix.lower(), "image/png")
2990 return f"data:{mime};base64," + base64.b64encode(raw).decode("ascii")
2993def _png_pixel_size(src: str | None) -> tuple[int, int] | None:
2994 """(width, height) of a PNG from its header, without Pillow; None otherwise."""
2995 if not src:
2996 return None
2997 try:
2998 with open(src, "rb") as fh:
2999 head = fh.read(24)
3000 except OSError:
3001 return None
3002 if head[:8] == b"\x89PNG\r\n\x1a\n" and head[12:16] == b"IHDR":
3003 width, height = struct.unpack(">II", head[16:24])
3004 return int(width), int(height)
3005 return None
3008def _background_image_spec(
3009 background_image: str | None,
3010 background_image_size: tuple[float, float] | None,
3011 background_image_origin: tuple[float, float] | None,
3012 background_image_opacity: float = 1.0,
3013) -> dict | None:
3014 """The ``layout.image`` dict for the stimulus-page background (VIZ-4).
3016 The rendered page sits at data coordinates ``(origin_x, origin_y)`` →
3017 ``(+image_w, +image_h)`` UNDER every other layer; the coords are where the
3018 (centered) stimulus appeared on the monitor, so the image aligns exactly and
3019 sidesteps CJK/RTL font rendering. ``background_image_origin`` defaults to
3020 ``(0, 0)``; ``yanchor="top"`` + the reversed y-axis put the image's top-left
3021 there. ``background_image_opacity`` dims a busy stimulus so the AOIs/scanpath
3022 read over it (1.0 = opaque). Returns ``None`` when there is nothing to draw
3023 (no image, no size, or an unreadable path), so callers can just skip it.
3024 """
3025 if not (background_image and background_image_size):
3026 return None
3027 uri = _image_to_data_uri(background_image)
3028 if not uri:
3029 return None
3030 image_w, image_h = background_image_size
3031 origin_x, origin_y = background_image_origin or (0.0, 0.0)
3032 return dict(
3033 source=uri,
3034 xref="x",
3035 yref="y",
3036 x=origin_x,
3037 y=origin_y,
3038 sizex=image_w,
3039 sizey=image_h,
3040 sizing="stretch",
3041 layer="below",
3042 xanchor="left",
3043 yanchor="top",
3044 opacity=float(background_image_opacity),
3045 )
3048def _add_background_image(
3049 fig: go.Figure,
3050 background_image: str | None,
3051 background_image_size: tuple[float, float] | None,
3052 background_image_origin: tuple[float, float] | None,
3053 background_image_opacity: float = 1.0,
3054 *,
3055 row: int | None = None,
3056 col: int | None = None,
3057) -> bool:
3058 """Add the stimulus-page background image to ``fig``; True when one was added.
3060 ``row``/``col`` bind it to one subplot's axes (the split comparison figure);
3061 without them it goes on the figure's single axis pair.
3062 """
3063 spec = _background_image_spec(
3064 background_image,
3065 background_image_size,
3066 background_image_origin,
3067 background_image_opacity,
3068 )
3069 if spec is None:
3070 return False
3071 if row is not None and col is not None:
3072 fig.add_layout_image(spec, row=row, col=col)
3073 else:
3074 fig.add_layout_image(spec)
3075 return True
3078def _add_saccade_layer(
3079 fig: go.Figure,
3080 fixations: pd.DataFrame,
3081 *,
3082 x_field: str,
3083 y_field: str,
3084 color: str,
3085 width: float,
3086 style: str,
3087 show_arrows: bool,
3088 saccade_classes: pd.Series | None = None,
3089 color_by_class: bool = True,
3090 class_colors: dict | None = None,
3091 class_legend: bool = True,
3092 visible_classes: Iterable[str] | None = None,
3093 render_mode: str = "Straight",
3094 two_way: bool = False,
3095) -> bool:
3096 """Add one scanpath's saccade lines (+ optional direction arrowheads) to ``fig``.
3098 Connects consecutive fixations in time order. When ``saccade_classes`` is
3099 given (the per-fixation reading class from ``measures.classify_saccades``)
3100 the segments are grouped by class; with ``color_by_class`` (VIZ-8 "By type")
3101 each group becomes its own sub-trace in its colour from ``class_colors``
3102 plus a small legend, and ``two_way`` (VIZ-19) first folds the five classes
3103 down to forward vs. regression. Otherwise a single uniform-``color`` trace is
3104 drawn. ``visible_classes`` (VIZ-31) is the reading-class **filter**: segments
3105 whose class isn't listed are not drawn at all — so it needs the
3106 classification even in uniform-colour mode, which is why ``color_by_class``
3107 exists separately. Filtering happens on the *unfolded* classes, before the
3108 two-way fold, so "show only regressions" means the same thing in every
3109 colour mode. It is read literally: an **empty** ``visible_classes`` draws no
3110 saccades. (The rail never sends one — ``controls._collect_viz_settings``
3111 reads a cleared multiselect as "no filter", since hiding the layer outright
3112 is what its toggle is for.) ``render_mode="Arc"`` (VIZ-9) draws each saccade as an upward
3113 arch instead of a straight connector. The arrowheads are a separate,
3114 independently-toggled trace drawn before the fixation markers so the dots sit
3115 on top (uniform ``color`` either way — they encode direction, the line colour
3116 encodes type), and they honour the same filter. The caller gates this on
3117 spatial axes, ``show_saccades`` and at least two fixations.
3119 Returns ``True`` when legend entries were added (by-type mode), so the caller
3120 reserves margin for the legend (mirrors ``_add_raw_gaze_layer``).
3121 """
3122 legend_added = False
3123 arch_frac = _ARCH_FRAC if render_mode == "Arc" else None
3124 keep = None if visible_classes is None else set(visible_classes)
3125 # Group once, on the raw (unfolded) classes, then filter — the two-way fold
3126 # and the uniform-colour merge below both work off the same grouping.
3127 segs = None
3128 if saccade_classes is not None:
3129 segs = _saccade_segments_by_class(
3130 fixations, x_field, y_field, saccade_classes, arch_frac
3131 )
3132 if keep is not None:
3133 segs = {c: s for c, s in segs.items() if c in keep}
3135 if segs is not None and color_by_class:
3136 # Merge over the full defaults so classes the caller omits (notably the
3137 # non-editable "other" catch-all — the UI palette only carries the five
3138 # reading classes) still get their intended colour instead of falling
3139 # back to the uniform line colour.
3140 palette = {**SACCADE_CLASS_COLORS, **(class_colors or {})}
3141 # VIZ-19: the two-way mode is the five-way one with the classes folded
3142 # into forward/regression buckets, so everything below — segments,
3143 # colours, legend — is shared. "Forward" takes the forward class's
3144 # colour, which is what the picker shows for it.
3145 if two_way:
3146 folded: dict = {}
3147 for cls_name, (fx, fy) in segs.items():
3148 bucket = SACCADE_DIRECTION_FOLD.get(cls_name, "other")
3149 bx, by = folded.setdefault(bucket, ([], []))
3150 bx.extend(fx)
3151 by.extend(fy)
3152 segs = folded
3153 draw_order = [*SACCADE_DIRECTION_CLASSES, "other"]
3154 labels = SACCADE_DIRECTION_LABELS
3155 legend_title = "Saccade direction"
3156 else:
3157 draw_order = SACCADE_CLASS_ORDER
3158 labels = SACCADE_CLASS_LABELS
3159 legend_title = "Saccade type"
3160 for cls_name in draw_order:
3161 seg = segs.get(cls_name)
3162 if not seg or not seg[0]:
3163 continue
3164 sx, sy = seg
3165 fig.add_trace(
3166 go.Scatter(
3167 x=sx,
3168 y=sy,
3169 mode="lines",
3170 line=dict(
3171 color=palette.get(cls_name, color), width=width, dash=style
3172 ),
3173 hoverinfo="skip",
3174 # VIZ-8: the colour key is optional — hide it (but keep the
3175 # coloured sub-traces) when class_legend is off.
3176 showlegend=class_legend,
3177 legendgroup="saccade_type",
3178 legendgrouptitle_text=legend_title,
3179 name=labels.get(cls_name, cls_name),
3180 )
3181 )
3182 # Reserve legend margin only when the key is actually shown.
3183 legend_added = class_legend
3184 else:
3185 if segs is None:
3186 sx, sy = _saccade_segments(fixations, x_field, y_field, arch_frac)
3187 else:
3188 # Filtered, but drawn in one uniform colour: concatenate the classes
3189 # that survived. Each class's arrays are already None-separated, so
3190 # joining them is the same trace the unfiltered path builds minus the
3191 # hidden segments (their order within the trace doesn't render).
3192 sx, sy = [], []
3193 for cls_name in [
3194 *SACCADE_CLASS_ORDER,
3195 *(c for c in segs if c not in SACCADE_CLASS_ORDER),
3196 ]:
3197 seg = segs.get(cls_name)
3198 if seg:
3199 sx.extend(seg[0])
3200 sy.extend(seg[1])
3201 if sx:
3202 fig.add_trace(
3203 go.Scatter(
3204 x=sx,
3205 y=sy,
3206 mode="lines",
3207 line=dict(color=color, width=width, dash=style),
3208 hoverinfo="skip",
3209 showlegend=False,
3210 name="saccades",
3211 )
3212 )
3213 if show_arrows:
3214 # BUG-9: same arch_frac as the segments above, so the arrowheads sit on
3215 # the drawn line (arched or straight) instead of on the chord.
3216 amx, amy, aang, aseg = _saccade_arrow_rows(
3217 fixations, x_field, y_field, arch_frac
3218 )
3219 if amx and keep is not None and saccade_classes is not None:
3220 mask = _arrow_class_mask(fixations, saccade_classes, keep, aseg)
3221 amx = [v for v, m in zip(amx, mask) if m]
3222 amy = [v for v, m in zip(amy, mask) if m]
3223 aang = [v for v, m in zip(aang, mask) if m]
3224 if amx:
3225 fig.add_trace(
3226 go.Scatter(
3227 x=amx,
3228 y=amy,
3229 mode="markers",
3230 marker=dict(
3231 symbol="arrow",
3232 size=12,
3233 angle=aang,
3234 angleref="up",
3235 color=color,
3236 line=dict(width=0),
3237 ),
3238 hoverinfo="skip",
3239 showlegend=False,
3240 name="saccade direction",
3241 )
3242 )
3243 return legend_added
3246def _add_raw_gaze_layer(
3247 fig: go.Figure,
3248 raw_gaze: pd.DataFrame | None,
3249 *,
3250 show_raw_gaze: bool,
3251 raw_gaze_color: str = "#888888",
3252 raw_gaze_marker_size: float = 4.0,
3253 raw_gaze_opacity: float = 0.6,
3254) -> bool:
3255 """Add the raw-gaze sample-point scatter (time-coloured when available).
3257 Returns ``True`` if a trace was added, so the caller can mark the legend
3258 active (it reserves margin for the legend). Self-gates on ``show_raw_gaze``
3259 and a non-empty frame. **UX-86**: ``raw_gaze_color`` is the flat colour
3260 used when the data carries no ``timestamp_ms`` — when it does, fixations
3261 stay time-mapped (Viridis) since that is the more informative default and
3262 a flat colour would throw it away. Samples imported without a clock carry
3263 ``sample_index`` instead: coloured by that order, with the legend titled
3264 *Sample order* and the hover saying ``sample n`` — never milliseconds.
3265 """
3266 if not (show_raw_gaze and raw_gaze is not None and not raw_gaze.empty):
3267 return False
3268 legend_title = None
3269 if "timestamp_ms" in raw_gaze.columns:
3270 color_vals = raw_gaze["timestamp_ms"]
3271 colorscale = "Viridis"
3272 customdata = raw_gaze["timestamp_ms"]
3273 when = "<br>Timestamp: %{customdata} ms"
3274 elif SAMPLE_INDEX in raw_gaze.columns:
3275 # No clock (the import mapped none): coloured by the samples' order,
3276 # and said so — a ramp with no title would read as time.
3277 color_vals = raw_gaze[SAMPLE_INDEX]
3278 colorscale = "Viridis"
3279 customdata = raw_gaze[SAMPLE_INDEX]
3280 when = "<br>Sample #: %{customdata}"
3281 legend_title = "Sample order"
3282 else:
3283 color_vals = raw_gaze_color
3284 colorscale = None
3285 customdata = None
3286 when = ""
3287 fig.add_trace(
3288 go.Scatter(
3289 x=raw_gaze["x"],
3290 y=raw_gaze["y"],
3291 mode="markers",
3292 marker=dict(
3293 size=raw_gaze_marker_size,
3294 color=color_vals,
3295 colorscale=colorscale,
3296 opacity=raw_gaze_opacity,
3297 showscale=False,
3298 ),
3299 hovertemplate=(
3300 "Raw gaze<br>x: %{x:.1f}<br>y: %{y:.1f}" + when + "<extra></extra>"
3301 ),
3302 customdata=customdata,
3303 legendgroup="raw_gaze" if legend_title else None,
3304 legendgrouptitle_text=legend_title,
3305 name="Raw gaze",
3306 showlegend=True,
3307 )
3308 )
3309 return True
3312# PRE-2 fixation classification (viz-only): SHORT / LONG / OUT-OF-BOUNDS, each
3313# Off / Highlight / Discard. Thresholds come from the caller's flags dict; these
3314# are the fallbacks the app's own controls default to.
3315_FIX_FLAG_SHORT_MS = 80.0
3316_FIX_FLAG_LONG_MS = 800.0
3317_FIX_FLAG_CATEGORIES = ("short", "long", "oob", "blink")
3318_FIX_FLAG_LABELS = {
3319 "short": "Short",
3320 "long": "Long",
3321 "oob": "Out of bounds",
3322 "blink": "Blink-adjacent",
3323}
3326def _fixation_flag_masks(
3327 fixations: pd.DataFrame,
3328 words: pd.DataFrame,
3329 flags: dict | None,
3330 *,
3331 spatial_axes: bool = True,
3332) -> dict[str, pd.Series]:
3333 """Boolean mask per PRE-2 fixation-flag category, over ``fixations``' index.
3335 ``{"short": …, "long": …, "oob": …}``, or ``{}`` when nothing is flagged.
3336 Out-of-bounds needs word boxes on spatial axes — without them no fixation
3337 counts as out of bounds. Shared by the static figure and the animated replay
3338 so the two classify identically (VIZ-23).
3339 """
3340 if not flags or fixations.empty:
3341 return {}
3342 dur = pd.to_numeric(fixations.get("duration_ms"), errors="coerce")
3343 oob = pd.Series(False, index=fixations.index)
3344 if spatial_axes and not words.empty:
3345 from .measures import fixation_in_text_mask
3347 oob = ~fixation_in_text_mask(fixations, words)
3348 short_ms = float(flags.get("short", {}).get("threshold_ms", _FIX_FLAG_SHORT_MS))
3349 long_ms = float(flags.get("long", {}).get("threshold_ms", _FIX_FLAG_LONG_MS))
3350 blink = pd.Series(False, index=fixations.index)
3351 for column in ("is_blink", "blink_before", "blink_after", "blink"):
3352 if column in fixations:
3353 blink |= fixations[column].fillna(False).astype(bool)
3354 return {
3355 "short": (dur < short_ms).fillna(False).astype(bool),
3356 "long": (dur > long_ms).fillna(False).astype(bool),
3357 "oob": oob.fillna(False).astype(bool),
3358 "blink": blink,
3359 }
3362def _discard_flagged_fixations(
3363 fixations: pd.DataFrame,
3364 words: pd.DataFrame,
3365 flags: dict | None,
3366 *,
3367 spatial_axes: bool = True,
3368) -> pd.DataFrame:
3369 """Drop the fixations whose PRE-2 flag mode is *Discard* (viz-only).
3371 Changes only what is DRAWN — the returned frame feeds the markers, the
3372 fixation-index labels and the marker-size scaling; reading measures and
3373 exports are untouched. Returns ``fixations`` itself when nothing is dropped.
3374 """
3375 masks = _fixation_flag_masks(fixations, words, flags, spatial_axes=spatial_axes)
3376 if not masks:
3377 return fixations
3378 drop = pd.Series(False, index=fixations.index)
3379 for category, mask in masks.items():
3380 if (flags or {}).get(category, {}).get("mode") == "Discard":
3381 drop = drop | mask
3382 return fixations[~drop] if bool(drop.any()) else fixations
3385def _render_scanpath_figure(
3386 words: pd.DataFrame,
3387 fixations: pd.DataFrame,
3388 *,
3389 settings: FigureSettings,
3390 raw_gaze: pd.DataFrame | None = None,
3391) -> go.Figure:
3392 canvas_width = settings.canvas_width
3393 canvas_height = settings.canvas_height
3394 base_font_size = settings.base_font_size
3395 font_family = settings.font_family
3396 x_field = settings.x_field
3397 y_field = settings.y_field
3398 show_words = settings.show_words
3399 show_word_labels = settings.show_word_labels
3400 show_fixations = settings.show_fixations
3401 show_order = settings.show_order
3402 show_saccades = settings.show_saccades
3403 show_heatmap = settings.show_heatmap
3404 color_by = settings.color_by
3405 heatmap_metric = settings.heatmap_metric
3406 show_saccade_arrows = settings.show_saccade_arrows
3407 heatmap_style = settings.heatmap_style
3408 heatmap_norm = settings.heatmap_norm
3409 heatmap_sigma_px = settings.heatmap_sigma_px
3410 marker_size_range = settings.marker_size_range
3411 order_font_size = settings.order_font_size
3412 order_font_color = settings.order_font_color
3413 show_colorbars = settings.show_fixation_colorbar
3414 show_heatmap_colorbar = settings.show_heatmap_colorbar
3415 fixation_color_range = settings.fixation_color_range
3416 heatmap_range = settings.heatmap_range
3417 fixation_colorscale = settings.fixation_colorscale
3418 heatmap_colorscale = settings.heatmap_colorscale
3419 show_raw_gaze = settings.show_raw_gaze
3420 raw_gaze_color = settings.raw_gaze_color
3421 raw_gaze_marker_size = settings.raw_gaze_marker_size
3422 raw_gaze_opacity = settings.raw_gaze_opacity
3423 critical_span_style = settings.critical_span_style
3424 highlight_column = settings.highlight_column
3425 saccade_color = settings.saccade_color
3426 saccade_style = settings.saccade_style
3427 saccade_width = settings.saccade_width
3428 saccade_color_mode = settings.saccade_color_mode
3429 saccade_class_colors = settings.saccade_class_colors
3430 saccade_type_legend = settings.saccade_type_legend
3431 saccade_classes = settings.saccade_classes
3432 saccade_render_mode = settings.saccade_render_mode
3433 fixation_snap_to_word = settings.fixation_snap_to_word
3434 hollow_fixations = settings.hollow_fixations
3435 fixation_opacity = settings.fixation_opacity
3436 fixation_color = settings.fixation_color
3437 fixation_symbol = settings.fixation_symbol
3438 text_color = settings.text_color
3439 highlight_text_color = settings.highlight_text_color
3440 background_color = settings.background_color
3441 # BUG-85: `color_by="line"` is the rail's own spelling of this (its "line"
3442 # option, a share link's `color_by=line`), so it colours by line from the
3443 # API and CLI too rather than falling through to a missing column.
3444 color_by_line = settings.color_by_line or settings.color_by == "line"
3445 fixation_flags = settings.fixation_flags
3446 span_border_color = settings.span_border_color
3447 # The fixations' colour bar; the heatmap's is `cb_style` below.
3448 colorbar_orientation = settings.fixation_colorbar_orientation
3449 colorbar_tickangle = settings.fixation_colorbar_tickangle
3450 colorbar_tickfont_size = settings.fixation_colorbar_tickfont_size
3451 line_spacing = settings.line_spacing
3452 scale_text_to_boxes = settings.scale_text_to_boxes
3453 background_image = settings.background_image
3454 background_image_size = settings.background_image_size
3455 background_image_origin = settings.background_image_origin
3456 background_image_opacity = settings.background_image_opacity
3457 fit_to_monitor = settings.fit_to_monitor
3458 show_coordinate_grid = settings.show_coordinate_grid
3459 coordinate_grid_spacing = settings.coordinate_grid_spacing
3460 word_heatmap_col = settings.word_heatmap_col
3461 word_heatmap_title = settings.word_heatmap_title
3462 word_hover_measure = settings.word_hover_measure
3463 word_hover_fields = settings.word_hover_fields
3464 fixation_hover_fields = settings.fixation_hover_fields
3465 show_connectors = settings.show_connectors
3466 connector_y = settings.connector_y
3467 illustration_reasons = settings.illustration_reasons
3468 fig = go.Figure()
3469 spatial_axes = x_field == "x" and y_field == "y"
3470 # Track whether a colorbar / legend will render, to reserve margin for them
3471 # below (so they don't shrink the equal-aspect plot). See _decoration_margins.
3472 legend_active = False
3473 heatmap_rendered = False
3474 is_numeric_color = False
3475 # The heatmap's colour-bar styling (orientation / tick angle / tick size).
3476 cb_style = dict(
3477 orientation=settings.heatmap_colorbar_orientation,
3478 tickangle=settings.heatmap_colorbar_tickangle,
3479 tickfont_size=settings.heatmap_colorbar_tickfont_size,
3480 )
3481 font_settings = dict(family=font_family or FONT_FAMILY, size=base_font_size)
3483 raw_for_range = raw_gaze if (show_raw_gaze and raw_gaze is not None) else None
3484 if spatial_axes:
3485 x_range, y_range, x_min_data, x_max_data, y_min_data, y_max_data = (
3486 _compute_axis_ranges(
3487 canvas_width,
3488 canvas_height,
3489 (fixations, x_field, y_field),
3490 (raw_for_range, "x", "y"),
3491 word_frames=[words] if not words.empty else [],
3492 fit_to_monitor=fit_to_monitor,
3493 )
3494 )
3495 else:
3496 x_range = [0, canvas_width]
3497 y_range = [canvas_height, 0]
3498 x_min_data = x_max_data = y_min_data = y_max_data = None
3500 # VIZ-9 "linear reading" mode: snap each fixation above the word it lands on,
3501 # so the saccade layer AND the fixation markers below draw from the snapped
3502 # positions. Off by default. The axis ranges above and the heatmaps below keep
3503 # the RECORDED positions (raw gaze density); only the drawn connectors and
3504 # markers move — and the Arc headroom just below, which must follow them.
3505 render_fix = fixations
3506 if (
3507 fixation_snap_to_word
3508 and spatial_axes
3509 and not fixations.empty
3510 and not words.empty
3511 ):
3512 render_fix = _snap_fixations_to_words(fixations, words, x_field, y_field)
3514 # VIZ-9 arc mode: the saccade arches rise ABOVE the fixations, so reserve
3515 # headroom at the top of the view (smaller y — the axis is inverted) or a wide
3516 # top-line saccade's apex gets clipped. Computed from the exact Bézier apex of
3517 # each segment so it's tight; only in Arc mode, so the default view is
3518 # unchanged. The apexes come from ``render_fix`` — the coordinates the
3519 # connectors are actually drawn from — because Snap to word can both widen a
3520 # saccade (two near-edge fixations jump to their words' centres) and lift its
3521 # endpoints (to the box tops), so an arc over the recorded positions would
3522 # under-reserve and clip the snapped curve.
3523 # Whole-monitor view (``fit_to_monitor``): the range still starts as the full
3524 # screen, and this only ever *grows* it — past the screen's top edge when an
3525 # arc would reach it — so a schematic arc is never silently cut off; the
3526 # screen itself is always shown whole.
3527 if (
3528 spatial_axes
3529 and saccade_render_mode == "Arc"
3530 and show_saccades
3531 and len(render_fix) > 1
3532 ):
3533 fo = render_fix.sort_values("timestamp_ms")
3534 fxv = pd.to_numeric(fo[x_field], errors="coerce").to_numpy(dtype=float)
3535 fyv = pd.to_numeric(fo[y_field], errors="coerce").to_numpy(dtype=float)
3536 apex = np.inf
3537 for i in range(len(fo) - 1):
3538 x0, y0, x1, y1 = fxv[i], fyv[i], fxv[i + 1], fyv[i + 1]
3539 if not np.isfinite((x0, y0, x1, y1)).all():
3540 continue
3541 # BUG-13: the curve's true peak, not its t=0.5 point — for endpoints
3542 # that differ in y the parabola crests higher than the midpoint
3543 # sample, and a wide top-line arc then clipped.
3544 apex = min(apex, _arch_apex_y(x0, y0, x1, y1, _ARCH_FRAC))
3545 if np.isfinite(apex):
3546 margin = 0.02 * abs(y_range[0] - y_range[1])
3547 y_range = [y_range[0], min(y_range[1], apex - margin)]
3549 # Stimulus-page background image (MultiplEYE / VIZ-4) — see
3550 # `_background_image_spec` for the placement contract.
3551 if spatial_axes:
3552 _add_background_image(
3553 fig,
3554 background_image,
3555 background_image_size,
3556 background_image_origin,
3557 background_image_opacity,
3558 )
3560 # Fix the display size up front so the data->screen scale is known: word
3561 # labels are then sized in that scale (true-to-scale text), and the same
3562 # fitted_w/fitted_h drive the final layout below.
3563 fitted_w, fitted_h = _fit_display_size(
3564 canvas_width, canvas_height, x_range, y_range, spatial_axes
3565 )
3566 scale = (
3567 _display_scale(x_range, y_range, fitted_w, fitted_h) if spatial_axes else 1.0
3568 )
3569 label_font_px = _word_label_font_px(
3570 words,
3571 scale=scale,
3572 line_spacing=line_spacing,
3573 manual_font_px=base_font_size,
3574 scale_text_to_boxes=scale_text_to_boxes,
3575 )
3577 # Words flagged by ``highlight_column`` (default the OneStop answer span
3578 # ``is_in_aspan``) get marked one of two ways:
3579 # - "Mark text": color the highlighted words dark pink (no border).
3580 # - "Mark border": draw a thin black outline around the span.
3581 has_highlight = (
3582 bool(highlight_column)
3583 and highlight_column in words.columns
3584 and not words.empty
3585 and bool(words[highlight_column].fillna(False).astype(bool).any())
3586 )
3587 highlight_text = has_highlight and critical_span_style == "Mark text"
3589 if spatial_axes and not words.empty:
3590 # Word-box grid (the "Bounding boxes" layer) and the "Mark border" span
3591 # overlay are independent: the span borders show even when the boxes are
3592 # off (then only the span outline is drawn).
3593 shapes = (
3594 build_word_boxes(
3595 words,
3596 color=settings.word_box_color,
3597 fill_color=settings.word_box_fill_color,
3598 fill_opacity=settings.word_box_fill_opacity,
3599 line_opacity=settings.word_box_line_opacity,
3600 )
3601 if show_words
3602 else []
3603 )
3604 if has_highlight and critical_span_style == "Mark border":
3605 shapes = shapes + build_critical_span_overlay(
3606 words, highlight_column, color=span_border_color
3607 )
3608 _add_highlight_key(fig, highlight_column, span_border_color, border=True)
3609 if shapes:
3610 fig.update_layout(shapes=shapes)
3611 if show_word_labels:
3612 _add_word_label_trace(
3613 fig,
3614 words,
3615 label_font_px,
3616 font_settings["family"],
3617 highlight_column=highlight_column if highlight_text else None,
3618 text_color=text_color,
3619 highlight_text_color=highlight_text_color,
3620 word_hover_measure=word_hover_measure,
3621 word_hover_fields=word_hover_fields,
3622 )
3624 if _add_raw_gaze_layer(
3625 fig,
3626 raw_gaze,
3627 show_raw_gaze=show_raw_gaze,
3628 raw_gaze_color=raw_gaze_color,
3629 raw_gaze_marker_size=raw_gaze_marker_size,
3630 raw_gaze_opacity=raw_gaze_opacity,
3631 ):
3632 legend_active = True
3634 if spatial_axes and show_heatmap and not fixations.empty:
3635 heatmap_rendered = True
3636 weights = fixations["duration_ms"] if heatmap_metric == "duration_ms" else None
3637 x_min = (
3638 x_min_data if x_min_data is not None else float(fixations[x_field].min())
3639 )
3640 x_max = (
3641 x_max_data if x_max_data is not None else float(fixations[x_field].max())
3642 )
3643 y_min = (
3644 y_min_data if y_min_data is not None else float(fixations[y_field].min())
3645 )
3646 y_max = (
3647 y_max_data if y_max_data is not None else float(fixations[y_field].max())
3648 )
3649 if heatmap_style == "Interpolated":
3650 _add_interpolated_heatmap(
3651 fig,
3652 fixations,
3653 x_field=x_field,
3654 y_field=y_field,
3655 x_min=x_min,
3656 x_max=x_max,
3657 y_min=y_min,
3658 y_max=y_max,
3659 weights=weights,
3660 heatmap_colorscale=heatmap_colorscale,
3661 show_colorbars=show_heatmap_colorbar,
3662 heatmap_norm=heatmap_norm,
3663 colorbar_style=cb_style,
3664 sigma_px=heatmap_sigma_px,
3665 )
3666 elif not words.empty:
3667 _add_word_level_heatmap(
3668 fig,
3669 words,
3670 fixations,
3671 x_field=x_field,
3672 y_field=y_field,
3673 weights=weights,
3674 heatmap_colorscale=heatmap_colorscale,
3675 heatmap_range=heatmap_range,
3676 show_colorbars=show_heatmap_colorbar,
3677 heatmap_norm=heatmap_norm,
3678 colorbar_style=cb_style,
3679 )
3680 else:
3681 _add_density_heatmap(
3682 fig,
3683 fixations,
3684 x_field=x_field,
3685 y_field=y_field,
3686 x_min=x_min,
3687 x_max=x_max,
3688 y_min=y_min,
3689 y_max=y_max,
3690 weights=weights,
3691 heatmap_colorscale=heatmap_colorscale,
3692 heatmap_range=heatmap_range,
3693 show_colorbars=show_heatmap_colorbar,
3694 heatmap_norm=heatmap_norm,
3695 colorbar_style=cb_style,
3696 )
3697 elif spatial_axes and show_heatmap and not words.empty:
3698 # Words-only dataset (no fixation report): fall back to the words
3699 # frame's own pre-aggregated reading measures for the box heatmap. The
3700 # corpus "word difficulty on the stimulus" view (AN-4) passes an explicit
3701 # ``word_heatmap_col`` + title so it can tint by any aggregate or rate.
3702 if word_heatmap_col is not None and word_heatmap_col in words.columns:
3703 heatmap_rendered = True
3704 values = pd.to_numeric(words[word_heatmap_col], errors="coerce").fillna(0.0)
3705 _draw_word_value_heatmap(
3706 fig,
3707 words,
3708 [float(v) for v in values],
3709 heatmap_colorscale=heatmap_colorscale,
3710 heatmap_range=heatmap_range,
3711 show_colorbars=show_heatmap_colorbar,
3712 heatmap_norm=heatmap_norm,
3713 colorbar_title=_plotly_literal(word_heatmap_title or "Value"),
3714 colorbar_style=cb_style,
3715 )
3716 else:
3717 measure = (
3718 "total_fixation_duration_ms"
3719 if heatmap_metric == "duration_ms"
3720 else "n_fixations"
3721 )
3722 if measure in words.columns:
3723 heatmap_rendered = True
3724 _add_word_measure_heatmap(
3725 fig,
3726 words,
3727 measure,
3728 heatmap_colorscale=heatmap_colorscale,
3729 heatmap_range=heatmap_range,
3730 show_colorbars=show_heatmap_colorbar,
3731 heatmap_norm=heatmap_norm,
3732 colorbar_style=cb_style,
3733 )
3735 # Saccade lines + optional direction arrowheads (drawn before the fixation
3736 # markers so the dots sit on top).
3737 if spatial_axes and show_saccades and len(fixations) > 1:
3738 # VIZ-8: "By type" colours each saccade by its reading class. The class is
3739 # geometry-derived per trial (like color-by-line above), so compute it
3740 # here from the trial's fixations + words — reuse a precomputed
3741 # `saccade_class` column if the pipeline already added one. Classify on the
3742 # RAW fixations (word_id is unchanged by the snap).
3743 # VIZ-19: "Forward / regression" is the same classification folded into
3744 # two buckets, so both non-uniform modes take this path.
3745 # VIZ-31: the same classification also backs the reading-class *filter*
3746 # (`saccade_classes` = the classes to draw, ``None`` meaning all), so
3747 # classify whenever either the colour mode or the filter needs it. A
3748 # filter naming every class is a no-op and takes the cheaper raw path.
3749 class_series = None
3750 two_way_saccades = saccade_color_mode == "Forward / regression"
3751 color_by_class = saccade_color_mode in ("By type", "Forward / regression")
3752 visible_classes = (
3753 None
3754 if saccade_classes is None
3755 or set(saccade_classes) >= set(SACCADE_CLASS_ORDER)
3756 else set(saccade_classes)
3757 )
3758 if color_by_class or visible_classes is not None:
3759 existing = fixations.get("saccade_class")
3760 if existing is not None:
3761 class_series = existing
3762 else:
3763 from .measures import classify_saccades
3765 class_series = classify_saccades(fixations, words)
3766 if _add_saccade_layer(
3767 fig,
3768 render_fix,
3769 x_field=x_field,
3770 y_field=y_field,
3771 color=saccade_color,
3772 width=saccade_width,
3773 style=saccade_style,
3774 show_arrows=show_saccade_arrows,
3775 saccade_classes=class_series,
3776 color_by_class=color_by_class,
3777 class_colors=saccade_class_colors,
3778 class_legend=saccade_type_legend,
3779 visible_classes=visible_classes,
3780 render_mode=saccade_render_mode,
3781 two_way=two_way_saccades,
3782 ):
3783 legend_active = True
3785 # Drift connectors (PRE-3): one faint grey vertical segment per fixation from
3786 # its original y (`connector_y`) to its drift-corrected y (the already-snapped
3787 # `fixations["y"]`). A SINGLE Scatter with None separators (scales with the
3788 # true-scale embed), drawn before the fixation markers so the dots sit on top.
3789 if (
3790 spatial_axes
3791 and show_connectors
3792 and connector_y is not None
3793 and not fixations.empty
3794 ):
3795 xs = pd.to_numeric(fixations[x_field], errors="coerce").to_numpy(dtype=float)
3796 y_corr = pd.to_numeric(fixations[y_field], errors="coerce").to_numpy(
3797 dtype=float
3798 )
3799 y_orig = np.asarray(connector_y, dtype=float)
3800 seg_x: list = []
3801 seg_y: list = []
3802 for xi, yo, yc in zip(xs, y_orig, y_corr):
3803 if not (np.isfinite(xi) and np.isfinite(yo) and np.isfinite(yc)):
3804 continue
3805 seg_x += [xi, xi, None]
3806 seg_y += [yo, yc, None]
3807 if seg_x:
3808 fig.add_trace(
3809 go.Scatter(
3810 x=seg_x,
3811 y=seg_y,
3812 mode="lines",
3813 line=dict(color="rgba(110,110,110,0.9)", width=1),
3814 opacity=0.3,
3815 hoverinfo="skip",
3816 showlegend=False,
3817 name="drift",
3818 )
3819 )
3821 if show_fixations and not fixations.empty:
3822 # ``render_fix`` == fixations unless VIZ-9 snap-to-word is on, in which
3823 # case the markers, order labels and colour-by-line use the snapped x/y.
3824 ordered = render_fix.sort_values("timestamp_ms")
3825 # Fixation classification (PRE-2, viz-only): SHORT / LONG / OUT-OF-BOUNDS,
3826 # each Off / Highlight / Discard. Apply Discard here — drop those rows from
3827 # `ordered` so they vanish from the markers, fixation-index labels and
3828 # marker-size scaling (the saccade layer above still bridges across them).
3829 # This changes only what's DRAWN; reading measures and exports are untouched.
3830 flags = fixation_flags or {}
3831 if flags:
3832 ordered = _discard_flagged_fixations(
3833 ordered, words, flags, spatial_axes=spatial_axes
3834 )
3835 # "Color by line" overrides the chosen color field: each fixation is
3836 # tinted by the text line it lands on (lines inferred from word
3837 # geometry). Rendered as discrete categories so the legend reads
3838 # "line: Line 1", "line: Line 2", …
3839 if color_by_line and spatial_axes and not words.empty:
3840 from .measures import assign_fixation_lines
3842 line_ids = assign_fixation_lines(ordered, words)
3843 color_data = line_ids.map(
3844 lambda v: f"Line {int(v) + 1}" if pd.notna(v) else "Out of bounds"
3845 )
3846 color_label = "line"
3847 is_numeric_color = False
3848 elif color_by == UNIFORM_COLOR_FIELD:
3849 # VIZ-17: no variable mapped to hue — size already encodes duration.
3850 color_data = None
3851 color_label = color_by
3852 is_numeric_color = False
3853 else:
3854 color_data = ordered[color_by] if color_by in ordered.columns else None
3855 color_label = color_by
3856 is_numeric_color = color_data is not None and pd.api.types.is_numeric_dtype(
3857 color_data
3858 )
3859 marker_color, category_legend = _resolve_marker_colors(
3860 color_data, is_numeric_color, fixation_color
3861 )
3862 sizes = _compute_marker_sizes(
3863 ordered["duration_ms"], marker_size_range, **_settings_size_scale(settings)
3864 )
3865 marker = dict(
3866 size=sizes,
3867 symbol=fixation_symbol or DEFAULT_FIXATION_SYMBOL,
3868 color=marker_color,
3869 colorscale=fixation_colorscale if is_numeric_color else None,
3870 showscale=show_colorbars and is_numeric_color,
3871 colorbar=_colorbar_dict(
3872 _column_title(color_label),
3873 orientation=colorbar_orientation,
3874 tickangle=colorbar_tickangle,
3875 tickfont_size=colorbar_tickfont_size,
3876 )
3877 if show_colorbars and is_numeric_color
3878 else None,
3879 cmin=fixation_color_range[0] if fixation_color_range else None,
3880 cmax=fixation_color_range[1] if fixation_color_range else None,
3881 line=dict(color=FIX_MARKER_OUTLINE, width=0.5),
3882 )
3883 # Marker alpha (VIZ-6): lower it so overlapping fixations show through.
3884 # Always set it (even at 1.0) so the slider is authoritative — Plotly's
3885 # variable-size scatter markers otherwise render at a ~0.7 default, so an
3886 # unset 1.0 looked translucent ("opacity at 1 wasn't really 1").
3887 marker["opacity"] = float(
3888 fixation_opacity if fixation_opacity is not None else 1.0
3889 )
3890 if hollow_fixations:
3891 marker = _make_hollow(marker)
3892 hover_fields = (
3893 ["order_in_trial", "duration_ms", "word_id"]
3894 if fixation_hover_fields is None
3895 else list(fixation_hover_fields)
3896 )
3897 customdata, hovertemplate = _hover_payload(
3898 ordered, hover_fields, fixation=True, words=words
3899 )
3900 glyph = FIXATION_GLYPH_SYMBOLS.get(fixation_symbol or "")
3901 if glyph:
3902 # VIZ-15: a shape Plotly's marker enum doesn't carry (♥), drawn as
3903 # text from the same marker dict (`_glyph_scatter_traces`); the
3904 # fixation-index labels move to their own trace since one Scatter has
3905 # only one text field, and a numeric colour bar to its own.
3906 for trace in _glyph_scatter_traces(
3907 ordered[x_field],
3908 ordered[y_field],
3909 marker,
3910 glyph,
3911 hovertemplate=hovertemplate,
3912 customdata=customdata,
3913 name="Fixations",
3914 showlegend=False,
3915 ):
3916 fig.add_trace(trace)
3917 bar = _glyph_colorbar_trace(marker, color_data if is_numeric_color else ())
3918 if bar is not None:
3919 fig.add_trace(bar)
3920 if show_order:
3921 fig.add_trace(
3922 go.Scatter(
3923 x=ordered[x_field],
3924 y=ordered[y_field],
3925 mode="text",
3926 text=ordered["order_in_trial"],
3927 textfont=dict(
3928 color=order_font_color,
3929 size=order_font_size,
3930 family=font_settings["family"],
3931 ),
3932 textposition="top center",
3933 hoverinfo="skip",
3934 name="Fixation index",
3935 showlegend=False,
3936 )
3937 )
3938 else:
3939 fig.add_trace(
3940 go.Scatter(
3941 x=ordered[x_field],
3942 y=ordered[y_field],
3943 mode="markers+text" if show_order else "markers",
3944 marker=marker,
3945 text=ordered["order_in_trial"] if show_order else None,
3946 textfont=dict(
3947 color=order_font_color,
3948 size=order_font_size,
3949 family=font_settings["family"],
3950 ),
3951 textposition="top center",
3952 hovertemplate=hovertemplate,
3953 customdata=customdata,
3954 name="Fixations",
3955 showlegend=False,
3956 )
3957 )
3958 legend_limit = len(_QUALITATIVE_PALETTE)
3959 truncated_legend = category_legend[:legend_limit]
3960 if category_legend:
3961 legend_active = True
3962 for category, color in truncated_legend:
3963 fig.add_trace(
3964 go.Scatter(
3965 x=[None],
3966 y=[None],
3967 mode="markers",
3968 marker=dict(
3969 size=10,
3970 color=color,
3971 line=dict(color=FIX_MARKER_OUTLINE, width=0.5),
3972 ),
3973 name=category
3974 if color_label == "line"
3975 else f"{_column_name(color_label)}: {category}",
3976 showlegend=True,
3977 hoverinfo="skip",
3978 )
3979 )
3980 if len(category_legend) > legend_limit:
3981 fig.add_trace(
3982 go.Scatter(
3983 x=[None],
3984 y=[None],
3985 mode="markers",
3986 marker=dict(size=10, color="#cccccc"),
3987 name=f"… +{len(category_legend) - legend_limit} more",
3988 showlegend=True,
3989 hoverinfo="skip",
3990 )
3991 )
3993 # Highlight overlays (PRE-2): mark SHORT / LONG / OUT-OF-BOUNDS fixations
3994 # in the chosen marker + colour, on top of the regular markers. Masks are
3995 # recomputed on the (post-discard) `ordered`; out-of-bounds needs word
3996 # boxes + spatial axes, short/long are duration-based and apply anywhere.
3997 if flags:
3998 _overlay = _fixation_flag_masks(
3999 ordered, words, flags, spatial_axes=spatial_axes
4000 )
4001 for _cat in _FIX_FLAG_CATEGORIES:
4002 _spec = flags.get(_cat, {})
4003 if _spec.get("mode") != "Highlight":
4004 continue
4005 _name = _FIX_FLAG_LABELS[_cat]
4006 hits = ordered[_overlay[_cat]]
4007 if hits.empty:
4008 continue
4009 legend_active = True
4010 fig.add_trace(
4011 go.Scatter(
4012 x=hits[x_field],
4013 y=hits[y_field],
4014 mode="markers",
4015 marker=dict(
4016 symbol=_spec.get("symbol") or "x",
4017 size=13,
4018 color=_spec.get("color") or OUT_OF_TEXT_COLOR,
4019 line=dict(color="#ffffff", width=1),
4020 ),
4021 name=_name,
4022 showlegend=True,
4023 hovertemplate=(
4024 f"{_name} fixation<br>x %{{x:.0f}}, y %{{y:.0f}}"
4025 "<extra></extra>"
4026 ),
4027 )
4028 )
4030 xaxis_cfg = dict(showticklabels=False, showgrid=False, zeroline=False, title=None)
4031 yaxis_cfg = dict(showticklabels=False, showgrid=False, zeroline=False, title=None)
4032 if spatial_axes:
4033 # automargin off: the colorbar/legend live in the reserved margin we size
4034 # below (_decoration_margins), so Plotly must not also shrink the
4035 # equal-aspect plot domain to fit them.
4036 xaxis_cfg.update(range=x_range, constrain="domain", automargin=False)
4037 yaxis_cfg.update(
4038 range=y_range,
4039 constrain="domain",
4040 scaleanchor="x",
4041 scaleratio=1,
4042 automargin=False,
4043 )
4044 _apply_coordinate_grid_axes(
4045 xaxis_cfg,
4046 yaxis_cfg,
4047 show=show_coordinate_grid,
4048 spacing=coordinate_grid_spacing,
4049 x_range=x_range,
4050 y_range=y_range,
4051 rendered_width=fitted_w,
4052 rendered_height=fitted_h,
4053 )
4054 else:
4055 xaxis_cfg.update(
4056 showticklabels=True, showgrid=True, title=_column_title(x_field)
4057 )
4058 yaxis_cfg.update(
4059 showticklabels=True, showgrid=True, title=_column_title(y_field)
4060 )
4062 shapes = list(fig.layout.shapes) if fig.layout.shapes else []
4063 if spatial_axes:
4064 shapes.append(
4065 dict(
4066 type="rect",
4067 x0=x_range[0],
4068 y0=y_range[1],
4069 x1=x_range[1],
4070 y1=y_range[0],
4071 line=dict(color="#000000", width=1),
4072 fillcolor="rgba(0,0,0,0)",
4073 # VIZ-5: the plot border is its own layer (a registration guide).
4074 name=_shape_layer_tag("frame"),
4075 )
4076 )
4078 # fitted_w / fitted_h were computed up front (so the label scale matched). A
4079 # colorbar (numeric colour / heatmap) or legend (discrete colour categories,
4080 # out-of-text, raw gaze) is given reserved margin so it never shrinks the
4081 # equal-aspect plot region — keeping the word labels matched to the boxes.
4082 decoration = (
4083 _decoration_margins(
4084 fitted_w,
4085 fitted_h,
4086 **_colorbar_reserves(
4087 (show_colorbars and is_numeric_color, colorbar_orientation),
4088 (show_heatmap_colorbar and heatmap_rendered, cb_style["orientation"]),
4089 ),
4090 legend=legend_active,
4091 coordinate_grid=show_coordinate_grid,
4092 )
4093 if spatial_axes
4094 else {"width": fitted_w, "height": fitted_h, "margin": dict(l=0, r=0, t=0, b=0)}
4095 )
4096 fig.update_layout(
4097 height=decoration["height"],
4098 width=decoration["width"],
4099 autosize=False,
4100 margin=decoration["margin"],
4101 xaxis=xaxis_cfg,
4102 yaxis=yaxis_cfg,
4103 legend=dict(orientation="h", yanchor="bottom", y=1.02, xanchor="right", x=1),
4104 template="plotly_white",
4105 # None leaves the template's default white; a hex value paints both the
4106 # plotting area and the surrounding paper (e.g. a neutral gray).
4107 plot_bgcolor=background_color,
4108 paper_bgcolor=background_color,
4109 font=font_settings,
4110 shapes=shapes,
4111 )
4112 add_illustration_label(fig, illustration_reasons, text=settings.illustration_text)
4113 return fig
4116def add_illustration_label(
4117 fig: go.Figure, reasons: Sequence[str] | None, *, text: str = ""
4118) -> go.Figure:
4119 """Stamp a figure and its metadata when it is schematic or transformed.
4121 ``text`` replaces the drawn wording; empty draws "Illustration · <reasons>".
4122 The reasons are recorded in the metadata either way."""
4123 reasons = [str(reason) for reason in (reasons or []) if reason]
4124 if not reasons:
4125 return fig
4126 fig.add_annotation(
4127 x=1,
4128 y=0,
4129 xref="paper",
4130 yref="paper",
4131 xanchor="right",
4132 yanchor="bottom",
4133 text=_plotly_literal(str(text).strip())
4134 # #374: a label the user switched on with nothing detected says
4135 # "Illustration" alone; "· manual label" told a reader nothing.
4136 or (
4137 "Illustration"
4138 if reasons == [MANUAL_LABEL_REASON]
4139 else "Illustration · " + "; ".join(reasons)
4140 ),
4141 showarrow=False,
4142 font=dict(size=10, color="#5f6368"),
4143 bgcolor="rgba(255,255,255,0.82)",
4144 borderpad=3,
4145 name=_ILLUSTRATION_LABEL_NAME,
4146 )
4147 _stack_bottom_right(fig)
4148 metadata = dict(fig.layout.meta or {})
4149 metadata["illustration"] = True
4150 metadata["illustration_reasons"] = reasons
4151 fig.update_layout(meta=metadata)
4152 return fig
4155# VIZ-3: alternative heatmap normalization. The colour of a heatmap cell maps
4156# LINEARLY between its z-range endpoints, so a few very-hot words (dwell times are
4157# heavy-tailed) can wash out the rest. "Log" instead maps colour to log1p(value),
4158# compressing the top of the range so mid-range words stay distinguishable. The
4159# transform is applied to the *values and the range endpoints together*, so the
4160# raw-unit `heatmap_range` slider keeps its meaning; only the colour curve changes.
4161_HEATMAP_NORMS = ("Linear", "Log")
4164def _apply_heatmap_norm(values, norm: str):
4165 """Transform heatmap values for the chosen normalization (VIZ-3).
4167 ``Log`` returns ``log1p(max(value, 0))`` (heavy-tail compression); anything
4168 else is the identity. Accepts a scalar or an array; returns the same shape."""
4169 arr = np.asarray(values, dtype=float)
4170 if norm == "Log":
4171 return np.log1p(np.clip(arr, 0.0, None))
4172 return arr
4175#: The duration-weighted word-box heatmap's colour-bar title: a box maps the
4176#: *summed* duration of its fixations, which a fixation-duration colour bar
4177#: beside it must not be mistaken for (round-7 review, findings 11 and 16).
4178_WORD_DWELL_TITLE = "Dwell time per word (ms)"
4181def _heatmap_title(base: str, norm: str) -> str:
4182 """Colour-bar title, marked ``(log)`` when the log normalization is active."""
4183 return f"{base} (log)" if norm == "Log" else base
4186def _add_word_level_heatmap(
4187 fig: go.Figure,
4188 words: pd.DataFrame,
4189 fixations: pd.DataFrame,
4190 *,
4191 x_field: str,
4192 y_field: str,
4193 weights: pd.Series | None,
4194 heatmap_colorscale: str,
4195 heatmap_range: tuple[float, float] | None,
4196 show_colorbars: bool,
4197 heatmap_norm: str = "Linear",
4198 colorbar_style: dict | None = None,
4199) -> None:
4200 # Pull the fixation coordinates (and optional weights) into numpy arrays once,
4201 # then test box membership per word against the arrays. Same O(words × fix)
4202 # work as before but without rebuilding pandas Series each iteration, and with
4203 # O(fix) memory (no full words × fix matrix).
4204 fx = pd.to_numeric(fixations[x_field], errors="coerce").to_numpy(dtype=float)
4205 fy = pd.to_numeric(fixations[y_field], errors="coerce").to_numpy(dtype=float)
4206 w_arr = (
4207 pd.to_numeric(weights, errors="coerce").to_numpy(dtype=float)
4208 if weights is not None
4209 else None
4210 )
4211 # Bin against the experiment's boxes (BUG-83) with the assignment's own
4212 # containment rule — the same boundary it uses and the heatmap then draws,
4213 # and half-open, so a fixation on a shared edge counts towards one word.
4214 from .measures import word_box_bounds, word_box_contains
4216 word_values = []
4217 for wx0, wy0, wx1, wy1 in zip(*word_box_bounds(words)):
4218 in_word = word_box_contains(fx, fy, wx0, wy0, wx1, wy1)
4219 val = (
4220 float(np.nansum(w_arr[in_word]))
4221 if w_arr is not None
4222 else float(in_word.sum())
4223 )
4224 word_values.append(val)
4226 _draw_word_value_heatmap(
4227 fig,
4228 words,
4229 word_values,
4230 heatmap_colorscale=heatmap_colorscale,
4231 heatmap_range=heatmap_range,
4232 show_colorbars=show_colorbars,
4233 heatmap_norm=heatmap_norm,
4234 colorbar_title="Fixation count" if weights is None else _WORD_DWELL_TITLE,
4235 colorbar_style=colorbar_style,
4236 )
4239def _add_word_measure_heatmap(
4240 fig: go.Figure,
4241 words: pd.DataFrame,
4242 measure: str,
4243 *,
4244 heatmap_colorscale: str,
4245 heatmap_range: tuple[float, float] | None,
4246 show_colorbars: bool,
4247 heatmap_norm: str = "Linear",
4248 colorbar_style: dict | None = None,
4249) -> None:
4250 """Word-box heatmap from a pre-aggregated per-word measure column.
4252 Used for words-only datasets (IA report without a fixation report): the
4253 usual heatmap aggregates fixation durations/counts into the boxes, but
4254 with no fixations the dataset's own reading measures (e.g. total fixation
4255 duration) carry the same information."""
4256 values = pd.to_numeric(words[measure], errors="coerce").fillna(0.0)
4257 _draw_word_value_heatmap(
4258 fig,
4259 words,
4260 [float(v) for v in values],
4261 heatmap_colorscale=heatmap_colorscale,
4262 heatmap_range=heatmap_range,
4263 show_colorbars=show_colorbars,
4264 heatmap_norm=heatmap_norm,
4265 colorbar_title="Fixation count"
4266 if measure == "n_fixations"
4267 else _WORD_DWELL_TITLE,
4268 colorbar_style=colorbar_style,
4269 )
4272def _draw_word_value_heatmap(
4273 fig: go.Figure,
4274 words: pd.DataFrame,
4275 word_values: list,
4276 *,
4277 heatmap_colorscale: str,
4278 heatmap_range: tuple[float, float] | None,
4279 show_colorbars: bool,
4280 heatmap_norm: str = "Linear",
4281 colorbar_title: str,
4282 colorbar_style: dict | None = None,
4283) -> None:
4284 from plotly.colors import sample_colorscale
4286 from .measures import word_box_bounds
4288 # Nonzero test on the RAW values (a word with no dwell stays uncoloured); the
4289 # colour position then maps through the chosen normalization (VIZ-3). Boxes
4290 # come from word_box_bounds so the tinted rects sit exactly on the outlines
4291 # build_word_boxes draws.
4292 boxes = zip(*word_box_bounds(words))
4293 nonzero_rows = [(box, v) for box, v in zip(boxes, word_values) if v > 0]
4294 if not nonzero_rows:
4295 return
4296 vals = [v for _, v in nonzero_rows]
4297 # Auto starts at 0: an empty word is the bottom of the scale.
4298 z_min_raw = heatmap_range[0] if heatmap_range else 0.0
4299 z_max_raw = heatmap_range[1] if heatmap_range else float(max(vals))
4300 z_min = float(_apply_heatmap_norm(z_min_raw, heatmap_norm))
4301 z_max = float(_apply_heatmap_norm(z_max_raw, heatmap_norm))
4302 z_span = max(z_max - z_min, 1e-9)
4304 heatmap_shapes = []
4305 for (x0, y0, x1, y1), v in nonzero_rows:
4306 tv = float(_apply_heatmap_norm(v, heatmap_norm))
4307 norm = max(0.0, min(1.0, (tv - z_min) / z_span))
4308 color = sample_colorscale(heatmap_colorscale, [norm])[0]
4309 heatmap_shapes.append(
4310 dict(
4311 type="rect",
4312 x0=x0,
4313 y0=y0,
4314 x1=x1,
4315 y1=y1,
4316 line=dict(width=0),
4317 fillcolor=color,
4318 opacity=0.5,
4319 layer="below",
4320 # VIZ-5: word-box heatmap rects belong to the heatmap layer.
4321 name=_shape_layer_tag("heatmap"),
4322 )
4323 )
4324 existing = list(fig.layout.shapes) if fig.layout.shapes else []
4325 fig.update_layout(shapes=existing + heatmap_shapes)
4326 if show_colorbars:
4327 fig.add_trace(
4328 go.Scatter(
4329 x=[None],
4330 y=[None],
4331 mode="markers",
4332 marker=dict(
4333 colorscale=heatmap_colorscale,
4334 showscale=True,
4335 cmin=z_min,
4336 cmax=z_max,
4337 colorbar=_colorbar_dict(
4338 _heatmap_title(colorbar_title, heatmap_norm),
4339 **(colorbar_style or {}),
4340 ),
4341 ),
4342 showlegend=False,
4343 hoverinfo="skip",
4344 # VIZ-5: the word-box heatmap's colorbar-carrier rides the heatmap
4345 # layer (name contains "heatmap" → classified there).
4346 name="heatmap colorbar",
4347 )
4348 )
4351def _add_density_heatmap(
4352 fig: go.Figure,
4353 fixations: pd.DataFrame,
4354 *,
4355 x_field: str,
4356 y_field: str,
4357 x_min: float,
4358 x_max: float,
4359 y_min: float,
4360 y_max: float,
4361 weights: pd.Series | None,
4362 heatmap_colorscale: str,
4363 heatmap_range: tuple[float, float] | None,
4364 show_colorbars: bool,
4365 heatmap_norm: str = "Linear",
4366 colorbar_style: dict | None = None,
4367) -> None:
4368 # A 40×40 count/duration grid drawn as a go.Heatmap (rather than
4369 # go.Histogram2d) so the colour mapping can go through _apply_heatmap_norm
4370 # (VIZ-3) — Plotly's Histogram2d bins internally and can't be log-scaled.
4371 xs = pd.to_numeric(fixations[x_field], errors="coerce")
4372 ys = pd.to_numeric(fixations[y_field], errors="coerce")
4373 valid = xs.notna() & ys.notna()
4374 if not valid.any():
4375 return
4376 if weights is not None:
4377 w = (
4378 pd.to_numeric(weights, errors="coerce")
4379 .reindex(fixations.index)
4380 .fillna(0.0)[valid]
4381 .to_numpy()
4382 )
4383 else:
4384 w = np.ones(int(valid.sum()))
4385 xv = xs[valid].to_numpy()
4386 yv = ys[valid].to_numpy()
4388 x_edges = np.linspace(x_min, x_max, 41)
4389 y_edges = np.linspace(y_min, y_max, 41)
4390 hist, _, _ = np.histogram2d(xv, yv, bins=[x_edges, y_edges], weights=w)
4391 grid = hist.T # rows index y, cols index x — the orientation go.Heatmap wants
4392 if grid.max() <= 0:
4393 return
4394 z = np.where(grid > 0, _apply_heatmap_norm(grid, heatmap_norm), np.nan)
4395 z_range = (
4396 _apply_heatmap_norm(np.asarray(heatmap_range, dtype=float), heatmap_norm)
4397 if heatmap_range
4398 else (None, None)
4399 )
4400 base_title = "Fixation density" if weights is None else "Dwell time per cell (ms)"
4401 fig.add_trace(
4402 go.Heatmap(
4403 x=(x_edges[:-1] + x_edges[1:]) / 2.0,
4404 y=(y_edges[:-1] + y_edges[1:]) / 2.0,
4405 z=z,
4406 colorscale=heatmap_colorscale,
4407 opacity=0.35,
4408 showscale=show_colorbars,
4409 colorbar=_colorbar_dict(
4410 _heatmap_title(base_title, heatmap_norm), **(colorbar_style or {})
4411 ),
4412 zmin=z_range[0],
4413 zmax=z_range[1],
4414 hoverinfo="skip",
4415 name="Fixation heatmap",
4416 )
4417 )
4420# Past this radius the 3-sigma kernel's normalising sum is taken from the
4421# Gaussian integral (exact to float precision at that sigma) instead of an array.
4422_KERNEL_EXACT_SUM_RADIUS = 100_000
4425def _gaussian_kernel_1d(sigma: float, max_radius: int | None = None) -> np.ndarray:
4426 """Normalized 1-D Gaussian kernel, truncated at 3 sigma.
4428 ``max_radius`` builds only the central taps, still normalized over the full
4429 3-sigma kernel, so a short axis pays for the taps it can reach rather than
4430 for sigma.
4431 """
4432 radius = max(1, round(sigma * 3))
4433 keep = radius if max_radius is None else max(0, min(radius, max_radius))
4434 offsets = np.arange(-keep, keep + 1)
4435 kernel = np.exp(-(offsets**2) / (2.0 * sigma * sigma))
4436 if keep == radius:
4437 total = kernel.sum()
4438 elif radius <= _KERNEL_EXACT_SUM_RADIUS:
4439 full = np.arange(-radius, radius + 1)
4440 total = np.exp(-(full**2) / (2.0 * sigma * sigma)).sum()
4441 else:
4442 total = (
4443 sigma
4444 * math.sqrt(2.0 * math.pi)
4445 * math.erf((radius + 0.5) / (sigma * math.sqrt(2.0)))
4446 )
4447 return kernel / total
4450def _blur_axis(grid: np.ndarray, sigma: float, axis: int) -> np.ndarray:
4451 """Zero-padded Gaussian blur along one axis; the output keeps ``grid``'s shape.
4453 ``np.convolve(mode="same")`` returns the *kernel's* length when the kernel
4454 is longer than the axis, which shifted the density off its coordinates on a
4455 short grid. Padding by the kernel radius and taking the ``valid`` part pins
4456 the length. Taps further out than the axis is long only ever meet the zero
4457 padding, so they are never built: the result is identical and the cost is
4458 bounded by the axis length, not by sigma.
4459 """
4460 kernel = _gaussian_kernel_1d(sigma, max_radius=grid.shape[axis] - 1)
4461 radius = len(kernel) // 2
4462 pad = [(0, 0)] * grid.ndim
4463 pad[axis] = (radius, radius)
4464 padded = np.pad(grid, pad)
4465 return np.apply_along_axis(
4466 lambda v: np.convolve(v, kernel, mode="valid"), axis, padded
4467 )
4470def _gaussian_blur_2d(
4471 grid: np.ndarray, sigma_rows: float, sigma_cols: float
4472) -> np.ndarray:
4473 """Separable Gaussian blur (a numpy-only stand-in for scipy.ndimage).
4475 Zero padding beyond the grid; the output always has the input's shape.
4476 """
4477 out = grid.astype(float)
4478 if out.size == 0:
4479 return out
4480 if sigma_rows and sigma_rows > 0:
4481 out = _blur_axis(out, sigma_rows, 0)
4482 if sigma_cols and sigma_cols > 0:
4483 out = _blur_axis(out, sigma_cols, 1)
4484 return out
4487# Interpolated-heatmap tuning. The Gaussian sigma defaults to a fraction of the
4488# larger data span — enough to merge a fixation cluster into one smooth blob
4489# without bleeding across neighbouring text lines.
4490_INTERP_GRID = 240 # cells along the wider axis
4491_INTERP_SIGMA_FRAC = 0.02 # sigma as a fraction of the larger data span
4492_INTERP_MIN_SIGMA_PX = 8.0
4495def interpolated_sigma_px(x_span: float, y_span: float) -> float:
4496 """The Interpolated heatmap's automatic Gaussian σ, in px: 2% of the data's
4497 larger span (fixations, word boxes and shown raw gaze), at least 8 px."""
4498 return max(_INTERP_MIN_SIGMA_PX, _INTERP_SIGMA_FRAC * max(x_span, y_span))
4501_INTERP_OPACITY = 0.45
4502_INTERP_FLOOR_FRAC = 0.02 # cells below this fraction of the peak render transparent
4503_INTERP_MIN_CELLS = 10 # cells along the narrower axis, at least
4504_INTERP_MIN_SPAN_PX = 2.0 * _INTERP_MIN_SIGMA_PX # narrower than this is widened
4507def _interp_grid_shape(x_span: float, y_span: float) -> tuple[int, int]:
4508 """(nx, ny) cells: _INTERP_GRID on the wider axis, a proportional share on
4509 the other, each within [_INTERP_MIN_CELLS, _INTERP_GRID]."""
4510 wide = max(x_span, y_span)
4511 narrow = min(x_span, y_span)
4512 share = round(_INTERP_GRID * narrow / wide) if wide > 0 else _INTERP_GRID
4513 n_narrow = int(min(_INTERP_GRID, max(_INTERP_MIN_CELLS, share)))
4514 if x_span >= y_span:
4515 return _INTERP_GRID, n_narrow
4516 return n_narrow, _INTERP_GRID
4519def _add_interpolated_heatmap(
4520 fig: go.Figure,
4521 fixations: pd.DataFrame,
4522 *,
4523 x_field: str,
4524 y_field: str,
4525 x_min: float,
4526 x_max: float,
4527 y_min: float,
4528 y_max: float,
4529 weights: pd.Series | None,
4530 heatmap_colorscale: str,
4531 show_colorbars: bool,
4532 heatmap_norm: str = "Linear",
4533 colorbar_style: dict | None = None,
4534 sigma_px: float | None = None,
4535 title: str | None = None,
4536) -> None:
4537 """Smooth, word-box-independent fixation heatmap (Gaussian-interpolated).
4539 Bins the fixations onto a fine grid (weighted by duration when ``weights``
4540 is given), then blurs with a Gaussian — the classic eye-movement heatmap
4541 (cf. PyGaze's gaze plotter). Empty cells render transparent so the reading
4542 text stays legible underneath.
4543 """
4544 xs = pd.to_numeric(fixations[x_field], errors="coerce")
4545 ys = pd.to_numeric(fixations[y_field], errors="coerce")
4546 valid = xs.notna() & ys.notna()
4547 if not valid.any():
4548 return
4549 if weights is not None:
4550 w = (
4551 pd.to_numeric(weights, errors="coerce")
4552 .reindex(fixations.index)
4553 .fillna(0.0)[valid]
4554 .to_numpy()
4555 )
4556 else:
4557 w = np.ones(int(valid.sum()))
4558 xs = xs[valid].to_numpy()
4559 ys = ys[valid].to_numpy()
4561 x_span = max(x_max - x_min, 1.0)
4562 y_span = max(y_max - y_min, 1.0)
4563 sigma_px = float(sigma_px or interpolated_sigma_px(x_span, y_span))
4564 # An axis with (next to) no extent — coincident fixations, one row or one
4565 # column of a fixation-only import — is widened around its centre to the
4566 # blob's own size (±3 sigma), so the edges match the span the cells and the
4567 # sigma are computed from, and the blob is round rather than a sliver.
4568 min_span = max(_INTERP_MIN_SPAN_PX, 6.0 * sigma_px)
4569 if x_max - x_min < _INTERP_MIN_SPAN_PX:
4570 x_mid = (x_min + x_max) / 2.0
4571 x_min, x_max, x_span = x_mid - min_span / 2, x_mid + min_span / 2, min_span
4572 if y_max - y_min < _INTERP_MIN_SPAN_PX:
4573 y_mid = (y_min + y_max) / 2.0
4574 y_min, y_max, y_span = y_mid - min_span / 2, y_mid + min_span / 2, min_span
4575 # The cell budget sits on the wider axis and the other gets its share of it,
4576 # so neither axis, nor the grid, outgrows _INTERP_GRID (squared): a tall,
4577 # narrow reading must not ask for 240 cells across and 96,000 down.
4578 nx, ny = _interp_grid_shape(x_span, y_span)
4579 x_edges = np.linspace(x_min, x_max, nx + 1)
4580 y_edges = np.linspace(y_min, y_max, ny + 1)
4581 # histogram2d returns shape (nx, ny); transpose so rows index y, cols index x
4582 # (the orientation go.Heatmap's z expects).
4583 hist, _, _ = np.histogram2d(xs, ys, bins=[x_edges, y_edges], weights=w)
4584 grid = hist.T
4586 blurred = _gaussian_blur_2d(
4587 grid, sigma_rows=sigma_px / (y_span / ny), sigma_cols=sigma_px / (x_span / nx)
4588 )
4589 peak = float(blurred.max())
4590 if peak <= 0:
4591 return
4592 # Near-zero cells -> NaN so Plotly renders them transparent (only populated
4593 # regions get tinted, keeping the text readable). The remaining density maps
4594 # through the chosen normalization (VIZ-3; log1p(0)=0 keeps the floor at 0).
4595 z = np.where(blurred < peak * _INTERP_FLOOR_FRAC, np.nan, blurred)
4596 z = _apply_heatmap_norm(z, heatmap_norm)
4598 base_title = title or (
4599 "Dwell-time density" if weights is not None else "Fixation density"
4600 )
4601 fig.add_trace(
4602 go.Heatmap(
4603 x=(x_edges[:-1] + x_edges[1:]) / 2.0,
4604 y=(y_edges[:-1] + y_edges[1:]) / 2.0,
4605 z=z,
4606 colorscale=heatmap_colorscale,
4607 opacity=_INTERP_OPACITY,
4608 showscale=show_colorbars,
4609 # z is a Gaussian-smoothed density in arbitrary (weighted) units, not
4610 # the per-word counts/ms the `heatmap_range` slider is calibrated for,
4611 # so it autoscales from 0 rather than borrowing that range.
4612 zmin=0.0,
4613 colorbar=_colorbar_dict(
4614 _heatmap_title(base_title, heatmap_norm), **(colorbar_style or {})
4615 ),
4616 hoverinfo="skip",
4617 name="Fixation heatmap",
4618 )
4619 )
4622# =============================================================================
4623# Scanpath animation — one or two scanpaths on a shared real reading-time clock
4624# =============================================================================
4626# Floor on the ▶ Play button's own per-frame duration: ~one 60 fps display frame.
4627# That duration only drives Plotly's frame queue, which is what a figure plays on
4628# where the wall-clock player isn't embedded (`fig.show()`); every HTML surface
4629# replays on `animation_player_post_script` instead (BUG-93).
4630_ANIM_MIN_FRAME_MS = 16
4631# VIZ-11: animation frames sit on a UNIFORM time grid (one every
4632# _ANIM_GRID_STEP_MS of reading) rather than one per fixation onset, so the
4633# slider scrubs linearly through seconds regardless of how fixations cluster or
4634# how many scanpaths overlay. The grid coarsens past _ANIM_MAX_FRAMES so a long
4635# reading doesn't emit thousands of frames (which would balloon the GIF/MP4
4636# export); quantization is then at most one grid step.
4637_ANIM_GRID_STEP_MS = 100.0
4638_ANIM_MAX_FRAMES = 360
4640# Vertical space (px) reserved BELOW the animation plot for the transport
4641# controls (play / pause / restart buttons + the time slider with its "Elapsed"
4642# readout). The figure is grown by this much (plus a small safety buffer) and
4643# the controls are placed in the bottom margin, so Plotly's automargin never has
4644# to shrink the equal-aspect plot to fit them — which would make the word boxes
4645# smaller than the true-to-scale label font computed for fitted_h (the
4646# text-too-large bug). Keeps the animation plot the SAME size as the static one.
4647_CONTROLS_MARGIN_PX = 116
4648_CONTROLS_SAFETY_PX = 24
4651def _scanpath_anim_specs(
4652 entries,
4653 marker_size_range,
4654 scale: str = DEFAULT_MARKER_SIZE_SCALE,
4655 duration_range=DEFAULT_MARKER_DURATION_RANGE,
4656 *,
4657 size_ranges=None,
4658):
4659 """Build per-scanpath animation specs from (fixations, color, label) entries.
4661 Empty/None fixations are skipped. Onsets are the recorded ``timestamp_ms``
4662 rebased to each reading's first fixation, so multiple scanpaths share one
4663 *real reading-time* clock. When timestamps aren't real times — missing, or
4664 the 0,1,2,… row index ``data.normalize_fixations`` synthesises when the
4665 source has no timestamp column — fixations are instead laid out back-to-back
4666 by their durations. Marker sizes use the figure's duration scale; under the
4667 relative scale they span the COMBINED durations, so equal durations still
4668 render at equal sizes across the two scanpaths.
4670 ``size_ranges`` (one per entry, default ``marker_size_range`` for each)
4671 gives each scanpath its own size range, as Compare's per-scanpath *Size*
4672 does: the duration scale stays shared — one duration is one *fraction* of
4673 the range on either side — and each side maps that fraction onto its own.
4674 """
4675 from .measures import rebased_fixation_onsets
4677 if size_ranges is None:
4678 size_ranges = [marker_size_range] * len(entries)
4679 specs = []
4680 for (fix_df, color, label), size_range in zip(entries, size_ranges):
4681 if fix_df is None or fix_df.empty:
4682 continue
4683 ordered = fix_df.sort_values("timestamp_ms").reset_index(drop=True)
4684 dur = pd.to_numeric(ordered["duration_ms"], errors="coerce").fillna(0)
4685 # Recorded-timestamp-vs-synthetic-index heuristic (shared with the
4686 # similarity time-curve): trust recorded timestamps only when they look
4687 # like real times, else lay fixations back-to-back by their durations.
4688 onsets = rebased_fixation_onsets(ordered)
4689 specs.append(
4690 dict(
4691 ordered=ordered,
4692 dur=dur,
4693 onsets=onsets,
4694 end=float(onsets[-1] + dur.iloc[-1]),
4695 color=color,
4696 label=label,
4697 size_range=tuple(size_range),
4698 )
4699 )
4700 if specs:
4701 # Each duration's place on the shared scale, 0..1, then sized in its
4702 # own scanpath's range — identical to sizing the combined durations in
4703 # one range whenever the two ranges agree.
4704 fractions = _compute_marker_sizes(
4705 pd.concat([s["dur"] for s in specs], ignore_index=True),
4706 (0.0, 1.0),
4707 scale,
4708 duration_range,
4709 )
4710 cursor = 0
4711 for s in specs:
4712 n = len(s["dur"])
4713 low, high = s["size_range"]
4714 s["sizes"] = low + np.asarray(
4715 fractions[cursor : cursor + n], dtype=float
4716 ) * (high - low)
4717 cursor += n
4718 return specs
4721def _anim_frame_duration_ms(frame_step_ms: float, playback_speed: float) -> int:
4722 """▶ Play's own per-frame duration: one grid step at the playback speed.
4724 Floored at ``_ANIM_MIN_FRAME_MS``. The only part of a replay the speed
4725 changes besides ``layout.meta`` — which is what lets `set_replay_clock`
4726 re-time a built replay (PERF-15)."""
4727 return int(max(frame_step_ms / max(playback_speed, 1e-6), _ANIM_MIN_FRAME_MS))
4730def _anim_timeline(specs, *, grid_step_ms=None, max_frames=None):
4731 """Uniform time-grid frame timeline across all scanpaths (VIZ-11).
4733 Returns ``(frame_times, frame_step_ms, reading_span_ms)``. Frames are
4734 emitted on a **uniform time grid** — one every ``step`` ms, where ``step`` is
4735 ``grid_step_ms`` unless that would exceed ``max_frames`` frames (then it
4736 coarsens) — so the slider scrubs linearly through reading time no matter how
4737 fixations cluster or how many scanpaths overlay (the union of onset sets is
4738 meaningless for >1 reader). ``frame_step_ms`` is that exact step (``0.0`` with
4739 no frames). None of it depends on the playback speed: ▶ Play's own per-frame
4740 duration is :func:`_anim_frame_duration_ms` of the step, used only where the
4741 wall-clock player isn't embedded, and the player shows frame k once
4742 ``frame_times[k] / playback_speed`` has elapsed, so a replay takes
4743 ``reading_span_ms / playback_speed`` (BUG-93). Frame *content* is
4744 unchanged — every fixation whose onset ≤ t shows at time t. All readings are
4745 rebased to t=0; ``reading_span_ms`` is the longest reading's span. Returns an
4746 empty grid when there is nothing to animate.
4748 Both knobs are user-facing (VIZ-11 follow-up): they trade smoothness against
4749 frame count, which is what the GIF/MP4 export size and render time are made
4750 of. Defaults are ``_ANIM_GRID_STEP_MS`` / ``_ANIM_MAX_FRAMES``.
4751 """
4752 step_pref = float(grid_step_ms if grid_step_ms else _ANIM_GRID_STEP_MS)
4753 cap = int(max_frames if max_frames else _ANIM_MAX_FRAMES)
4754 reading_span_ms = max((s["end"] for s in specs), default=0.0)
4755 if not specs or reading_span_ms <= 0:
4756 return [], 0.0, reading_span_ms
4757 step = max(step_pref, reading_span_ms / max(cap, 1))
4758 frame_times = [
4759 min(k * step, reading_span_ms) for k in range(int(reading_span_ms // step) + 1)
4760 ]
4761 # Land the final frame exactly on the reading end so it reveals everything.
4762 if frame_times[-1] < reading_span_ms:
4763 frame_times.append(reading_span_ms)
4764 return frame_times, step, reading_span_ms
4767def _revealed_xy(all_x, all_y, kk):
4768 """Full-length x/y with only the first ``kk`` fixations revealed.
4770 Not-yet-reached fixations are masked to ``None`` so Plotly draws nothing
4771 there. The array length is the SAME in every frame — the replay reveals a
4772 fixation by un-masking its coordinate, never by growing the array — which is
4773 what lets the Play button animate with ``redraw=False`` (only positions
4774 change, so Plotly skips redrawing the static word boxes/labels each frame).
4775 """
4776 n = len(all_x)
4777 xs = [all_x[i] if i < kk else None for i in range(n)]
4778 ys = [all_y[i] if i < kk else None for i in range(n)]
4779 return xs, ys
4782def _revealed_saccade_xy(all_x, all_y, kk):
4783 """Constant-length saccade polyline for the first ``kk`` fixations.
4785 Every consecutive fixation pair occupies a fixed ``(x0, x1, None)`` slot;
4786 segments past the ``kk``-th fixation are blanked to ``None`` so the trace
4787 length never changes frame to frame (same ``redraw=False`` requirement as
4788 :func:`_revealed_xy`). Only which segments are drawn changes.
4789 """
4790 sx, sy = [], []
4791 for j in range(len(all_x) - 1):
4792 if j < kk - 1:
4793 sx.extend([all_x[j], all_x[j + 1], None])
4794 sy.extend([all_y[j], all_y[j + 1], None])
4795 else:
4796 sx.extend([None, None, None])
4797 sy.extend([None, None, None])
4798 return sx, sy
4801def _revealed_arrow_xy(all_x, all_y, seg_index, kk):
4802 """Constant-length saccade-arrow positions for the first ``kk`` fixations.
4804 An arrowhead belongs to the saccade leaving fixation ``seg_index[j]``, so it
4805 appears exactly when :func:`_revealed_saccade_xy` draws that segment — the
4806 arrows reveal *with* their saccades instead of all standing there from frame
4807 zero. Hidden arrows are masked to ``None`` rather than dropped, keeping the
4808 array (and its ``marker.angle``) the same length every frame, which is what
4809 the ``redraw=False`` playback needs.
4810 """
4811 xs = [x if seg_index[j] < kk - 1 else None for j, x in enumerate(all_x)]
4812 ys = [y if seg_index[j] < kk - 1 else None for j, y in enumerate(all_y)]
4813 return xs, ys
4816def animation_playback_ms(
4817 fixations_list, playback_speed, *, grid_step_ms=None, max_frames=None
4818):
4819 """Reading span and *actual* animation runtime for the given scanpath(s).
4821 Returns ``(reading_span_ms, playback_ms)``. ``playback_ms`` is what the replay
4822 takes: the wall-clock player (:func:`animation_player_post_script`) reaches the
4823 last frame once ``reading_span_ms / playback_speed`` has elapsed, so that is
4824 the time the side panel quotes and a GIF/MP4 lasts (BUG-93). Both 0 when there
4825 are no fixations.
4826 """
4827 summary = animation_timeline_summary(
4828 fixations_list, playback_speed, grid_step_ms=grid_step_ms, max_frames=max_frames
4829 )
4830 return summary["reading_span_ms"], summary["playback_ms"]
4833def animation_timeline_summary(
4834 fixations_list, playback_speed, *, grid_step_ms=None, max_frames=None
4835) -> dict:
4836 """What the chosen frame grid actually produces, without building the figure.
4838 VIZ-11 follow-up: the grid step and the frame cap are user controls, so the UI
4839 has to show their consequence — frame count, the effective step, and whether
4840 the cap *coarsened* the requested step. Silently coarsening is the thing that
4841 made the old hard-coded behaviour opaque.
4843 Returns ``{"n_frames", "step_ms", "requested_step_ms", "coarsened",
4844 "frame_duration_ms", "reading_span_ms", "playback_ms"}``.
4845 """
4846 requested = float(grid_step_ms if grid_step_ms else _ANIM_GRID_STEP_MS)
4847 specs = _scanpath_anim_specs(
4848 [(f, None, None) for f in fixations_list], DEFAULT_MARKER_SIZE_RANGE
4849 )
4850 frame_times, frame_step_ms, reading_span_ms = _anim_timeline(
4851 specs, grid_step_ms=grid_step_ms, max_frames=max_frames
4852 )
4853 n_frames = len(frame_times)
4854 step = (frame_times[1] - frame_times[0]) if n_frames > 1 else float(reading_span_ms)
4855 return {
4856 "n_frames": n_frames,
4857 "step_ms": float(step),
4858 "requested_step_ms": requested,
4859 "coarsened": bool(n_frames > 1 and step > requested + 1e-6),
4860 "frame_duration_ms": _anim_frame_duration_ms(frame_step_ms, playback_speed),
4861 "reading_span_ms": float(reading_span_ms),
4862 "playback_ms": float(reading_span_ms) / max(playback_speed, 1e-6),
4863 }
4866# BUG-93 — the replay's clock. Plotly's own ▶ Play steps a frame on the first
4867# display tick *after* its duration and restarts the next frame's clock from
4868# there, so every hold rounds up to whole ticks and the rounding accumulates: on
4869# a 60 Hz screen a 40 ms frame lasts 50 ms, and a 20.8 s reading replayed in 26 s.
4870# No duration can fix that from here — the tick is the viewer's. So every HTML
4871# surface (`tabs._true_scale_plot_html`, `tabs._animation_html`,
4872# `api.save_figure`) embeds a small player that shows whichever frame the wall
4873# clock has reached, reading the frame times, the speed and VIZ-10's autoplay
4874# intent off `fig.layout.meta`, where `make_scanpath_animation` stamps them.
4875_AUTOPLAY_META_FLAG = "scanpath_autoplay"
4876_REPLAY_META_TIMES = "scanpath_frame_times_ms"
4877_REPLAY_META_SPEED = "scanpath_playback_speed"
4879# `{plot_id}` stays literal: plotly.py substitutes it (a plain `str.replace`, so
4880# the braces need no escaping) and runs the script in a `.then()` after `newPlot`.
4881_REPLAY_PLAYER_JS = """(function () {
4882 var gd = document.getElementById('{plot_id}');
4883 if (!gd) { return; }
4884 var tries = 0;
4885 // Frames live on gd._transitionData._frames (gd.frames is undefined) and are
4886 // attached after newPlot resolves, so wait for them rather than a fixed delay.
4887 (function init() {
4888 var meta = gd.layout && gd.layout.meta;
4889 var td = gd._transitionData;
4890 if (typeof Plotly === 'undefined' || !gd.on || !meta ||
4891 !(td && td._frames && td._frames.length)) {
4892 if (++tries < 200) { setTimeout(init, 50); }
4893 return;
4894 }
4895 run(meta);
4896 })();
4898 function run(meta) {
4899 var times = meta.scanpath_frame_times_ms;
4900 var speed = meta.scanpath_playback_speed;
4901 if (!times || !times.length || !(speed > 0)) { return; }
4902 var last = times.length - 1;
4903 var jump = {mode: 'immediate', frame: {duration: 0, redraw: false},
4904 transition: {duration: 0}};
4905 var shown = 0, raf = null, t0 = 0, resume = false;
4907 function frameAt(ms) { // the last frame whose reading time has been reached
4908 var lo = 0, hi = last;
4909 while (lo < hi) {
4910 var mid = (lo + hi + 1) >> 1;
4911 if (times[mid] <= ms) { lo = mid; } else { hi = mid - 1; }
4912 }
4913 return lo;
4914 }
4915 function show(k) {
4916 shown = k;
4917 Plotly.animate(gd, [String(k)], jump);
4918 }
4919 // A late tick skips frames rather than falling behind the clock.
4920 function tick() {
4921 if (!gd.isConnected) { raf = null; leave(); return; }
4922 var k = frameAt((performance.now() - t0) * speed);
4923 if (k !== shown) { show(k); }
4924 raf = k < last ? requestAnimationFrame(tick) : null;
4925 }
4926 function stop() {
4927 if (raf !== null) { cancelAnimationFrame(raf); raf = null; }
4928 }
4929 function play() {
4930 if (raf !== null) { return; } // already playing: keep the clock
4931 show(shown < last ? shown : 0); // at the end, Play starts over
4932 t0 = performance.now() - times[shown] / speed;
4933 raf = requestAnimationFrame(tick);
4934 }
4936 // Whatever put a frame on screen — this clock, the slider, Restart — Play
4937 // resumes from it.
4938 gd.on('plotly_animatingframe', function (e) {
4939 var k = parseInt(e && e.name, 10);
4940 if (k >= 0 && k <= last) { shown = k; }
4941 });
4942 gd.on('plotly_buttonclicked', function (e) {
4943 if (e && e.button && e.button.name === 'play') { play(); } else { stop(); }
4944 });
4945 gd.on('plotly_sliderstart', stop);
4946 gd.on('plotly_sliderchange', function (e) { if (e && e.interaction) { stop(); } });
4947 // A background tab gets no ticks; carry on from the same frame on return.
4948 function onVisibility() {
4949 if (!gd.isConnected) { stop(); leave(); return; }
4950 if (document.hidden) { resume = raf !== null; stop(); }
4951 else if (resume) { resume = false; play(); }
4952 }
4953 // A page that swaps content without reloading (the docs site) can drop the
4954 // plot; let go of the document then, so the plot can be collected.
4955 function leave() {
4956 document.removeEventListener('visibilitychange', onVisibility);
4957 }
4958 document.addEventListener('visibilitychange', onVisibility);
4960 // Plotly still draws the ▶ Play button; this clock takes over what it does.
4961 var edit = {};
4962 (gd.layout.updatemenus || []).forEach(function (menu, i) {
4963 (menu.buttons || []).forEach(function (button, j) {
4964 if (button.name === 'play') {
4965 edit['updatemenus[' + i + '].buttons[' + j + '].execute'] = false;
4966 }
4967 });
4968 });
4969 Promise.resolve(Plotly.relayout(gd, edit)).then(function () {
4970 if (meta.scanpath_autoplay) { play(); }
4971 });
4972 }
4973})();"""
4976def animation_player_post_script(fig) -> str | None:
4977 """The replay player for an animated scanpath, or ``None`` if there is none.
4979 Pass it to ``fig.to_html(post_script=…)`` / ``write_html(post_script=…)``
4980 (with ``auto_play=False``) for any figure :func:`make_scanpath_animation`
4981 built — or its ``to_dict()``; ``None`` — for a static figure, or a replay
4982 with no frames — is what those calls take for "no script" (BUG-93).
4984 The player keeps the replay on the wall clock: at every display tick it
4985 shows the last frame whose reading time ``elapsed × playback_speed`` has
4986 reached, so a replay takes ``reading span / playback_speed`` exactly — a slow
4987 tick skips frames instead of pushing every later one back — and a background
4988 tab pauses it. It takes the ▶ Play button over (the button's own command is
4989 switched off with ``execute: false``, Plotly's hook for exactly this, and the
4990 click still arrives as ``plotly_buttonclicked``); Pause, Restart and the time
4991 slider keep their own commands, and any of them stops the clock. VIZ-10's
4992 autoplay starts it on load, from the first frame.
4994 It **polls** for Plotly and the figure's frames before starting. Two things
4995 made a one-shot kick-off silently never fire (VIZ-10), both confirmed
4996 against a live Plotly build: frames live on ``gd._transitionData._frames``,
4997 **not** ``gd.frames`` (``undefined``), and the library can arrive late (CDN
4998 latency, the true-scale iframe mount) and attaches its frames only after
4999 ``newPlot`` resolves. Polling every 50 ms (capped at ~10 s) covers all of it,
5000 on the live embed and saved HTML alike.
5002 Without the script — ``fig.show()``, or a plain ``write_html`` — the figure
5003 still plays on Plotly's own queue, at the frame duration the ▶ Play button
5004 carries.
5005 """
5006 if isinstance(fig, dict): # a figure's `to_dict()`
5007 meta = (fig.get("layout") or {}).get("meta")
5008 else:
5009 meta = getattr(fig.layout, "meta", None)
5010 if not isinstance(meta, dict) or not meta.get(_REPLAY_META_TIMES):
5011 return None
5012 return _REPLAY_PLAYER_JS
5015# PERF-17 — the replay's frames, as its HTML page carries them. Every frame
5016# restates every animated trace at full length (see `_revealed_xy`), ~33 KB a
5017# frame however little changed, so a 2,001-frame replay was a 66 MB page. The
5018# page instead carries frame 0 whole and, for each later frame, only what changed
5019# since the one before; this decoder rebuilds the exact frame list in the browser
5020# and hands it to `Plotly.addFrames`, ahead of the player (which already polls for
5021# the frames). Delta-against-the-previous-frame can't be Plotly's own frames: the
5022# slider jumps from any frame to any other, so each frame has to be complete.
5023#
5024# A delta node is `null` (unchanged), `[0, value]` (replaced), `[1, {key: node},
5025# [dropped keys]?]` (an object's changed keys), or `[2, [indices], [values]]` (a
5026# same-length array's changed slots). Unchanged values are shared between frames
5027# rather than copied: Plotly copies a frame's objects before applying it and
5028# never writes into a frame. `__SCANPATH_PACKED_FRAMES__` is replaced by the
5029# JSON; `{plot_id}` stays literal for plotly.py, as in the player.
5030_PACKED_FRAMES_TOKEN = "__SCANPATH_PACKED_FRAMES__"
5031_REPLAY_FRAMES_JS = """(function () {
5032 var gd = document.getElementById('{plot_id}');
5033 if (!gd || typeof Plotly === 'undefined') { return; }
5034 var packed = __SCANPATH_PACKED_FRAMES__;
5035 var own = Object.prototype.hasOwnProperty;
5036 function patch(prev, node) {
5037 if (node === null) { return prev; }
5038 var out, i, k;
5039 if (node[0] === 0) { return node[1]; }
5040 if (node[0] === 1) {
5041 out = {};
5042 for (k in prev) { if (own.call(prev, k)) { out[k] = prev[k]; } }
5043 for (k in node[1]) { if (own.call(node[1], k)) { out[k] = patch(prev[k], node[1][k]); } }
5044 for (i = 0; node[2] && i < node[2].length; i++) { delete out[node[2][i]]; }
5045 return out;
5046 }
5047 out = prev.slice();
5048 for (i = 0; i < node[1].length; i++) { out[node[1][i]] = node[2][i]; }
5049 return out;
5050 }
5051 var state = {};
5052 var frames = packed.map(function (f) {
5053 var frame = {}, k;
5054 for (k in f) { if (own.call(f, k) && k !== 'p') { frame[k] = f[k]; } }
5055 if (f.p) {
5056 frame.data = f.p.map(function (node, i) {
5057 var slot = f.traces ? f.traces[i] : i;
5058 state[slot] = patch(state[slot], node);
5059 return state[slot];
5060 });
5061 }
5062 return frame;
5063 });
5064 Plotly.addFrames(gd, frames);
5065})();"""
5067_ABSENT = object()
5070def _same_value(a, b) -> bool:
5071 """Whether two figure values serialize to the same JSON value.
5073 Strict where JSON is: ``True`` is not ``1``. Errs towards *different* —
5074 a value judged different is merely carried again, never lost.
5075 """
5076 if type(a) is not type(b):
5077 return False
5078 if isinstance(a, np.ndarray):
5079 if a.dtype != b.dtype or a.shape != b.shape:
5080 return False
5081 if a.dtype.hasobject:
5082 return _same_value(a.tolist(), b.tolist())
5083 return a.tobytes() == b.tobytes()
5084 if isinstance(a, dict):
5085 return a.keys() == b.keys() and all(_same_value(a[k], b[k]) for k in a)
5086 if isinstance(a, (list, tuple)):
5087 if len(a) != len(b):
5088 return False
5089 try:
5090 if a != b: # C speed for the common case: plain numbers and strings
5091 return False
5092 except ValueError: # an ndarray inside: no truth value
5093 pass
5094 except TypeError: # `pd.NA` against anything else: no truth value either
5095 return False
5096 kinds = list(map(type, a))
5097 if kinds != list(map(type, b)):
5098 return False
5099 kind_set = set(kinds)
5100 if any(k in (list, tuple, dict, np.ndarray) for k in kind_set):
5101 return all(_same_value(x, y) for x, y in zip(a, b))
5102 # `==` holds -0.0 equal to 0.0, which JSON writes apart. Only an array
5103 # holding a zero pays for the sign check.
5104 if any(issubclass(k, float) for k in kind_set) and 0.0 in a:
5105 return all(
5106 not isinstance(x, float) or x != 0.0 or _same_sign(x, y)
5107 for x, y in zip(a, b)
5108 )
5109 return True
5110 try:
5111 same = bool(a == b)
5112 except (TypeError, ValueError):
5113 return False
5114 return same and (not isinstance(a, float) or a != 0.0 or _same_sign(a, b))
5117def _same_sign(a: float, b: float) -> bool:
5118 return math.copysign(1.0, a) == math.copysign(1.0, b)
5121def _frame_delta(prev, cur):
5122 """``cur`` as a change to ``prev`` — the delta node `_REPLAY_FRAMES_JS` applies."""
5123 if isinstance(cur, dict) and isinstance(prev, dict):
5124 changed = {}
5125 for key, value in cur.items():
5126 node = _frame_delta(prev.get(key, _ABSENT), value)
5127 if node is not None:
5128 changed[key] = node
5129 dropped = [key for key in prev if key not in cur]
5130 if dropped:
5131 return [1, changed, dropped]
5132 return [1, changed] if changed else None
5133 if prev is not _ABSENT and _same_value(prev, cur):
5134 return None
5135 if isinstance(cur, list) and isinstance(prev, list) and len(cur) == len(prev):
5136 slots = [i for i, (a, b) in enumerate(zip(prev, cur)) if not _same_value(a, b)]
5137 # A patch costs an index per slot; past half the array, restate it.
5138 if 2 * len(slots) < len(cur):
5139 return [2, slots, [cur[i] for i in slots]]
5140 return [0, cur]
5143def pack_replay_frames(frames: Sequence[Mapping]) -> list[dict]:
5144 """Delta-encode a replay's frames for `_REPLAY_FRAMES_JS` (PERF-17).
5146 ``frames`` are the figure's ``to_dict()["frames"]`` (or the same as plain
5147 JSON values). Each packed frame keeps its own keys (``name``, ``traces``, …)
5148 and replaces ``data`` with ``p``: one delta node per trace against that
5149 trace's state in the frame before — the trace index is ``traces[i]``, or
5150 ``i`` without it. Frame 0 has no frame before it, so it is carried whole.
5151 Values are carried as they are, so serializing the result the way
5152 ``to_html`` serializes frames (``to_json_plotly``) writes each exactly as
5153 the frames would have.
5154 """
5155 state: dict = {}
5156 packed = []
5157 for frame in frames:
5158 data = frame.get("data")
5159 # A frame whose `data` is null keeps it as it was.
5160 out = {k: v for k, v in frame.items() if k != "data" or data is None}
5161 if data is not None:
5162 traces = frame.get("traces")
5163 nodes = []
5164 for i, trace in enumerate(data):
5165 slot = traces[i] if traces is not None else i
5166 nodes.append(_frame_delta(state.get(slot, _ABSENT), trace))
5167 state[slot] = trace
5168 out["p"] = nodes
5169 packed.append(out)
5170 return packed
5173def replay_page(fig) -> tuple[dict, str] | None:
5174 """A replay as its HTML page carries it: ``(figure dict, post_script)``.
5176 PERF-17: the dict is the figure without its ``frames``, and the script
5177 rebuilds them in the browser from :func:`pack_replay_frames`' deltas, then
5178 runs :func:`animation_player_post_script`'s player — a 2,001-frame replay's
5179 page falls from 66 MB to under 1 MB. Serialize the dict with
5180 ``to_html(…, validate=False, post_script=script)``; ``auto_play`` no longer
5181 matters, since plotly.py sees no frames. ``fig`` may be a figure or its
5182 ``to_dict()``, which is not modified. ``None`` for a figure with no player (a
5183 static figure, or a replay with no frames): serialize that one as it is.
5185 The figure itself keeps its frames: `api.animate_scanpath`, the GIF/MP4
5186 export and ``fig.show()`` use them as they are.
5187 """
5188 player = animation_player_post_script(fig)
5189 if player is None:
5190 return None
5191 if not isinstance(fig, dict) and not fig.frames:
5192 return None
5193 fig_dict = fig if isinstance(fig, dict) else fig.to_dict()
5194 frames = fig_dict.get("frames") or []
5195 if not frames:
5196 return None
5197 if not all(_packable(frame) for frame in frames):
5198 # Frames this encoding can't address (a typed-array `traces`) travel as
5199 # plotly.py writes them, still on the player.
5200 return fig_dict, player
5201 page = {key: value for key, value in fig_dict.items() if key != "frames"}
5202 return page, _packed_frames_script(frames) + "\n" + player
5205def _packable(frame: Mapping) -> bool:
5206 """Whether `pack_replay_frames` can address ``frame``'s traces by index."""
5207 data = frame.get("data")
5208 traces = frame.get("traces")
5209 if data is not None and not isinstance(data, (list, tuple)):
5210 return False
5211 return traces is None or (
5212 isinstance(traces, (list, tuple))
5213 and all(isinstance(i, int) and not isinstance(i, bool) for i in traces)
5214 )
5217def _packed_frames_script(frames: Sequence[Mapping]) -> str:
5218 """`_REPLAY_FRAMES_JS` carrying ``frames``, packed (PERF-17)."""
5219 from plotly.io.json import to_json_plotly
5221 packed = to_json_plotly(pack_replay_frames(frames))
5222 # Inside a <script>: no `<` may close it, and plotly.py substitutes
5223 # `{plot_id}` across the whole script — both only ever occur inside a JSON
5224 # string, where the escapes decode to the same text.
5225 packed = packed.replace("<", "\\u003c").replace("{plot_id}", "\\u007bplot_id}")
5226 return _REPLAY_FRAMES_JS.replace(_PACKED_FRAMES_TOKEN, packed)
5229def animation_clip_frame_ms(fig) -> float | None:
5230 """How long a GIF/MP4 of ``fig`` holds each frame to last as long as its replay.
5232 The replay takes ``reading span / playback_speed`` — its last frame's
5233 reading time over the speed stamped on ``layout.meta`` — and a clip spreads
5234 that evenly over the frames (BUG-93). ``None`` for a figure
5235 :func:`make_scanpath_animation` didn't build."""
5236 meta = getattr(fig.layout, "meta", None)
5237 if not isinstance(meta, dict):
5238 return None
5239 times = meta.get(_REPLAY_META_TIMES)
5240 speed = meta.get(_REPLAY_META_SPEED)
5241 if not times or not speed or speed <= 0:
5242 return None
5243 return float(times[-1]) / float(speed) / len(times)
5246def _replay_clock_meta(frame_times, playback_speed: float, autoplay: bool) -> dict:
5247 """The replay's ``layout.meta``: the clock the player reads, and autoplay."""
5248 return {
5249 _AUTOPLAY_META_FLAG: bool(autoplay and frame_times),
5250 _REPLAY_META_TIMES: frame_times,
5251 _REPLAY_META_SPEED: float(playback_speed),
5252 }
5255def set_replay_clock(
5256 fig: go.Figure, frame_step_ms: float, *, playback_speed: float, autoplay: bool
5257) -> None:
5258 """Re-time a replay in place: a new playback speed and autoplay, same frames.
5260 PERF-15: the frames depend on neither (BUG-93), only ▶ Play's own frame
5261 duration and the clock on ``layout.meta`` do, so the app builds a replay once
5262 and stamps these onto the copy each cache hit returns. ``frame_step_ms`` is
5263 the exact grid step :func:`build_scanpath_replay` returned with the figure —
5264 the rounded frame times on ``layout.meta`` could truncate Play's duration to
5265 a different whole millisecond. The result is byte-identical to building the
5266 replay at that speed and autoplay. Only the clock's keys change: anything
5267 else on ``layout.meta`` (the Illustration label's) stays where it is.
5268 """
5269 meta = fig.layout.meta if isinstance(fig.layout.meta, dict) else {}
5270 times = list(meta.get(_REPLAY_META_TIMES) or [])
5271 if times:
5272 fig.layout.updatemenus = _animation_play_buttons(
5273 _anim_frame_duration_ms(frame_step_ms, playback_speed)
5274 )
5275 fig.layout.meta = {**meta, **_replay_clock_meta(times, playback_speed, autoplay)}
5278#: The replay's transport controls are app chrome, drawn in the app's font.
5279_REPLAY_UI_FONT = APP_THEME["font"]
5282def _animation_play_buttons(frame_duration):
5283 """Play / Pause / Restart buttons.
5285 Each carries a ``name`` the replay player looks for: on every HTML surface
5286 :func:`animation_player_post_script` takes ▶ Play over and runs the frames on
5287 the wall clock (BUG-93), so Play's own ``frame_duration`` drives only a
5288 figure shown without it (``fig.show()``).
5290 Frames step with ``redraw=False``: every animated trace is full length with
5291 not-yet-reached fixations masked to ``None`` (see :func:`_revealed_xy`), so
5292 advancing a frame only changes point positions — Plotly updates just those
5293 few traces instead of redrawing the whole figure (the static word boxes +
5294 labels) every frame. A full redraw of the scanpath figure costs ~50 ms, which
5295 on a long trial dwarfed the per-frame budget. Transitions are 0 so frames snap
5296 into place (no tweening), and the constant array length means a new
5297 fixation/number appears on its mark instead of gliding in from the corner.
5298 """
5299 return [
5300 dict(
5301 type="buttons",
5302 showactive=False,
5303 # A horizontal row above the plot. The scrubber shares this top
5304 # transport band to keep playback controls together.
5305 direction="right",
5306 y=1.0,
5307 x=0.0,
5308 xanchor="left",
5309 yanchor="bottom",
5310 pad=dict(b=12, l=8),
5311 # #374 F23: the app's font, not the figure's (often a monospace
5312 # stimulus font), so the buttons read as the app's own.
5313 font=dict(family=_REPLAY_UI_FONT),
5314 buttons=[
5315 dict(
5316 label="▶ Play",
5317 name="play",
5318 method="animate",
5319 args=[
5320 None,
5321 dict(
5322 frame=dict(duration=frame_duration, redraw=False),
5323 fromcurrent=True,
5324 transition=dict(duration=0),
5325 ),
5326 ],
5327 ),
5328 dict(
5329 label="⏸ Pause",
5330 name="pause",
5331 method="animate",
5332 args=[
5333 [None],
5334 dict(
5335 frame=dict(duration=0, redraw=False),
5336 mode="immediate",
5337 transition=dict(duration=0),
5338 ),
5339 ],
5340 ),
5341 dict(
5342 label="⟲ Restart",
5343 name="restart",
5344 method="animate",
5345 args=[
5346 ["0"],
5347 dict(
5348 frame=dict(duration=0, redraw=True),
5349 mode="immediate",
5350 transition=dict(duration=0),
5351 ),
5352 ],
5353 ),
5354 ],
5355 )
5356 ]
5359def _animation_time_slider(frame_times, total_ms):
5360 """Linear time-scrubber slider (VIZ-11).
5362 Frame times sit on a uniform grid, so the handle moves linearly through
5363 reading time. Each step's label is **"elapsed / total s"** (e.g. "1.2 /
5364 30.0s"), surfaced in the single ``currentvalue`` readout — meaningful for any
5365 number of overlaid scanpaths, unlike a fixation index. A long reading would
5366 render a wall of overlapping numbers if every step drew a tick + label, so the
5367 per-step tick ruler (``ticklen``/``minorticklen`` = 0) and per-step labels
5368 (transparent ``font``) are hidden; the readout is the one time display.
5369 """
5370 total_s = total_ms / 1000.0
5371 return [
5372 dict(
5373 active=0,
5374 # Share the top transport band with Play / Pause / Restart.
5375 yanchor="bottom",
5376 xanchor="left",
5377 ticklen=0,
5378 minorticklen=0,
5379 # Per-step labels feed the readout but must not pile up under the
5380 # track, so draw them fully transparent.
5381 font=dict(color="rgba(0,0,0,0)"),
5382 currentvalue=dict(
5383 font=dict(size=14, color="#444", family=_REPLAY_UI_FONT),
5384 # #374 F23/F8: the trial's own clock, first fixation onward.
5385 prefix="Trial time ",
5386 visible=True,
5387 xanchor="right",
5388 ),
5389 transition=dict(duration=0),
5390 pad=dict(b=12),
5391 len=0.6,
5392 x=0.38,
5393 y=1.0,
5394 steps=[
5395 dict(
5396 args=[
5397 [str(k)],
5398 dict(
5399 frame=dict(duration=0, redraw=True),
5400 mode="immediate",
5401 transition=dict(duration=0),
5402 ),
5403 ],
5404 label=f"{frame_times[k] / 1000:.1f} / {total_s:.1f} s",
5405 method="animate",
5406 )
5407 for k in range(len(frame_times))
5408 ],
5409 )
5410 ]
5413def _render_scanpath_animation(
5414 words: pd.DataFrame,
5415 fixations: pd.DataFrame,
5416 *,
5417 settings: FigureSettings,
5418 fixations_b: pd.DataFrame | None = None,
5419 words_b: pd.DataFrame | None = None,
5420) -> tuple[go.Figure, float]:
5421 """Frame-by-frame scanpath replay on a real reading-time clock.
5423 Returns the figure and the exact grid step its frames sit on, which
5424 :func:`set_replay_clock` needs to re-time it (PERF-15).
5426 Pass ``fixations_b`` (and optionally ``words_b``) to overlay a SECOND
5427 scanpath animated on the same clock. Every scanpath is rebased to its first
5428 fixation's ``timestamp_ms``, so they share *real reading time* including the
5429 saccade/blink gaps between fixations; a frame is emitted at every fixation
5430 onset across all scanpaths, and the shorter reading finishes first and holds
5431 while the longer keeps going. The wall-clock player every HTML surface embeds
5432 (:func:`animation_player_post_script`) shows each frame once its reading time
5433 over ``playback_speed`` has elapsed, so the whole replay takes
5434 ``reading_span / playback_speed`` — exactly what
5435 :func:`animation_playback_ms` reports (and the side panel quotes).
5437 With two scanpaths each trail wears its own style — ``style_a`` /
5438 ``style_b``, resolved exactly as :func:`make_comparison_figure` resolves
5439 them (colour, size range, opacity, hollow markers and the saccade line's
5440 colour, dash and width; the comparison palette where a style names none) —
5441 order numbers are tinted per-scanpath, and an optional A/B legend
5442 (``show_legend``) names them; word boxes/labels come from
5443 ``words`` (scanpath A), so the overlay is meaningful for two readings of the
5444 same text. With one scanpath the behaviour matches the classic single replay
5445 (order numbers honour ``order_font_color``, no legend).
5447 The single replay honours the same fixation-colouring options as
5448 :func:`make_scanpath_figure`: ``color_by`` (numeric → ``fixation_colorscale``
5449 pinned to the whole trial's range so colours stay stable as the trail grows,
5450 categorical → discrete palette + legend), ``color_by_line``, and an optional
5451 colorbar (styled by the ``fixation_colorbar_*`` settings, like the static
5452 figure). The dual overlay
5453 colours as :func:`make_comparison_figure` does: the metric (on one range
5454 shared by both readings) or one shared category→colour mapping fills the
5455 markers, and each reading's flat A/B colour becomes its marker outline.
5457 VIZ-23 brought the remaining word-label, arrow and flag options across from
5458 the static figure, each defaulting to the replay's previous behaviour:
5460 - the word labels take ``text_color`` / ``highlight_column`` /
5461 ``highlight_text_color`` / ``word_hover_measure`` (``highlight_column`` is
5462 the *text*-marking channel — there is no border-overlay style here);
5463 - ``show_saccade_arrows`` adds the direction arrowheads, each revealed with
5464 the saccade it belongs to rather than all at frame zero;
5465 - ``fixation_flags`` applies the PRE-2 short/long/out-of-bounds
5466 classification: *Discard* drops those fixations from the replay entirely,
5467 *Highlight* overlays them in their flag marker as the replay reaches them.
5469 ``layout.meta`` carries the replay's clock (each frame's reading time and the
5470 speed) and, with ``autoplay`` (default on, VIZ-10), the intent to start on
5471 load *at the configured playback speed*; the player reads both. The figure
5472 itself is always built paused — autoplay is the embedder's to start.
5473 """
5474 canvas_width = settings.canvas_width
5475 canvas_height = settings.canvas_height
5476 base_font_size = settings.base_font_size
5477 font_family = settings.font_family
5478 playback_speed = settings.playback_speed
5479 show_words = settings.show_words
5480 show_word_labels = settings.show_word_labels
5481 show_saccades = settings.show_saccades
5482 show_saccade_arrows = settings.show_saccade_arrows
5483 show_order = settings.show_order
5484 marker_size_range = settings.marker_size_range
5485 order_font_size = settings.order_font_size
5486 order_font_color = settings.order_font_color
5487 color_by = settings.color_by
5488 # BUG-85: as in `make_scanpath_figure` — "line" is colour-by-line.
5489 color_by_line = settings.color_by_line or color_by == "line"
5490 fixation_colorscale = settings.fixation_colorscale
5491 fixation_color_range = settings.fixation_color_range
5492 fixation_flags = settings.fixation_flags
5493 fixation_flags_b = settings.fixation_flags_b
5494 # The replay has no heatmap: its one bar is the fixations'.
5495 show_colorbars = settings.show_fixation_colorbar
5496 colorbar_orientation = settings.fixation_colorbar_orientation
5497 colorbar_tickangle = settings.fixation_colorbar_tickangle
5498 colorbar_tickfont_size = settings.fixation_colorbar_tickfont_size
5499 saccade_color = settings.saccade_color
5500 saccade_style = settings.saccade_style
5501 saccade_width = settings.saccade_width
5502 hollow_fixations = settings.hollow_fixations
5503 fixation_opacity = settings.fixation_opacity
5504 fixation_color = settings.fixation_color
5505 fixation_symbol = settings.fixation_symbol
5506 text_color = settings.text_color
5507 highlight_column = settings.highlight_column
5508 highlight_text_color = settings.highlight_text_color
5509 word_hover_measure = settings.word_hover_measure
5510 word_hover_fields = settings.word_hover_fields
5511 fixation_hover_fields = settings.fixation_hover_fields
5512 background_color = settings.background_color
5513 # Drawn as written: a label is a name, not Plotly markup (round 9).
5514 label_a = _plotly_literal(settings.label_a)
5515 label_b = _plotly_literal(settings.label_b)
5516 show_legend = settings.show_legend
5517 line_spacing = settings.line_spacing
5518 scale_text_to_boxes = settings.scale_text_to_boxes
5519 background_image = settings.background_image
5520 background_image_size = settings.background_image_size
5521 background_image_origin = settings.background_image_origin
5522 background_image_opacity = settings.background_image_opacity
5523 fit_to_monitor = settings.fit_to_monitor
5524 show_coordinate_grid = settings.show_coordinate_grid
5525 coordinate_grid_spacing = settings.coordinate_grid_spacing
5526 autoplay = settings.autoplay
5527 anim_grid_step_ms = settings.anim_grid_step_ms
5528 anim_max_frames = settings.anim_max_frames
5529 fig = go.Figure()
5530 font_settings = dict(family=font_family or FONT_FAMILY, size=base_font_size)
5532 word_frames = [w for w in (words, words_b) if w is not None and not w.empty]
5533 x_range, y_range, *_ = _compute_axis_ranges(
5534 canvas_width,
5535 canvas_height,
5536 (fixations, "x", "y"),
5537 (fixations_b, "x", "y"),
5538 word_frames=word_frames,
5539 fit_to_monitor=fit_to_monitor,
5540 )
5542 # Fix the display size first so word labels are sized in the data->screen
5543 # scale (true-to-scale text); the same fitted_w/fitted_h drive the layout.
5544 fitted_w, fitted_h = _fit_display_size(
5545 canvas_width, canvas_height, x_range, y_range, spatial_axes=True
5546 )
5547 scale = _display_scale(x_range, y_range, fitted_w, fitted_h)
5548 label_font_px = _word_label_font_px(
5549 words,
5550 scale=scale,
5551 line_spacing=line_spacing,
5552 manual_font_px=base_font_size,
5553 scale_text_to_boxes=scale_text_to_boxes,
5554 )
5556 # CMP-11: which reading's stimulus the replay draws. The replay has only ever
5557 # had ONE stimulus layer, so "both" keeps meaning A's here rather than
5558 # stacking a second identical set of rectangles onto every existing
5559 # same-dataset co-animation. "b" is the one that matters: a cross-dataset
5560 # co-animation would otherwise run B's trace over A's text.
5561 stimulus_words = words
5562 if (
5563 _compare_stimulus_sides(settings.compare_stimulus) == (False, True)
5564 and words_b is not None
5565 and not words_b.empty
5566 ):
5567 stimulus_words = words_b
5568 shapes = (
5569 build_word_boxes(
5570 stimulus_words,
5571 color=settings.word_box_color,
5572 fill_color=settings.word_box_fill_color,
5573 fill_opacity=settings.word_box_fill_opacity,
5574 line_opacity=settings.word_box_line_opacity,
5575 )
5576 if show_words and not stimulus_words.empty
5577 else []
5578 )
5579 if show_word_labels and not stimulus_words.empty:
5580 _add_word_label_trace(
5581 fig,
5582 stimulus_words,
5583 label_font_px,
5584 font_settings["family"],
5585 highlight_column=highlight_column,
5586 text_color=text_color,
5587 highlight_text_color=highlight_text_color,
5588 word_hover_measure=word_hover_measure,
5589 word_hover_fields=word_hover_fields,
5590 )
5592 # PRE-2 fixation flags (VIZ-23), mirroring the static figure: *Discard* drops
5593 # the flagged rows before the specs are built, so they vanish from the trail,
5594 # the saccade polyline, the index labels and the marker-size scaling. Applied
5595 # AFTER the axis ranges above, exactly as in `make_scanpath_figure` — the view
5596 # is framed on the trial as recorded, not on what survives the filter.
5597 flags = fixation_flags or {}
5598 # B is flagged against its own word boxes when it brought them, else against
5599 # A's (the overlay draws A's boxes, and two readings of one text share them).
5600 entry_words = [
5601 words,
5602 words_b if (words_b is not None and not words_b.empty) else words,
5603 ]
5604 # CMP-24: B carries its own flags when it was given any — as
5605 # `fixation_flags_b`, or (the comparison's spelling) on its `style_b`.
5606 if fixation_flags_b is None and isinstance(settings.style_b, dict):
5607 fixation_flags_b = settings.style_b.get("fixation_flags")
5608 flags_b = flags if fixation_flags_b is None else (fixation_flags_b or {})
5609 entry_flags = [flags, flags_b]
5610 if flags:
5611 fixations = _discard_flagged_fixations(fixations, entry_words[0], flags)
5612 if flags_b and fixations_b is not None:
5613 fixations_b = _discard_flagged_fixations(fixations_b, entry_words[1], flags_b)
5615 # The co-animation draws each scanpath in its own style, exactly as the
5616 # static comparison resolves it (`_comparison_scanpath_style`: the rail's
5617 # per-scanpath colour, size range, opacity, hollow and saccade line). A lone
5618 # scanpath keeps the figure-wide settings below.
5619 dual_input = all(f is not None and not f.empty for f in (fixations, fixations_b))
5620 styles = [
5621 _comparison_scanpath_style(
5622 idx, style, default_marker_size_range=marker_size_range
5623 )
5624 for idx, style in enumerate((settings.style_a, settings.style_b))
5625 ]
5626 entries = [
5627 (frame, style["fix_color"], label)
5628 for frame, style, label in zip(
5629 (fixations, fixations_b), styles, (label_a, label_b)
5630 )
5631 ]
5632 specs = _scanpath_anim_specs(
5633 entries,
5634 marker_size_range,
5635 size_ranges=[
5636 style["marker_size_range"] if dual_input else marker_size_range
5637 for style in styles
5638 ],
5639 **_settings_size_scale(settings),
5640 )
5641 # The words frame each surviving scanpath is flagged against (the highlight
5642 # overlay's out-of-bounds test). `_scanpath_anim_specs` skips empty
5643 # scanpaths, so apply the same skip rule here to stay aligned with `specs`.
5644 surviving = [
5645 (w, f, style)
5646 for (fix_df, _color, _label), w, f, style in zip(
5647 entries, entry_words, entry_flags, styles
5648 )
5649 if fix_df is not None and not fix_df.empty
5650 ]
5651 for spec, (spec_words, spec_flags, spec_style) in zip(specs, surviving):
5652 spec["words"] = spec_words
5653 spec["flags"] = spec_flags
5654 spec["style"] = spec_style
5655 dual = len(specs) > 1
5656 if not dual and specs:
5657 # A lone scanpath always wears the canonical single-replay colour,
5658 # whether it arrived as `fixations` or (degenerately) only as
5659 # `fixations_b`, so the trail never silently renders in the B colour.
5660 # VIZ-17/18: honour the caller's uniform fixation colour when one is
5661 # given, so the replay matches the static figure (and the palette).
5662 specs[0]["color"] = fixation_color or COMPARISON_PALETTE[0]
5664 # Metric colouring, mirroring the static figure's fixation trace (the dual
5665 # overlay's, mirroring the comparison figure's, comes first). Numeric
5666 # metrics map through
5667 # `fixation_colorscale` with cmin/cmax pinned to the WHOLE trial (or the
5668 # caller's range) up front — otherwise the scale would renormalise to the
5669 # partial trail on every frame and colours would drift during playback.
5670 for s in specs:
5671 s["marker_colors"] = None
5672 s["marker_extra"] = {}
5673 s["marker_line"] = None
5674 category_legend: list = []
5675 color_label = color_by or ""
5676 # VIZ-17: the uniform sentinel means "no variable mapped to hue" — leave the
5677 # trail on its flat colour rather than looking for a column by that name.
5678 if color_by == UNIFORM_COLOR_FIELD:
5679 color_by, color_label = None, ""
5680 if dual and (color_by or color_by_line):
5681 # The co-animation colours as the static comparison does: the metric
5682 # (one shared range) or the shared category colours fill each marker,
5683 # and each scanpath's own colour becomes its outline — the A/B cue.
5684 labels = [
5685 _fixation_category_labels(s["ordered"], s["words"], color_by, color_by_line)
5686 for s in specs
5687 ]
5688 colors, category_legend = _shared_category_colors(
5689 labels, avoid=[s["color"] for s in specs]
5690 )
5691 if category_legend:
5692 color_label = "line" if (color_by_line or color_by == "line") else color_by
5693 for s, spec_colors in zip(specs, colors):
5694 if spec_colors is not None:
5695 s["marker_colors"] = spec_colors
5696 s["marker_line"] = dict(color=s["color"], width=1.4)
5697 elif color_by and all(
5698 color_by in s["ordered"].columns
5699 and pd.api.types.is_numeric_dtype(s["ordered"][color_by])
5700 for s in specs
5701 ):
5702 values = pd.concat([s["ordered"][color_by] for s in specs])
5703 if values.notna().any():
5704 rng = fixation_color_range or (
5705 float(values.min()),
5706 float(values.max()),
5707 )
5708 for i, s in enumerate(specs):
5709 # One bar, on A's trail — the scale is shared.
5710 bar = show_colorbars and i == 0
5711 s["marker_colors"] = list(s["ordered"][color_by])
5712 s["marker_line"] = dict(color=s["color"], width=1.4)
5713 s["marker_extra"] = dict(
5714 colorscale=fixation_colorscale,
5715 cmin=rng[0],
5716 cmax=rng[1],
5717 showscale=bar,
5718 colorbar=_colorbar_dict(
5719 _column_title(color_label),
5720 orientation=colorbar_orientation,
5721 tickangle=colorbar_tickangle,
5722 tickfont_size=colorbar_tickfont_size,
5723 )
5724 if bar
5725 else None,
5726 )
5727 if not dual and specs and (color_by or color_by_line):
5728 ordered0 = specs[0]["ordered"]
5729 if color_by_line and not words.empty:
5730 from .measures import assign_fixation_lines
5732 line_ids = assign_fixation_lines(ordered0, words)
5733 color_data = line_ids.map(
5734 lambda v: f"Line {int(v) + 1}" if pd.notna(v) else "Out of bounds"
5735 )
5736 color_label = "line"
5737 is_numeric_color = False
5738 else:
5739 color_data = ordered0[color_by] if color_by in ordered0.columns else None
5740 is_numeric_color = color_data is not None and pd.api.types.is_numeric_dtype(
5741 color_data
5742 )
5743 if color_data is not None:
5744 marker_color, category_legend = _resolve_marker_colors(
5745 color_data, is_numeric_color
5746 )
5747 specs[0]["marker_colors"] = list(marker_color)
5748 if is_numeric_color:
5749 rng = fixation_color_range or (
5750 float(color_data.min()),
5751 float(color_data.max()),
5752 )
5753 # VIZ-23: the same styled colorbar the static figure builds, so
5754 # orientation / tick angle / tick size apply here too.
5755 colorbar = None
5756 if show_colorbars:
5757 colorbar = _colorbar_dict(
5758 _column_title(color_label),
5759 orientation=colorbar_orientation,
5760 tickangle=colorbar_tickangle,
5761 tickfont_size=colorbar_tickfont_size,
5762 )
5763 specs[0]["marker_extra"] = dict(
5764 colorscale=fixation_colorscale,
5765 cmin=rng[0],
5766 cmax=rng[1],
5767 showscale=show_colorbars,
5768 colorbar=colorbar,
5769 )
5771 def _trail_marker(s):
5772 """Marker dict for a (full-length) trail trace.
5774 Every animated trace is full length with not-yet-reached fixations masked
5775 to ``None`` positions (see :func:`_revealed_xy`), so the size/colour
5776 arrays are stated once at full length and never change frame to frame —
5777 only which positions are revealed does. Restating the whole marker keeps
5778 the colorscale/cmin/cmax/colorbar attached to the trail."""
5779 colors = s["marker_colors"]
5780 marker = dict(
5781 size=list(s["sizes"]),
5782 # A glyph shape (♥) has no Plotly symbol: `_trail_traces` draws this
5783 # dict as text instead, so the symbol here only needs to be valid.
5784 symbol=_marker_symbol(fixation_symbol),
5785 color=colors if colors is not None else s["color"],
5786 line=s["marker_line"] or dict(color=FIX_MARKER_OUTLINE, width=0.5),
5787 **s["marker_extra"],
5788 )
5789 # Always set the alpha (even 1.0) so the control overrides Plotly's ~0.7
5790 # default for variable-size scatter markers (VIZ-6).
5791 marker["opacity"] = float(s["opacity"] if s["opacity"] is not None else 1.0)
5792 if s["hollow"]:
5793 marker = _make_hollow(marker)
5794 return marker
5796 glyph = FIXATION_GLYPH_SYMBOLS.get(fixation_symbol or "")
5798 def _trail_style(s) -> tuple[dict, list[dict] | None]:
5799 """The trail's marker dict and, for a glyph shape, its text layers —
5800 stated once per scanpath and reused by every frame, which only moves
5801 positions. Building them per frame re-sampled the colorscale (glyph and
5802 hollow markers) for every fixation on each of ~360 frames."""
5803 if "trail_style" not in s:
5804 marker = _trail_marker(s)
5805 layers = _glyph_layers(marker, glyph, s["n_total"]) if glyph else None
5806 s["trail_style"] = (marker, layers)
5807 return s["trail_style"]
5809 def _trail_traces(s, x, y, *, make: Callable = go.Scatter, **top) -> list:
5810 """The trail as drawn at positions ``x``/``y`` — one marker trace, or
5811 for a glyph shape (♥) its text layers (`_glyph_scatter_traces`), which
5812 un-mask exactly like the markers do. Full length either way, so the
5813 frames still only move positions. ``make=dict`` for a frame."""
5814 marker, layers = _trail_style(s)
5815 if layers is not None:
5816 return _glyph_layer_traces(
5817 x, y, layers, make=make, customdata=s["customdata"], **top
5818 )
5819 return [
5820 make(
5821 x=x,
5822 y=y,
5823 mode="markers",
5824 marker=marker,
5825 text=s["order_text"],
5826 customdata=s["customdata"],
5827 **top,
5828 )
5829 ]
5831 # Base traces, with stable indices the frames update by position. Each
5832 # animated trace is built at FULL length (one slot per fixation); the replay
5833 # reveals a fixation by un-masking its x/y, never by growing the array or
5834 # rewriting `text`. Constant length + position-only changes are what let the
5835 # Play button animate with `redraw=False` (see `_animation_play_buttons`):
5836 # Plotly then re-renders only these few traces per frame instead of redrawing
5837 # the static word boxes + labels every time — the redraw cost that made a
5838 # long replay run far slower than its quoted time. It also keeps the trail's
5839 # fixation number in `text` (hover only); the visible order numbers live in a
5840 # separate text trace (below).
5841 # Scanpath legend entries of their own (dual + legend only): see the trail.
5842 own_entries = bool(category_legend) or bool(glyph)
5843 for s in specs:
5844 ordered = s["ordered"]
5845 n_total = len(ordered)
5846 all_x = ordered["x"].tolist()
5847 all_y = ordered["y"].tolist()
5848 s["all_x"] = all_x
5849 s["all_y"] = all_y
5850 s["n_total"] = n_total
5851 hover_fields = (
5852 ["order_in_trial", "duration_ms"]
5853 if fixation_hover_fields is None
5854 else list(fixation_hover_fields)
5855 )
5856 s["customdata"], s["hovertemplate"] = _hover_payload(
5857 ordered, hover_fields, fixation=True, words=s.get("words")
5858 )
5859 # The trial's own fixation numbers, as the static figure and the hover
5860 # show them — never a 1..n renumbering of what survived the filters.
5861 s["order_text"] = _fixation_order_labels(ordered)
5862 s["text_color"] = s["color"] if dual else order_font_color
5863 # The co-animation draws each scanpath's saccades, opacity and hollow
5864 # markers from its own style, as the static comparison does; a lone
5865 # replay keeps the figure-wide settings.
5866 style = s["style"]
5867 s["sac_color"] = style["saccade_color"] if dual else saccade_color
5868 s["sac_width"] = style["saccade_width"] if dual else saccade_width
5869 s["sac_dash"] = style["saccade_style"] if dual else saccade_style
5870 s["opacity"] = style["opacity"] if dual else fixation_opacity
5871 s["hollow"] = bool(style["hollow"]) if dual else hollow_fixations
5872 s["curr_outline"] = s["color"] if dual else CURRENT_FIX_OUTLINE
5873 s["curr_outline_w"] = 2.5 if dual else 2
5875 base_x, base_y = _revealed_xy(all_x, all_y, 1)
5876 trail = _trail_traces(
5877 s,
5878 base_x,
5879 base_y,
5880 # A/B legend on the dual overlay only — off by default, honours the
5881 # compare-legend toggle (CMP-2). The single-replay colour-by legend
5882 # below is separate and unaffected. Under shared category colours
5883 # the swatch would show a category, and a glyph has no swatch, so a
5884 # separate entry (below) names the scanpath instead.
5885 showlegend=dual and show_legend and not own_entries,
5886 name=s["label"],
5887 legendgroup=s["label"],
5888 hovertemplate=(s["label"] + "<br>" if dual else "") + s["hovertemplate"],
5889 )
5890 s["idx_trails"] = list(range(len(fig.data), len(fig.data) + len(trail)))
5891 for trace in trail:
5892 fig.add_trace(trace)
5893 # Order numbers: a text trace holding EVERY fixation's final position,
5894 # with not-yet-reached fixations masked to None x/y (so nothing is drawn
5895 # there). A number snaps on at its fixation when that position un-masks —
5896 # no gliding in from the (0,0) corner — and because the `text` strings
5897 # never change frame to frame, `redraw=False` renders the reveal purely
5898 # from the position change.
5899 if show_order:
5900 s["idx_order"] = len(fig.data)
5901 fig.add_trace(
5902 go.Scatter(
5903 x=base_x,
5904 y=base_y,
5905 mode="text",
5906 text=s["order_text"],
5907 textfont=dict(
5908 color=s["text_color"],
5909 size=order_font_size,
5910 family=font_settings["family"],
5911 ),
5912 textposition="top center",
5913 showlegend=False,
5914 legendgroup=s["label"],
5915 hoverinfo="skip",
5916 )
5917 )
5918 else:
5919 s["idx_order"] = None
5920 if show_saccades:
5921 sac_x, sac_y = _revealed_saccade_xy(all_x, all_y, 1)
5922 s["idx_sac"] = len(fig.data)
5923 fig.add_trace(
5924 go.Scatter(
5925 x=sac_x,
5926 y=sac_y,
5927 mode="lines",
5928 line=dict(
5929 color=s["sac_color"], width=s["sac_width"], dash=s["sac_dash"]
5930 ),
5931 showlegend=False,
5932 legendgroup=s["label"],
5933 hoverinfo="skip",
5934 )
5935 )
5936 else:
5937 s["idx_sac"] = None
5938 # Saccade direction arrowheads (VIZ-23). Same marker as the static
5939 # figure's, but each arrow is revealed with its own saccade (its position
5940 # un-masks when `_revealed_saccade_xy` draws that segment) rather than the
5941 # whole set standing there from frame zero. Angles are stated once at full
5942 # length and never change, so `redraw=False` still applies.
5943 arrow_x, arrow_y, arrow_angle, arrow_seg = (
5944 _saccade_arrow_rows(ordered, "x", "y")
5945 if (show_saccades and show_saccade_arrows)
5946 else ([], [], [], [])
5947 )
5948 s["arrow_x"], s["arrow_y"], s["arrow_seg"] = arrow_x, arrow_y, arrow_seg
5949 s["idx_arrow"] = None
5950 if arrow_x:
5951 ax0, ay0 = _revealed_arrow_xy(arrow_x, arrow_y, arrow_seg, 1)
5952 s["idx_arrow"] = len(fig.data)
5953 fig.add_trace(
5954 go.Scatter(
5955 x=ax0,
5956 y=ay0,
5957 mode="markers",
5958 marker=dict(
5959 symbol="arrow",
5960 size=12,
5961 angle=arrow_angle,
5962 angleref="up",
5963 color=s["sac_color"],
5964 line=dict(width=0),
5965 ),
5966 showlegend=False,
5967 legendgroup=s["label"],
5968 hoverinfo="skip",
5969 name="saccade direction",
5970 )
5971 )
5972 s["idx_curr"] = len(fig.data)
5973 fig.add_trace(
5974 go.Scatter(
5975 x=[all_x[0]],
5976 y=[all_y[0]],
5977 mode="markers",
5978 marker=dict(
5979 size=[float(s["sizes"][0]) + 8],
5980 color=CURRENT_FIX_COLOR,
5981 line=dict(color=s["curr_outline"], width=s["curr_outline_w"]),
5982 ),
5983 showlegend=False,
5984 legendgroup=s["label"],
5985 hoverinfo="skip",
5986 )
5987 )
5988 # PRE-2 *Highlight* overlays (VIZ-23): one trace per flagged category, in
5989 # that category's marker + colour, drawn over the trail. Full-length like
5990 # every animated trace — fixations that aren't flagged are masked out
5991 # permanently, the rest un-mask as the replay reaches them.
5992 s["flag_overlays"] = []
5993 s_flags = s.get("flags", flags)
5994 if s_flags:
5995 overlay_masks = _fixation_flag_masks(ordered, s["words"], s_flags)
5996 for category in _FIX_FLAG_CATEGORIES:
5997 spec_flags = s_flags.get(category, {})
5998 if spec_flags.get("mode") != "Highlight":
5999 continue
6000 hit = overlay_masks[category].to_numpy()
6001 if not hit.any():
6002 continue
6003 hx = [all_x[j] if hit[j] else None for j in range(n_total)]
6004 hy = [all_y[j] if hit[j] else None for j in range(n_total)]
6005 label = _FIX_FLAG_LABELS[category]
6006 name = f"{s['label']} · {label}" if dual else label
6007 fx0, fy0 = _revealed_xy(hx, hy, 1)
6008 s["flag_overlays"].append(
6009 dict(
6010 idx=len(fig.data),
6011 x=hx,
6012 y=hy,
6013 marker=dict(
6014 symbol=spec_flags.get("symbol") or "x",
6015 size=13,
6016 color=spec_flags.get("color") or OUT_OF_TEXT_COLOR,
6017 line=dict(color="#ffffff", width=1),
6018 ),
6019 name=name,
6020 )
6021 )
6022 fig.add_trace(
6023 go.Scatter(
6024 x=fx0,
6025 y=fy0,
6026 mode="markers",
6027 marker=s["flag_overlays"][-1]["marker"],
6028 name=name,
6029 legendgroup=s["label"],
6030 showlegend=True,
6031 hovertemplate=(
6032 f"{label} fixation<br>x %{{x:.0f}}, y %{{y:.0f}}"
6033 "<extra></extra>"
6034 ),
6035 )
6036 )
6038 # Categorical colour legend, as in the static figure. These dummy traces sit
6039 # AFTER the per-scanpath traces so the frame indices recorded above stay
6040 # valid; frames never touch them.
6041 if dual and show_legend and own_entries:
6042 for s in specs:
6043 coloured = s["marker_line"] is not None
6044 fig.add_trace(
6045 go.Scatter(
6046 x=[None],
6047 y=[None],
6048 mode="markers",
6049 marker=dict(
6050 size=10,
6051 symbol=_marker_symbol(fixation_symbol),
6052 color="#ffffff" if coloured else s["color"],
6053 line=dict(
6054 color=s["color"] if coloured else FIX_MARKER_OUTLINE,
6055 width=2 if coloured else 0.5,
6056 ),
6057 ),
6058 name=s["label"],
6059 legendgroup=s["label"],
6060 showlegend=True,
6061 hoverinfo="skip",
6062 )
6063 )
6064 _add_category_legend(fig, category_legend, color_label)
6065 if glyph and specs:
6066 # A glyph carries no colorscale, so A's numeric colour bar rides on a
6067 # trace of its own (after the animated ones, like the legend entries).
6068 bar = _glyph_colorbar_trace(
6069 _trail_style(specs[0])[0], specs[0]["marker_colors"] or ()
6070 )
6071 if bar is not None:
6072 fig.add_trace(bar)
6074 frame_times, frame_step_ms, reading_span_ms = _anim_timeline(
6075 specs,
6076 grid_step_ms=anim_grid_step_ms,
6077 max_frames=anim_max_frames,
6078 )
6080 # Frames are plain dicts, validated once, by the `fig.frames` assignment
6081 # below. Built from `go.Scatter`/`go.Frame` they were validated (and deep-
6082 # copied) three times over — each trace, each frame, then the figure —
6083 # which on a long replay is most of the build: every frame restates the
6084 # trail's full-length sizes and colours (see `_trail_marker`).
6085 frames = []
6086 n_frames = len(frame_times)
6087 for k, t in enumerate(frame_times):
6088 # UX-169: the card's "120 of 361 frames" — and a cancel checkpoint, so an
6089 # abandoned build stops within a frame. A no-op outside a card.
6090 progress.report(k + 1, n_frames, unit="frames")
6091 traces_in_frame = []
6092 traces_idx_in_frame = []
6093 for s in specs:
6094 all_x = s["all_x"]
6095 all_y = s["all_y"]
6096 # Fixations whose recorded onset has been reached by time t.
6097 kk = max(int(np.searchsorted(s["onsets"], t, side="right")), 1)
6099 # Trail: full-length, fixations past kk masked to None. The marker
6100 # (sizes/colours) and `text` are full-length and identical every
6101 # frame, so only positions change — `redraw=False` then re-renders
6102 # just this trace, not the whole figure.
6103 tx, ty = _revealed_xy(all_x, all_y, kk)
6104 for idx, trace in zip(s["idx_trails"], _trail_traces(s, tx, ty, make=dict)):
6105 traces_in_frame.append(trace)
6106 traces_idx_in_frame.append(idx)
6108 if show_order:
6109 # Same full-length positions/text as the base order trace; the
6110 # reveal is purely the un-masking of x/y for reached fixations,
6111 # so numbers appear in place (and `redraw=False` shows them).
6112 ox, oy = _revealed_xy(all_x, all_y, kk)
6113 traces_in_frame.append(
6114 dict(
6115 x=ox,
6116 y=oy,
6117 mode="text",
6118 text=s["order_text"],
6119 textfont=dict(
6120 color=s["text_color"],
6121 size=order_font_size,
6122 family=font_settings["family"],
6123 ),
6124 textposition="top center",
6125 )
6126 )
6127 traces_idx_in_frame.append(s["idx_order"])
6129 if show_saccades:
6130 sac_x, sac_y = _revealed_saccade_xy(all_x, all_y, kk)
6131 traces_in_frame.append(
6132 dict(
6133 x=sac_x,
6134 y=sac_y,
6135 mode="lines",
6136 line=dict(
6137 color=s["sac_color"],
6138 width=s["sac_width"],
6139 dash=s["sac_dash"],
6140 ),
6141 )
6142 )
6143 traces_idx_in_frame.append(s["idx_sac"])
6145 if s["idx_arrow"] is not None:
6146 # Arrowheads reveal with the saccades above: same constant-length
6147 # array, only the mask moves (angles are set on the base trace).
6148 arw_x, arw_y = _revealed_arrow_xy(
6149 s["arrow_x"], s["arrow_y"], s["arrow_seg"], kk
6150 )
6151 traces_in_frame.append(dict(x=arw_x, y=arw_y, mode="markers"))
6152 traces_idx_in_frame.append(s["idx_arrow"])
6154 ci = kk - 1
6155 traces_in_frame.append(
6156 dict(
6157 x=[all_x[ci]],
6158 y=[all_y[ci]],
6159 mode="markers",
6160 marker=dict(
6161 size=[float(s["sizes"][ci]) + 8],
6162 color=CURRENT_FIX_COLOR,
6163 line=dict(color=s["curr_outline"], width=s["curr_outline_w"]),
6164 ),
6165 )
6166 )
6167 traces_idx_in_frame.append(s["idx_curr"])
6169 for overlay in s["flag_overlays"]:
6170 # A flagged fixation's highlight appears with the fixation itself.
6171 ox_f, oy_f = _revealed_xy(overlay["x"], overlay["y"], kk)
6172 traces_in_frame.append(
6173 dict(x=ox_f, y=oy_f, mode="markers", marker=overlay["marker"])
6174 )
6175 traces_idx_in_frame.append(overlay["idx"])
6177 frames.append(
6178 dict(data=traces_in_frame, name=str(k), traces=traces_idx_in_frame)
6179 )
6180 fig.frames = frames
6182 shapes.append(
6183 dict(
6184 type="rect",
6185 x0=x_range[0],
6186 y0=y_range[1],
6187 x1=x_range[1],
6188 y1=y_range[0],
6189 line=dict(color="#000000", width=1),
6190 fillcolor="rgba(0,0,0,0)",
6191 )
6192 )
6194 sliders = (
6195 _animation_time_slider(frame_times, reading_span_ms) if frame_times else []
6196 )
6197 updatemenus = (
6198 _animation_play_buttons(_anim_frame_duration_ms(frame_step_ms, playback_speed))
6199 if frame_times
6200 else []
6201 )
6203 # fitted_w / fitted_h were computed up front (so the label scale matched).
6204 # ALL transport controls (play/pause/restart buttons + the time slider with
6205 # its elapsed-time readout) sit ABOVE the plot in the top margin. Critically,
6206 # the figure is made tall enough that the plot region stays >= fitted_h after
6207 # Plotly's automargin reserves space for those controls — otherwise the
6208 # equal-aspect (`scaleanchor`) plot would shrink to fit the leftover height,
6209 # making the word boxes smaller than the true-to-scale label font computed
6210 # for fitted_h (text-too-large bug). _CONTROLS_MARGIN_PX is that reserve.
6211 # A single-replay numeric colorbar gets the same treatment on the right
6212 # (the dual-overlay legend overlays the plot, so it needs no reserve).
6213 # A HORIZONTAL colorbar (VIZ-23) stays below the plot, so it takes bottom
6214 # reserve rather than right — the same trade `_decoration_margins` makes for
6215 # the static figure.
6216 anim_colorbar = bool(specs and specs[0].get("marker_extra", {}).get("showscale"))
6217 horizontal_colorbar = anim_colorbar and colorbar_orientation == "Horizontal"
6218 right_reserve = (
6219 _COLORBAR_RESERVE_PX if (anim_colorbar and not horizontal_colorbar) else 0
6220 )
6221 top_reserve = _CONTROLS_MARGIN_PX
6222 bottom_reserve = _COLORBAR_BOTTOM_PX if horizontal_colorbar else 0
6223 grid_left = _GRID_LEFT_RESERVE_PX if show_coordinate_grid else 0
6224 grid_bottom = _GRID_BOTTOM_RESERVE_PX if show_coordinate_grid else 0
6225 # Stimulus-page background image (MultiplEYE) — same layout image as
6226 # make_scanpath_figure: placed at its (centered) origin, UNDER every trace,
6227 # and persisting across frames (a layout image, not per-frame data). Lets the
6228 # animated replay show the rendered page exactly like the static plot.
6229 bg_spec = _background_image_spec(
6230 background_image,
6231 background_image_size,
6232 background_image_origin,
6233 background_image_opacity,
6234 )
6235 bg_images = [bg_spec] if bg_spec else []
6236 xaxis = dict(
6237 showticklabels=False,
6238 showgrid=False,
6239 zeroline=False,
6240 title=None,
6241 range=x_range,
6242 constrain="domain",
6243 automargin=False,
6244 )
6245 yaxis = dict(
6246 showticklabels=False,
6247 showgrid=False,
6248 zeroline=False,
6249 title=None,
6250 range=y_range,
6251 constrain="domain",
6252 scaleanchor="x",
6253 scaleratio=1,
6254 automargin=False,
6255 )
6256 _apply_coordinate_grid_axes(
6257 xaxis,
6258 yaxis,
6259 show=show_coordinate_grid,
6260 spacing=coordinate_grid_spacing,
6261 x_range=x_range,
6262 y_range=y_range,
6263 rendered_width=fitted_w,
6264 rendered_height=fitted_h,
6265 )
6266 layout = dict(
6267 height=(
6268 fitted_h + top_reserve + bottom_reserve + grid_bottom + _CONTROLS_SAFETY_PX
6269 ),
6270 width=fitted_w + grid_left + right_reserve,
6271 autosize=False,
6272 images=bg_images,
6273 margin=dict(
6274 l=grid_left,
6275 r=right_reserve,
6276 t=top_reserve,
6277 b=bottom_reserve + grid_bottom,
6278 ),
6279 xaxis=xaxis,
6280 yaxis=yaxis,
6281 template="plotly_white",
6282 plot_bgcolor=background_color,
6283 paper_bgcolor=background_color,
6284 font=font_settings,
6285 shapes=shapes,
6286 sliders=sliders,
6287 updatemenus=updatemenus,
6288 )
6289 # The PRE-2 highlight overlays carry legend entries too, so they get the same
6290 # floating key as the A/B and colour-category legends.
6291 flag_legend = any(s.get("flag_overlays") for s in specs)
6292 if dual or category_legend or flag_legend:
6293 layout["legend"] = dict(
6294 orientation="h",
6295 yanchor="top",
6296 y=0.99,
6297 xanchor="right",
6298 x=0.99,
6299 bgcolor="rgba(255,255,255,0.7)",
6300 bordercolor="#cccccc",
6301 borderwidth=1,
6302 )
6303 if dual:
6304 layout["legend"]["font"] = _compare_legend_font(base_font_size, font_family)
6305 fig.update_layout(**layout)
6306 # BUG-93: the replay's clock — each frame's reading time and the speed — for
6307 # the wall-clock player every HTML surface embeds, plus VIZ-10's autoplay
6308 # intent, which that player reads on load (Plotly's own `auto_play` ignores
6309 # the frame duration). No frames → no times, so no player and no autoplay.
6310 fig.layout.meta = _replay_clock_meta(
6311 [round(float(t), 3) for t in frame_times], playback_speed, autoplay
6312 )
6313 return fig, frame_step_ms
6316def _resolve_trial_display_name(
6317 participant: str,
6318 trial_id: str,
6319 trial_words: pd.DataFrame,
6320 trial_labels: tuple[str, str] | None,
6321 idx: int,
6322) -> str:
6323 """Scanpath ``idx``'s name as written; the builders make it literal
6324 (:func:`_plotly_literal`) where they draw it."""
6325 if trial_labels is not None and len(trial_labels) > idx:
6326 return trial_labels[idx]
6327 text_id = None
6328 if "text_id" in trial_words.columns and not trial_words.empty:
6329 text_id = trial_words["text_id"].iloc[0]
6330 text_str = str(text_id) if text_id is not None else ""
6331 trial_str = str(trial_id)
6332 contains_text = text_str and text_str.lower() in trial_str.lower()
6333 if text_str:
6334 return (
6335 f"{text_str} · {participant}"
6336 if contains_text
6337 else f"{text_str} · {participant} (trial {trial_str})"
6338 )
6339 return f"{trial_str} · {participant}"
6342def _comparison_scanpath_style(
6343 idx: int,
6344 override: dict | None = None,
6345 *,
6346 default_marker_size_range: tuple[int, int] = DEFAULT_MARKER_SIZE_RANGE,
6347) -> dict:
6348 """Resolve the per-scanpath style for a comparison trace.
6350 Defaults reproduce the classic two-flat-colour look (``COMPARISON_PALETTE``);
6351 ``override`` (from the rail's per-scanpath styling panel) wins per key.
6352 """
6353 base = {
6354 "fix_color": compare_palette_color(idx),
6355 "saccade_color": compare_palette_color(idx),
6356 "saccade_style": "solid",
6357 "saccade_width": DEFAULT_SACCADE_WIDTH,
6358 "marker_size_range": default_marker_size_range,
6359 "hollow": False,
6360 "opacity": COMPARE_FIXATION_OPACITY,
6361 }
6362 if override:
6363 # Drop falsy values (None / "") so a blank colour can't override the
6364 # palette default and reach Plotly as a dark/None marker colour. The
6365 # per-scanpath filters (CMP-24) are the exception: an empty one is a
6366 # real answer — "no filter on this scanpath" — not a missing colour.
6367 base.update(
6368 {
6369 k: v
6370 for k, v in override.items()
6371 if v or (k in COMPARE_FILTER_STYLE_KEYS and v is not None)
6372 }
6373 )
6374 return base
6377#: CMP-24: the style keys that carry a scanpath's own *filters* rather than its
6378#: look. A comparison draws each reading under its own — the figure-level
6379#: ``fixation_flags`` / ``saccade_classes`` apply to a scanpath whose style names
6380#: none, so one setting filters both and a style entry overrides it per side.
6381COMPARE_FILTER_STYLE_KEYS = frozenset({"fixation_flags", "saccade_classes"})
6384def _visible_saccade_classes(classes: Iterable[str] | None) -> set[str] | None:
6385 """The saccade classes to draw, ``None`` meaning all (VIZ-31's rule: an empty
6386 or complete list is no filter)."""
6387 if not classes or set(classes) >= set(SACCADE_CLASS_ORDER):
6388 return None
6389 return set(classes)
6392def _comparison_filters(style: dict, settings: FigureSettings) -> dict:
6393 """One scanpath's filters (CMP-24): its style's own, else the figure's."""
6394 return dict(
6395 fixation_flags=style.get("fixation_flags", settings.fixation_flags),
6396 saccade_classes=style.get("saccade_classes", settings.saccade_classes),
6397 )
6400def _arrow_class_mask(
6401 fixations: pd.DataFrame, saccade_classes: pd.Series, keep: set, aseg: list
6402) -> list[bool]:
6403 """Which arrowheads belong to a visible saccade class (VIZ-31).
6405 An arrowhead belongs to the saccade leaving fixation ``aseg[j]`` in time
6406 order, which is exactly how ``_saccade_segments_by_class`` keys a segment —
6407 so the filter drops arrows for hidden saccades instead of leaving them
6408 floating over nothing."""
6409 ordered_cls = saccade_classes.reindex(
6410 fixations.sort_values("timestamp_ms").index
6411 ).tolist()
6412 return [
6413 (
6414 "other"
6415 if i >= len(ordered_cls) or pd.isna(ordered_cls[i])
6416 else ordered_cls[i]
6417 )
6418 in keep
6419 for i in aseg
6420 ]
6423def _comparison_raw_gaze(
6424 raw_gaze: pd.DataFrame | None, trial: tuple[str, str], *, show: bool
6425) -> pd.DataFrame | None:
6426 """One reading's raw-gaze samples out of the comparison's frame, or ``None``.
6428 ``raw_gaze`` carries both readings keyed exactly as the words / fixations
6429 frames are — B's ids already namespaced or renamed apart by the caller — so
6430 it is sliced by the same ``(participant, trial)`` pair.
6431 """
6432 if not show or raw_gaze is None or raw_gaze.empty:
6433 return None
6434 rows = raw_gaze[
6435 (raw_gaze["participant_id"] == trial[0]) & (raw_gaze["trial_id"] == trial[1])
6436 ]
6437 return rows if not rows.empty else None
6440def _add_comparison_raw_gaze_trace(
6441 fig: go.Figure,
6442 samples: pd.DataFrame | None,
6443 display_name: str,
6444 color: str,
6445 settings: FigureSettings,
6446 *,
6447 row: int | None = None,
6448 col: int | None = None,
6449) -> None:
6450 """One reading's raw-gaze samples in a comparison figure (VIZ-48).
6452 Drawn in that scanpath's own colour (its style's ``raw_gaze_color``, else
6453 its fixation colour) rather than the single-trial figure's time scale: two clouds on one Viridis ramp could not be told apart in an
6454 overlay, and the A/B colour is the cue every other comparison layer keeps.
6455 Size and opacity are the 🔵 Raw gaze settings, as on the single figure. The
6456 trace joins its scanpath's legend group, so toggling A in the legend hides
6457 A's samples with it.
6458 """
6459 if samples is None or samples.empty:
6460 return
6461 if "timestamp_ms" in samples.columns:
6462 customdata, when = samples["timestamp_ms"], "<br>Timestamp: %{customdata} ms"
6463 elif SAMPLE_INDEX in samples.columns: # no clock: the sample's number
6464 customdata, when = samples[SAMPLE_INDEX], "<br>Sample #: %{customdata}"
6465 else:
6466 customdata, when = None, ""
6467 trace = go.Scatter(
6468 x=samples["x"],
6469 y=samples["y"],
6470 mode="markers",
6471 marker=dict(
6472 size=settings.raw_gaze_marker_size,
6473 color=color,
6474 opacity=settings.raw_gaze_opacity,
6475 ),
6476 hovertemplate=(
6477 f"Raw gaze · {display_name}<br>x: %{{x:.1f}}<br>y: %{{y:.1f}}"
6478 + when
6479 + "<extra></extra>"
6480 ),
6481 customdata=customdata,
6482 name=f"{display_name} · raw gaze",
6483 legendgroup=display_name,
6484 showlegend=bool(settings.show_legend),
6485 )
6486 if row is not None:
6487 fig.add_trace(trace, row=row, col=col)
6488 else:
6489 fig.add_trace(trace)
6492def _add_comparison_fixation_trace(
6493 fig: go.Figure,
6494 trial_fix: pd.DataFrame,
6495 display_name: str,
6496 style: dict,
6497 font_settings: dict,
6498 *,
6499 show_fixations: bool = True,
6500 show_saccades: bool = True,
6501 show_saccade_arrows: bool = False,
6502 show_order: bool = True,
6503 order_font_size: int | None = None,
6504 show_legend: bool = False,
6505 color_by: str | None = None,
6506 colorscale: str = DEFAULT_FIXATION_COLORSCALE,
6507 color_range: tuple[float, float] | None = None,
6508 show_colorbar: bool = False,
6509 colorbar_style: dict | None = None,
6510 fixation_symbol: str = DEFAULT_FIXATION_SYMBOL,
6511 fixation_hover_fields: Sequence[str] | None = None,
6512 row: int | None = None,
6513 col: int | None = None,
6514 trial_words: pd.DataFrame | None = None,
6515 fixation_flags: dict | None = None,
6516 saccade_classes: Iterable[str] | None = None,
6517 duration_scale: dict | None = None,
6518 category_colors: Sequence[str] | None = None,
6519) -> None:
6520 """Add one scanpath's saccades + fixation markers to a comparison figure.
6522 Saccades and markers are separate traces (mirroring the single-trial figure)
6523 so the per-scanpath saccade colour/line-style/line-width and hollow markers
6524 all apply, and the shared ``show_saccades`` / ``show_saccade_arrows`` /
6525 ``show_order`` toggles take effect.
6527 Fixation colour: by default each scanpath uses its flat per-scanpath colour
6528 (the A/B cue). When ``color_by`` names a numeric column, the marker **fill** is
6529 coloured by that metric (shared ``colorscale`` / ``color_range`` across both
6530 scanpaths) and the per-scanpath flat colour becomes the marker **outline**, so
6531 the readings stay distinguishable while still showing the metric. Order numbers
6532 are tinted to the per-scanpath colour either way.
6534 ``category_colors`` is the discrete counterpart — one literal colour per row
6535 of ``trial_fix``, from the figure's shared category→colour mapping
6536 (:func:`_shared_category_colors`, for a categorical column or colour-by-line).
6537 It is drawn the same way: category fill, per-scanpath outline.
6539 ``fixation_symbol`` (VIZ-15/23) sets the marker shape — shape is the channel
6540 that survives a greyscale print, which is exactly what comparison figures get
6541 used for. A glyph shape (♥) is drawn as text, as on the static figure
6542 (:func:`_glyph_scatter_traces`), its A/B outline a larger glyph beneath.
6544 ``show_fixations=False`` (CMP-7) drops the marker trace, and with it the
6545 fixation-index labels that ride on it as marker text — the same thing the
6546 toggle does on the static figure. The saccade and arrow layers are
6547 independent and keep their own toggles, so a lines-only comparison is still
6548 reachable. This is what makes a comparison *heatmap* readable: the whole
6549 point of the split word boxes is lost under two full sets of markers.
6551 CMP-24: ``fixation_flags`` and ``saccade_classes`` are *this* scanpath's
6552 filters, applied as the static figure applies them — *Discard* drops markers
6553 and their index labels (the saccades still bridge across them), *Highlight*
6554 overlays its marker, and hidden saccade classes lose their line and arrow.
6555 Both need ``trial_words`` for the geometry they classify against.
6557 ``duration_scale`` is the figure's (``_settings_size_scale``): both
6558 scanpaths share it, so under a fixed scale one duration draws at one size
6559 on either side; only the size *range* is per scanpath.
6560 """
6561 if trial_fix.empty:
6562 return
6563 if category_colors is not None:
6564 # Positional, before *Discard* drops rows, so the two stay aligned.
6565 trial_fix = trial_fix.assign(_category_color=list(category_colors))
6566 words_for_flags = trial_words if trial_words is not None else pd.DataFrame()
6567 keep = _visible_saccade_classes(saccade_classes)
6568 class_series = None
6569 if keep is not None and (show_saccades or show_saccade_arrows):
6570 existing = trial_fix.get("saccade_class")
6571 if existing is not None:
6572 class_series = existing
6573 else:
6574 from .measures import classify_saccades
6576 class_series = classify_saccades(trial_fix, words_for_flags)
6577 fix_color = style["fix_color"]
6578 saccade_color = style["saccade_color"]
6579 saccade_style = style.get("saccade_style", "solid")
6580 saccade_width = style.get("saccade_width", DEFAULT_SACCADE_WIDTH)
6582 def _add(trace):
6583 if row is not None and col is not None:
6584 fig.add_trace(trace, row=row, col=col)
6585 else:
6586 fig.add_trace(trace)
6588 # Comparison figures always draw straight connectors (no Arc mode here); bind
6589 # it once so the segments and the arrowheads can never disagree (BUG-9).
6590 arch_frac: float | None = None
6591 if show_saccades and len(trial_fix) > 1:
6592 if class_series is not None:
6593 segs = _saccade_segments_by_class(
6594 trial_fix, "x", "y", class_series, arch_frac
6595 )
6596 sx, sy = [], []
6597 for cls_name, (cx, cy) in segs.items():
6598 if cls_name in keep:
6599 sx.extend(cx)
6600 sy.extend(cy)
6601 else:
6602 sx, sy = _saccade_segments(trial_fix, "x", "y", arch_frac)
6603 if sx:
6604 _add(
6605 go.Scatter(
6606 x=sx,
6607 y=sy,
6608 mode="lines",
6609 line=dict(
6610 color=saccade_color, width=saccade_width, dash=saccade_style
6611 ),
6612 name=display_name,
6613 legendgroup=display_name,
6614 showlegend=False,
6615 hoverinfo="skip",
6616 )
6617 )
6618 if show_saccade_arrows and len(trial_fix) > 1:
6619 amx, amy, aang, aseg = _saccade_arrow_rows(trial_fix, "x", "y", arch_frac)
6620 if amx and class_series is not None:
6621 mask = _arrow_class_mask(trial_fix, class_series, keep, aseg)
6622 amx = [v for v, m in zip(amx, mask) if m]
6623 amy = [v for v, m in zip(amy, mask) if m]
6624 aang = [v for v, m in zip(aang, mask) if m]
6625 if amx:
6626 _add(
6627 go.Scatter(
6628 x=amx,
6629 y=amy,
6630 mode="markers",
6631 marker=dict(
6632 symbol="arrow",
6633 size=12,
6634 angle=aang,
6635 angleref="up",
6636 color=saccade_color,
6637 line=dict(width=0),
6638 ),
6639 legendgroup=display_name,
6640 showlegend=False,
6641 hoverinfo="skip",
6642 )
6643 )
6645 if not show_fixations:
6646 return
6648 # Only the categories doing something: the rail always sends all four, so
6649 # an untouched set must cost nothing (the classification scans the boxes).
6650 flags = {
6651 cat: spec
6652 for cat, spec in (fixation_flags or {}).items()
6653 if isinstance(spec, dict) and str(spec.get("mode") or "Off") != "Off"
6654 }
6655 if flags:
6656 trial_fix = _discard_flagged_fixations(trial_fix, words_for_flags, flags)
6657 if trial_fix.empty:
6658 return
6659 sizes = _compute_marker_sizes(
6660 trial_fix["duration_ms"],
6661 style["marker_size_range"],
6662 **(duration_scale or {}),
6663 )
6664 # Metric colouring ("Color fixations by") when a numeric column is chosen:
6665 # colour the FILL by the metric (shared colorscale/range across both
6666 # scanpaths) and keep the per-scanpath flat colour as the marker OUTLINE so
6667 # A/B stay distinguishable. Otherwise the fill is the flat per-scanpath colour.
6668 metric_color = bool(
6669 color_by
6670 and color_by != "line"
6671 and color_by in trial_fix.columns
6672 and pd.api.types.is_numeric_dtype(trial_fix[color_by])
6673 )
6674 symbol = _marker_symbol(fixation_symbol)
6675 if metric_color:
6676 marker = dict(
6677 size=sizes,
6678 symbol=symbol,
6679 color=trial_fix[color_by],
6680 colorscale=colorscale,
6681 cmin=color_range[0] if color_range else None,
6682 cmax=color_range[1] if color_range else None,
6683 showscale=bool(show_colorbar),
6684 # VIZ-23: the same styled colorbar the static figure builds, so the
6685 # orientation / tick-angle / tick-size controls reach Compare too.
6686 colorbar=_colorbar_dict(_column_title(color_by), **(colorbar_style or {}))
6687 if show_colorbar
6688 else None,
6689 line=dict(color=fix_color, width=1.4),
6690 )
6691 elif category_colors is not None:
6692 # The discrete counterpart: the shared category colour fills, the
6693 # per-scanpath colour outlines — the same A/B cue as a numeric metric.
6694 marker = dict(
6695 size=sizes,
6696 symbol=symbol,
6697 color=trial_fix["_category_color"].tolist(),
6698 line=dict(color=fix_color, width=1.4),
6699 )
6700 else:
6701 marker = dict(
6702 size=sizes,
6703 symbol=symbol,
6704 color=fix_color,
6705 line=dict(color=FIX_MARKER_OUTLINE, width=0.5),
6706 )
6707 # Per-scanpath marker alpha (VIZ-6): always set it (even 1.0) so the control
6708 # overrides Plotly's ~0.7 default for variable-size scatter markers.
6709 opacity = style.get("opacity", 1.0)
6710 marker["opacity"] = float(opacity if opacity is not None else 1.0)
6711 if style.get("hollow"):
6712 marker = _make_hollow(marker)
6713 order_font = dict(font_settings)
6714 order_font["color"] = fix_color
6715 if order_font_size is not None:
6716 order_font["size"] = order_font_size
6717 # VIZ-26 hover fields (the "Hover fields" multiselect under 👁️ Fixations)
6718 # reach the comparison figure too — it used to hard-code Order/Time/Duration
6719 # regardless of that setting, which is what made the control look inert
6720 # while comparing. Same fallback list the static + animation builders use
6721 # when nothing is explicitly chosen, so the three render paths agree.
6722 hover_fields = (
6723 ["order_in_trial", "duration_ms", "word_id"]
6724 if fixation_hover_fields is None
6725 else list(fixation_hover_fields)
6726 )
6727 customdata, hovertemplate = _hover_payload(
6728 trial_fix, hover_fields, fixation=True, words=trial_words
6729 )
6730 glyph = FIXATION_GLYPH_SYMBOLS.get(fixation_symbol or "")
6731 # The trace's own legend swatch would mislead under category colours (it
6732 # shows the first fixation's category) and cannot draw a glyph (♥), so in
6733 # either case a separate entry names the scanpath.
6734 own_entry = category_colors is not None or bool(glyph)
6735 if glyph:
6736 # VIZ-15: ♥ as text, as on the static figure (`_glyph_scatter_traces`);
6737 # the index labels and a numeric colour bar get traces of their own.
6738 for trace in _glyph_scatter_traces(
6739 trial_fix["x"],
6740 trial_fix["y"],
6741 marker,
6742 glyph,
6743 name=display_name,
6744 legendgroup=display_name,
6745 showlegend=False,
6746 hovertemplate=f"{display_name}<br>{hovertemplate}",
6747 customdata=customdata,
6748 ):
6749 _add(trace)
6750 if show_order:
6751 _add(
6752 go.Scatter(
6753 x=trial_fix["x"],
6754 y=trial_fix["y"],
6755 mode="text",
6756 text=trial_fix["order_in_trial"],
6757 textposition="top center",
6758 textfont=order_font,
6759 name=f"{display_name} · index",
6760 legendgroup=display_name,
6761 showlegend=False,
6762 hoverinfo="skip",
6763 )
6764 )
6765 bar = _glyph_colorbar_trace(marker, trial_fix[color_by] if metric_color else ())
6766 if bar is not None:
6767 _add(bar)
6768 else:
6769 _add(
6770 go.Scatter(
6771 x=trial_fix["x"],
6772 y=trial_fix["y"],
6773 mode="markers+text" if show_order else "markers",
6774 marker=marker,
6775 name=display_name,
6776 legendgroup=display_name,
6777 showlegend=show_legend and not own_entry,
6778 text=trial_fix["order_in_trial"] if show_order else None,
6779 textposition="top center",
6780 textfont=order_font,
6781 hovertemplate=f"{display_name}<br>{hovertemplate}",
6782 customdata=customdata,
6783 )
6784 )
6785 if show_legend and own_entry:
6786 coloured = metric_color or category_colors is not None
6787 _add(
6788 go.Scatter(
6789 x=[None],
6790 y=[None],
6791 mode="markers",
6792 marker=dict(
6793 size=10,
6794 symbol=symbol,
6795 color="#ffffff" if coloured else fix_color,
6796 line=dict(
6797 color=fix_color if coloured else FIX_MARKER_OUTLINE,
6798 width=2 if coloured else 0.5,
6799 ),
6800 ),
6801 name=display_name,
6802 legendgroup=display_name,
6803 showlegend=True,
6804 hoverinfo="skip",
6805 )
6806 )
6807 if flags:
6808 overlay = _fixation_flag_masks(trial_fix, words_for_flags, flags)
6809 for cat in _FIX_FLAG_CATEGORIES:
6810 spec = flags.get(cat, {})
6811 if spec.get("mode") != "Highlight" or cat not in overlay:
6812 continue
6813 hits = trial_fix[overlay[cat]]
6814 if hits.empty:
6815 continue
6816 name = _FIX_FLAG_LABELS[cat]
6817 _add(
6818 go.Scatter(
6819 x=hits["x"],
6820 y=hits["y"],
6821 mode="markers",
6822 marker=dict(
6823 symbol=spec.get("symbol") or "x",
6824 size=13,
6825 color=spec.get("color") or OUT_OF_TEXT_COLOR,
6826 line=dict(color=fix_color, width=1.5),
6827 ),
6828 name=f"{display_name} · {name}",
6829 legendgroup=display_name,
6830 showlegend=show_legend,
6831 hovertemplate=(
6832 f"{display_name} · {name} fixation<br>"
6833 "x %{x:.0f}, y %{y:.0f}<extra></extra>"
6834 ),
6835 )
6836 )
6839def _comparison_metric_colorbar(
6840 fixations: pd.DataFrame, color_by: str | None, show_colorbars: bool
6841) -> bool:
6842 """Whether a comparison figure will actually draw a metric colorbar.
6844 Same test :func:`_add_comparison_fixation_trace` makes per trace, hoisted so
6845 the layout can reserve room for a horizontal bar below the plot (VIZ-23).
6846 """
6847 return bool(
6848 show_colorbars
6849 and color_by
6850 and color_by != "line"
6851 and color_by in fixations.columns
6852 and pd.api.types.is_numeric_dtype(fixations[color_by])
6853 )
6856def _word_id_keys(values: pd.Series) -> pd.Series:
6857 """Word ids as join keys that mean the same thing on both frames (CMP-7).
6859 The words table and the fixations table routinely disagree on dtype: word
6860 boxes carry an integer ``word_id`` while a fixation's is a float, because it
6861 is NaN wherever the fixation landed outside every box. A plain ``str()`` then
6862 yields ``"7"`` on one side and ``"7.0"`` on the other, so a keyed join
6863 silently matches nothing — which is exactly how the comparison heatmap came
6864 out empty. Whole numbers lose the decimal tail here; anything non-numeric
6865 keeps its stripped string, so datasets with string word ids still join.
6866 """
6867 numeric = pd.to_numeric(values, errors="coerce")
6868 integral = numeric.notna() & (numeric % 1 == 0)
6869 text = values.astype(str).str.strip()
6870 if not integral.any():
6871 return text
6872 return text.mask(integral, numeric.where(integral, 0).astype("int64").astype(str))
6875def _comparison_word_heatmap_data(
6876 trial_specs: Sequence[dict],
6877 *,
6878 metric: str,
6879 heatmap_range: tuple[float, float] | None,
6880 heatmap_norm: str,
6881) -> tuple[list[dict[str, float]], float, float, str]:
6882 """Per-trial word values and one shared transformed colour range (CMP-7)."""
6883 value_maps: list[dict[str, float]] = []
6884 all_values: list[float] = []
6885 duration_weighted = metric == "duration_ms"
6886 for spec in trial_specs:
6887 fixations = spec["trial_fix"]
6888 if fixations.empty or "word_id" not in fixations.columns:
6889 values: dict[str, float] = {}
6890 else:
6891 valid = fixations[fixations["word_id"].notna()].copy()
6892 keys = _word_id_keys(valid["word_id"])
6893 if duration_weighted and "duration_ms" in valid.columns:
6894 grouped = (
6895 pd.to_numeric(valid["duration_ms"], errors="coerce")
6896 .groupby(keys)
6897 .sum()
6898 )
6899 else:
6900 grouped = valid.groupby(keys).size()
6901 values = {str(key): float(value) for key, value in grouped.items()}
6902 value_maps.append(values)
6903 all_values.extend(value for value in values.values() if value > 0)
6904 if heatmap_range is not None:
6905 raw_min, raw_max = map(float, heatmap_range)
6906 elif all_values:
6907 # Auto starts at 0, as the single-trial word heatmap does.
6908 raw_min, raw_max = 0.0, max(all_values)
6909 else:
6910 raw_min, raw_max = 0.0, 1.0
6911 z_min = float(_apply_heatmap_norm(raw_min, heatmap_norm))
6912 z_max = float(_apply_heatmap_norm(raw_max, heatmap_norm))
6913 if z_max <= z_min:
6914 z_max = z_min + 1.0
6915 title = _WORD_DWELL_TITLE if duration_weighted else "Fixation count"
6916 return value_maps, z_min, z_max, title
6919def _comparison_heatmap_shapes(
6920 words: pd.DataFrame,
6921 values: dict[str, float],
6922 *,
6923 heatmap_colorscale: str,
6924 heatmap_norm: str,
6925 z_min: float,
6926 z_max: float,
6927 half: str | None = None,
6928 xref: str | None = None,
6929 yref: str | None = None,
6930) -> list[dict]:
6931 """Tint full word boxes or their left/right half on a shared scale."""
6932 if words.empty or not values:
6933 return []
6934 from plotly.colors import sample_colorscale
6936 from .measures import word_box_bounds
6938 shapes: list[dict] = []
6939 z_span = max(z_max - z_min, 1e-9)
6940 if "word_id" not in words.columns:
6941 return []
6942 keys = _word_id_keys(words["word_id"])
6943 for key, (x0, y0, x1, y1) in zip(keys, zip(*word_box_bounds(words))):
6944 value = values.get(key, 0.0)
6945 if value <= 0:
6946 continue
6947 midpoint = (x0 + x1) / 2.0
6948 if half == "left":
6949 x1 = midpoint
6950 elif half == "right":
6951 x0 = midpoint
6952 transformed = float(_apply_heatmap_norm(value, heatmap_norm))
6953 position = max(0.0, min(1.0, (transformed - z_min) / z_span))
6954 shape = dict(
6955 type="rect",
6956 x0=x0,
6957 y0=y0,
6958 x1=x1,
6959 y1=y1,
6960 line=dict(width=0),
6961 fillcolor=sample_colorscale(heatmap_colorscale, [position])[0],
6962 opacity=0.55,
6963 layer="below",
6964 name=_shape_layer_tag("heatmap"),
6965 )
6966 if xref is not None:
6967 shape["xref"] = xref
6968 if yref is not None:
6969 shape["yref"] = yref
6970 shapes.append(shape)
6971 return shapes
6974def _comparison_heatmap_colorbar_trace(
6975 *,
6976 colorscale: str,
6977 z_min: float,
6978 z_max: float,
6979 title: str,
6980 heatmap_norm: str,
6981 colorbar_style: dict,
6982) -> go.Scatter:
6983 return go.Scatter(
6984 x=[None],
6985 y=[None],
6986 mode="markers",
6987 marker=dict(
6988 colorscale=colorscale,
6989 showscale=True,
6990 cmin=z_min,
6991 cmax=z_max,
6992 colorbar=_colorbar_dict(
6993 _heatmap_title(title, heatmap_norm), **colorbar_style
6994 ),
6995 ),
6996 showlegend=False,
6997 hoverinfo="skip",
6998 name="comparison heatmap colorbar",
6999 )
7002def _comparison_heatmap_colorbar_traces(
7003 trial_specs: Sequence[dict],
7004 *,
7005 z_min: float,
7006 z_max: float,
7007 title: str,
7008 heatmap_norm: str,
7009 colorbar_style: dict,
7010 overlay: bool,
7011) -> list[go.Scatter]:
7012 """The comparison heatmap's colour bar(s): one while A and B share a colour
7013 scale, else one per scanpath on the same range (`_arrange_colorbars` sets
7014 them side by side). The overlay's titles say which half is whose."""
7015 scales = [spec["heatmap_colorscale"] for spec in trial_specs]
7016 sides = ("left A", "right B") if overlay else ("A", "B")
7017 if len(set(scales)) == 1:
7018 named = [(scales[0], " · A left half, B right" if overlay else "")]
7019 else:
7020 named = [(scale, f" · {side}") for scale, side in zip(scales, sides)]
7021 return [
7022 _comparison_heatmap_colorbar_trace(
7023 colorscale=scale,
7024 z_min=z_min,
7025 z_max=z_max,
7026 title=title + suffix,
7027 heatmap_norm=heatmap_norm,
7028 colorbar_style=colorbar_style,
7029 )
7030 for scale, suffix in named
7031 ]
7034def _make_split_comparison_figure(
7035 words: pd.DataFrame,
7036 fixations: pd.DataFrame,
7037 trial_a: tuple[str, str],
7038 trial_b: tuple[str, str],
7039 *,
7040 settings: FigureSettings,
7041 orientation: str,
7042 styles: tuple[dict, dict] | None = None,
7043 raw_gaze: pd.DataFrame | None = None,
7044) -> go.Figure:
7045 """Two-panel comparison, either horizontal (side-by-side) or vertical (stacked).
7047 Each panel computes its **own** axis ranges from its trial's words +
7048 fixations, which is what lets the two panels sit in different coordinate
7049 spaces at all (CMP-8).
7051 The stimulus-image background (VIZ-4) is added to both panels. It used to be
7052 the *same* image on the grounds that "the two readings are of the same
7053 text" — no longer true once B may come from another corpus, so B's panel
7054 takes ``background_image_b`` (and its size/origin) when given and falls back
7055 to A's otherwise, which is every same-dataset comparison.
7057 Likewise ``canvas_b``: B's screen, defaulting to A's. It feeds B's axis
7058 ranges *and* its label-fitting pair, so B's reading text stays true-to-scale
7059 on **B's** monitor rather than being sized against A's.
7060 """
7061 from plotly.subplots import make_subplots
7063 canvas_width = settings.canvas_width
7064 canvas_height = settings.canvas_height
7065 font_family = settings.font_family
7066 base_font_size = settings.base_font_size
7067 show_words = settings.show_words
7068 show_word_labels = settings.show_word_labels
7069 trial_labels = settings.trial_labels
7070 marker_size_range = settings.marker_size_range
7071 show_fixations = settings.show_fixations
7072 show_saccades = settings.show_saccades
7073 show_saccade_arrows = settings.show_saccade_arrows
7074 show_order = settings.show_order
7075 show_legend = settings.show_legend
7076 order_font_size = settings.order_font_size
7077 color_by = settings.color_by
7078 color_by_line = settings.color_by_line
7079 fixation_colorscale = settings.fixation_colorscale
7080 fixation_color_range = settings.fixation_color_range
7081 fixation_symbol = settings.fixation_symbol
7082 # The fixations' colour bar and the heatmap's, each with its own style.
7083 show_colorbars = settings.show_fixation_colorbar
7084 show_heatmap_colorbar = settings.show_heatmap_colorbar
7085 show_heatmap = settings.show_heatmap
7086 heatmap_metric = settings.heatmap_metric
7087 heatmap_range = settings.heatmap_range
7088 heatmap_norm = settings.heatmap_norm
7089 colorbar_orientation = settings.fixation_colorbar_orientation
7090 colorbar_tickangle = settings.fixation_colorbar_tickangle
7091 colorbar_tickfont_size = settings.fixation_colorbar_tickfont_size
7092 heat_cb_style = dict(
7093 orientation=settings.heatmap_colorbar_orientation,
7094 tickangle=settings.heatmap_colorbar_tickangle,
7095 tickfont_size=settings.heatmap_colorbar_tickfont_size,
7096 )
7097 text_color = settings.text_color
7098 highlight_column = settings.highlight_column
7099 highlight_text_color = settings.highlight_text_color
7100 word_hover_measure = settings.word_hover_measure
7101 word_hover_fields = settings.word_hover_fields
7102 fixation_hover_fields = settings.fixation_hover_fields
7103 background_color = settings.background_color
7104 line_spacing = settings.line_spacing
7105 scale_text_to_boxes = settings.scale_text_to_boxes
7106 background_image = settings.background_image
7107 background_image_size = settings.background_image_size
7108 background_image_origin = settings.background_image_origin
7109 background_image_opacity = settings.background_image_opacity
7110 fit_to_monitor = settings.fit_to_monitor
7111 show_coordinate_grid = settings.show_coordinate_grid
7112 coordinate_grid_spacing = settings.coordinate_grid_spacing
7114 font_settings = dict(family=font_family or FONT_FAMILY, size=base_font_size)
7115 cb_style = dict(
7116 orientation=colorbar_orientation,
7117 tickangle=colorbar_tickangle,
7118 tickfont_size=colorbar_tickfont_size,
7119 )
7120 is_stacked = orientation == "stacked"
7121 # Per-panel canvas: B falls back to A's, so a same-dataset comparison is
7122 # byte-identical to the pre-CMP-8 figure.
7123 canvas_b = settings.canvas_b or (canvas_width, canvas_height)
7124 panel_canvas = [(canvas_width, canvas_height), (int(canvas_b[0]), int(canvas_b[1]))]
7125 # Each panel's share of its screen, for the preliminary fit that picks the
7126 # figure's size. The word labels are *not* sized from it: they are sized
7127 # from the subplot space each panel finally gets (round-8 review, finding 5).
7128 panel_widths = [w if is_stacked else w // 2 for w, _ in panel_canvas]
7129 # B's stimulus page. Inheriting A's is right for a same-dataset pair (two
7130 # readings of the same text) and *wrong* across datasets — a PoTeC panel with
7131 # a OneStop page under it is a picture of two different things. So when B has
7132 # its own screen (`canvas_b`, i.e. a cross-dataset pair) B's panel draws only
7133 # an image explicitly given for it, and otherwise draws none.
7134 if settings.background_image_b is not None:
7135 image_b = (
7136 settings.background_image_b,
7137 settings.background_image_size_b,
7138 settings.background_image_origin_b,
7139 )
7140 elif settings.canvas_b is None:
7141 image_b = (background_image, background_image_size, background_image_origin)
7142 else:
7143 image_b = (None, None, None)
7144 panel_images = [
7145 (background_image, background_image_size, background_image_origin),
7146 image_b,
7147 ]
7149 # Shared metric colour range across both panels (when colouring by a metric
7150 # and no explicit range), so the two scanpaths use one comparable scale.
7151 metric_range = fixation_color_range
7152 if (
7153 color_by
7154 and color_by != "line"
7155 and metric_range is None
7156 and color_by in fixations.columns
7157 and pd.api.types.is_numeric_dtype(fixations[color_by])
7158 ):
7159 both = fixations[
7160 (
7161 (fixations["participant_id"] == trial_a[0])
7162 & (fixations["trial_id"] == trial_a[1])
7163 )
7164 | (
7165 (fixations["participant_id"] == trial_b[0])
7166 & (fixations["trial_id"] == trial_b[1])
7167 )
7168 ][color_by]
7169 if len(both) and pd.notna(both.min()) and pd.notna(both.max()):
7170 metric_range = (float(both.min()), float(both.max()))
7172 trial_specs = []
7173 for idx, trial in enumerate([trial_a, trial_b]):
7174 participant, trial_id = trial
7175 trial_words = words[
7176 (words["participant_id"] == participant) & (words["trial_id"] == trial_id)
7177 ]
7178 trial_fix = fixations[
7179 (fixations["participant_id"] == participant)
7180 & (fixations["trial_id"] == trial_id)
7181 ].sort_values("timestamp_ms")
7182 display_name = _plotly_literal(
7183 _resolve_trial_display_name(
7184 participant, trial_id, trial_words, trial_labels, idx
7185 )
7186 )
7187 style = _comparison_scanpath_style(
7188 idx,
7189 styles[idx] if styles else None,
7190 default_marker_size_range=marker_size_range,
7191 )
7192 trial_specs.append(
7193 dict(
7194 trial_words=trial_words,
7195 trial_fix=trial_fix,
7196 raw_gaze=_comparison_raw_gaze(
7197 raw_gaze, trial, show=settings.show_raw_gaze
7198 ),
7199 display_name=display_name,
7200 style=style,
7201 color=style["fix_color"],
7202 # The word-box outline: the scanpath's own colour unless its
7203 # style names one (`box_color`).
7204 box_color=style.get("box_color") or style["fix_color"],
7205 # Its fill: the figure's unless the style names one.
7206 box_fill_color=(
7207 style.get("box_fill_color") or settings.word_box_fill_color
7208 ),
7209 # Its raw-gaze samples: the scanpath's own colour unless its
7210 # style names one (`raw_gaze_color`).
7211 raw_gaze_color=style.get("raw_gaze_color") or style["fix_color"],
7212 # Its heatmap's colour scale: the figure's unless the style
7213 # names one (`heatmap_colorscale`). The range stays shared.
7214 heatmap_colorscale=(
7215 style.get("heatmap_colorscale") or settings.heatmap_colorscale
7216 ),
7217 )
7218 )
7220 heatmap_maps, heatmap_min, heatmap_max, heatmap_title = (
7221 _comparison_word_heatmap_data(
7222 trial_specs,
7223 metric=heatmap_metric,
7224 heatmap_range=heatmap_range,
7225 heatmap_norm=heatmap_norm,
7226 )
7227 if show_heatmap
7228 else ([], 0.0, 1.0, "")
7229 )
7231 # A categorical column or colour-by-line: one category→colour mapping for
7232 # both panels, each reading's lines against its own word boxes.
7233 category_colors, category_legend = _shared_category_colors(
7234 [
7235 _fixation_category_labels(
7236 spec["trial_fix"], spec["trial_words"], color_by, color_by_line
7237 )
7238 for spec in trial_specs
7239 ],
7240 avoid=[spec["color"] for spec in trial_specs],
7241 )
7242 category_label = "line" if (color_by_line or color_by == "line") else color_by
7244 # #374 F26: each panel always says which scanpath it is, "A · …" / "B · …",
7245 # legend or not — so the top band is reserved for the titles too (BUG-90:
7246 # without it the upper title was clipped off the canvas).
7247 subplot_titles = [
7248 f"{side} · {spec['display_name']}" for side, spec in zip("AB", trial_specs)
7249 ]
7251 # Each panel's own axis ranges, from its trial's words + fixations (CMP-8).
7252 panel_ranges = []
7253 for idx, spec in enumerate(trial_specs):
7254 panel_cw, panel_ch = panel_canvas[idx]
7255 x_range, y_range, *_ = _compute_axis_ranges(
7256 panel_cw,
7257 panel_ch,
7258 (spec["trial_fix"], "x", "y"),
7259 (spec["raw_gaze"], "x", "y"),
7260 word_frames=[spec["trial_words"]] if not spec["trial_words"].empty else [],
7261 fit_to_monitor=fit_to_monitor,
7262 )
7263 panel_ranges.append((x_range, y_range))
7264 panel_fits = [
7265 _fit_display_size(
7266 panel_widths[idx], panel_canvas[idx][1], x_range, y_range, spatial_axes=True
7267 )
7268 for idx, (x_range, y_range) in enumerate(panel_ranges)
7269 ]
7271 # The figure's size, chosen before anything is sized against it. Two panels
7272 # that share a screen reuse the last panel's fit — which keeps every
7273 # same-dataset figure the size it always was. Two *different* screens can't
7274 # be reconciled that way: the panels then get the widest / tallest fit of the
7275 # pair, so neither is clipped. (Which is also why the caption in
7276 # `tabs._render_comparison_figure` says sizes are not comparable across
7277 # panels — each panel is true-to-scale on its own monitor.)
7278 if settings.canvas_b is None:
7279 panel_w, panel_h = panel_fits[-1]
7280 else:
7281 panel_w = max(fit[0] for fit in panel_fits)
7282 panel_h = max(fit[1] for fit in panel_fits)
7283 if is_stacked:
7284 total_width = panel_w
7285 total_height = panel_h * 2 + 40
7286 else: # side-by-side
7287 total_width = panel_w * 2
7288 total_height = panel_h
7289 # A colour bar gets its own reserved band — below the panels when horizontal
7290 # (VIZ-23), to their right when vertical — so the figure grows by it rather
7291 # than Plotly's automargin shrinking the panels under text already sized for
7292 # them (the single-trial figure's `_decoration_margins` rule).
7293 reserves = _colorbar_reserves(
7294 (
7295 _comparison_metric_colorbar(fixations, color_by, show_colorbars),
7296 colorbar_orientation,
7297 ),
7298 (
7299 bool(show_heatmap and show_heatmap_colorbar and any(heatmap_maps)),
7300 heat_cb_style["orientation"],
7301 ),
7302 )
7303 bottom_px = _COLORBAR_BOTTOM_PX if reserves["colorbar_below"] else 0
7304 right_px = _COLORBAR_RESERVE_PX if reserves["colorbar_right"] else 0
7305 grid_left = _GRID_LEFT_RESERVE_PX if show_coordinate_grid else 0
7306 grid_bottom = _GRID_BOTTOM_RESERVE_PX if show_coordinate_grid else 0
7307 # The t band was the (now-removed) title; keep a slim band only for the
7308 # optional legend.
7309 top_px = _compare_legend_font(base_font_size)["size"] + 14
7310 figure_width = total_width + grid_left + right_px
7311 figure_height = total_height + bottom_px + grid_bottom
7312 plot_area = (
7313 max(figure_width - grid_left - right_px, 1),
7314 max(figure_height - top_px - bottom_px - grid_bottom, 1),
7315 )
7316 if is_stacked:
7317 fig = make_subplots(
7318 rows=2,
7319 cols=1,
7320 vertical_spacing=0.08,
7321 subplot_titles=subplot_titles,
7322 )
7323 else:
7324 fig = make_subplots(
7325 rows=1,
7326 cols=2,
7327 horizontal_spacing=0.04,
7328 subplot_titles=subplot_titles,
7329 )
7331 # Each panel's data→screen scale, from the subplot space it actually gets
7332 # in the final figure: its domain's share of the plot area, then the
7333 # equal-aspect constraint, which shrinks whichever side has room to spare.
7334 panel_displays = []
7335 for idx, (x_range, y_range) in enumerate(panel_ranges):
7336 suffix = "" if idx == 0 else str(idx + 1)
7337 x_domain = fig.layout[f"xaxis{suffix}"].domain
7338 y_domain = fig.layout[f"yaxis{suffix}"].domain
7339 scale = _display_scale(
7340 x_range,
7341 y_range,
7342 (x_domain[1] - x_domain[0]) * plot_area[0],
7343 (y_domain[1] - y_domain[0]) * plot_area[1],
7344 )
7345 panel_displays.append(
7346 (
7347 scale,
7348 round((x_range[1] - x_range[0]) * scale),
7349 round((y_range[0] - y_range[1]) * scale),
7350 )
7351 )
7353 all_shapes: list = []
7354 for idx, spec in enumerate(trial_specs):
7355 if is_stacked:
7356 row, col = idx + 1, 1
7357 axis_suffix = "" if idx == 0 else str(idx + 1)
7358 else:
7359 row, col = 1, idx + 1
7360 axis_suffix = "" if idx == 0 else str(idx + 1)
7361 xref = f"x{axis_suffix}"
7362 yref = f"y{axis_suffix}"
7363 trial_words = spec["trial_words"]
7364 trial_fix = spec["trial_fix"]
7365 x_range, y_range = panel_ranges[idx]
7366 panel_scale, panel_display_w, panel_display_h = panel_displays[idx]
7368 # Stimulus-page background image (VIZ-4/23), one per panel, UNDER every
7369 # trace — `row`/`col` bind it to this panel's axes. B may carry its own
7370 # (CMP-8); it falls back to A's when it doesn't.
7371 panel_image, panel_image_size, panel_image_origin = panel_images[idx]
7372 _add_background_image(
7373 fig,
7374 panel_image,
7375 panel_image_size,
7376 panel_image_origin,
7377 background_image_opacity,
7378 row=row,
7379 col=col,
7380 )
7382 if show_heatmap:
7383 all_shapes.extend(
7384 _comparison_heatmap_shapes(
7385 trial_words,
7386 heatmap_maps[idx],
7387 heatmap_colorscale=spec["heatmap_colorscale"],
7388 heatmap_norm=heatmap_norm,
7389 z_min=heatmap_min,
7390 z_max=heatmap_max,
7391 xref=xref,
7392 yref=yref,
7393 )
7394 )
7396 if show_words and not trial_words.empty:
7397 for box in build_word_boxes(
7398 trial_words,
7399 color=spec["box_color"],
7400 fill_color=spec["box_fill_color"],
7401 fill_opacity=settings.word_box_fill_opacity,
7402 line_opacity=settings.word_box_line_opacity,
7403 ):
7404 box = dict(box)
7405 box["xref"] = xref
7406 box["yref"] = yref
7407 all_shapes.append(box)
7409 all_shapes.append(
7410 dict(
7411 type="rect",
7412 xref=xref,
7413 yref=yref,
7414 x0=x_range[0],
7415 y0=y_range[1],
7416 x1=x_range[1],
7417 y1=y_range[0],
7418 line=dict(color="#000000", width=1),
7419 fillcolor="rgba(0,0,0,0)",
7420 )
7421 )
7423 # Under the scanpath, as on the single-trial figure.
7424 _add_comparison_raw_gaze_trace(
7425 fig,
7426 spec["raw_gaze"],
7427 spec["display_name"],
7428 spec["raw_gaze_color"],
7429 settings,
7430 row=row,
7431 col=col,
7432 )
7433 _add_comparison_fixation_trace(
7434 fig,
7435 trial_fix,
7436 spec["display_name"],
7437 spec["style"],
7438 font_settings,
7439 show_fixations=show_fixations,
7440 show_saccades=show_saccades,
7441 show_saccade_arrows=show_saccade_arrows,
7442 show_order=show_order,
7443 order_font_size=order_font_size,
7444 show_legend=show_legend,
7445 color_by=color_by,
7446 colorscale=fixation_colorscale,
7447 color_range=metric_range,
7448 show_colorbar=show_colorbars and idx == 0,
7449 colorbar_style=cb_style,
7450 fixation_symbol=fixation_symbol,
7451 fixation_hover_fields=fixation_hover_fields,
7452 row=row,
7453 col=col,
7454 trial_words=spec["trial_words"],
7455 **_comparison_filters(spec["style"], settings),
7456 duration_scale=_settings_size_scale(settings),
7457 category_colors=category_colors[idx],
7458 )
7460 if show_word_labels:
7461 _add_word_label_trace(
7462 fig,
7463 trial_words,
7464 _word_label_font_px(
7465 trial_words,
7466 scale=panel_scale,
7467 line_spacing=line_spacing,
7468 manual_font_px=base_font_size,
7469 scale_text_to_boxes=scale_text_to_boxes,
7470 ),
7471 font_settings["family"],
7472 row=row,
7473 col=col,
7474 highlight_column=highlight_column,
7475 text_color=text_color,
7476 highlight_text_color=highlight_text_color,
7477 word_hover_measure=word_hover_measure,
7478 word_hover_fields=word_hover_fields,
7479 )
7481 xaxis_key = "xaxis" if idx == 0 else f"xaxis{idx + 1}"
7482 yaxis_key = "yaxis" if idx == 0 else f"yaxis{idx + 1}"
7483 xaxis = dict(
7484 showticklabels=False,
7485 showgrid=False,
7486 zeroline=False,
7487 title=None,
7488 range=x_range,
7489 constrain="domain",
7490 )
7491 yaxis = dict(
7492 showticklabels=False,
7493 showgrid=False,
7494 zeroline=False,
7495 title=None,
7496 range=y_range,
7497 constrain="domain",
7498 scaleanchor=xref,
7499 scaleratio=1,
7500 )
7501 _apply_coordinate_grid_axes(
7502 xaxis,
7503 yaxis,
7504 show=show_coordinate_grid,
7505 spacing=coordinate_grid_spacing,
7506 x_range=x_range,
7507 y_range=y_range,
7508 rendered_width=panel_display_w,
7509 rendered_height=panel_display_h,
7510 )
7511 fig.update_layout(**{xaxis_key: xaxis, yaxis_key: yaxis})
7513 if show_heatmap and show_heatmap_colorbar and any(heatmap_maps):
7514 for trace in _comparison_heatmap_colorbar_traces(
7515 trial_specs,
7516 z_min=heatmap_min,
7517 z_max=heatmap_max,
7518 title=heatmap_title,
7519 heatmap_norm=heatmap_norm,
7520 colorbar_style=heat_cb_style,
7521 overlay=False,
7522 ):
7523 fig.add_trace(trace)
7524 _add_category_legend(fig, category_legend, category_label or "")
7526 fig.update_layout(
7527 height=figure_height,
7528 width=figure_width,
7529 autosize=False,
7530 margin=dict(l=grid_left, r=right_px, t=top_px, b=bottom_px + grid_bottom),
7531 legend=dict(
7532 orientation="h",
7533 yanchor="bottom",
7534 y=1.05,
7535 xanchor="right",
7536 x=1,
7537 font=_compare_legend_font(base_font_size, font_family),
7538 ),
7539 template="plotly_white",
7540 plot_bgcolor=background_color,
7541 paper_bgcolor=background_color,
7542 font=font_settings,
7543 shapes=all_shapes,
7544 )
7545 return fig
7548def _compare_stimulus_sides(value: str | None) -> tuple[bool, bool]:
7549 """``compare_stimulus`` → ``(draw A's stimulus, draw B's)`` (CMP-11).
7551 Tolerant of an unrecognised value on purpose: this reads a share-link param
7552 and a saved config, and drawing both sets of boxes is the honest fallback —
7553 it shows what is there rather than silently hiding one reading's AOIs.
7554 """
7555 normalized = str(value or "both").strip().lower()
7556 if normalized == "a":
7557 return True, False
7558 if normalized == "b":
7559 return False, True
7560 return True, True
7563def _render_comparison_figure(
7564 words: pd.DataFrame,
7565 fixations: pd.DataFrame,
7566 trial_a: tuple[str, str],
7567 trial_b: tuple[str, str],
7568 *,
7569 settings: FigureSettings,
7570 raw_gaze: pd.DataFrame | None = None,
7571) -> go.Figure:
7572 """Two scanpaths on one canvas — overlaid, side by side, or stacked.
7574 The shared settings contract keeps marker shape, highlighted text, stimulus
7575 image, and colorbar styling consistent across all three layouts (VIZ-23).
7576 """
7577 canvas_width = settings.canvas_width
7578 canvas_height = settings.canvas_height
7579 font_family = settings.font_family
7580 base_font_size = settings.base_font_size
7581 show_words = settings.show_words
7582 show_word_labels = settings.show_word_labels
7583 trial_labels = settings.trial_labels
7584 layout = settings.layout
7585 marker_size_range = settings.marker_size_range
7586 style_a = settings.style_a
7587 style_b = settings.style_b
7588 show_fixations = settings.show_fixations
7589 show_saccades = settings.show_saccades
7590 show_saccade_arrows = settings.show_saccade_arrows
7591 show_order = settings.show_order
7592 show_legend = settings.show_legend
7593 order_font_size = settings.order_font_size
7594 color_by = settings.color_by
7595 color_by_line = settings.color_by_line
7596 fixation_colorscale = settings.fixation_colorscale
7597 fixation_color_range = settings.fixation_color_range
7598 fixation_symbol = settings.fixation_symbol
7599 # The fixations' colour bar and the heatmap's, each with its own style.
7600 show_colorbars = settings.show_fixation_colorbar
7601 show_heatmap_colorbar = settings.show_heatmap_colorbar
7602 show_heatmap = settings.show_heatmap
7603 heatmap_metric = settings.heatmap_metric
7604 heatmap_range = settings.heatmap_range
7605 heatmap_norm = settings.heatmap_norm
7606 colorbar_orientation = settings.fixation_colorbar_orientation
7607 colorbar_tickangle = settings.fixation_colorbar_tickangle
7608 colorbar_tickfont_size = settings.fixation_colorbar_tickfont_size
7609 heat_cb_style = dict(
7610 orientation=settings.heatmap_colorbar_orientation,
7611 tickangle=settings.heatmap_colorbar_tickangle,
7612 tickfont_size=settings.heatmap_colorbar_tickfont_size,
7613 )
7614 text_color = settings.text_color
7615 highlight_column = settings.highlight_column
7616 highlight_text_color = settings.highlight_text_color
7617 word_hover_measure = settings.word_hover_measure
7618 word_hover_fields = settings.word_hover_fields
7619 fixation_hover_fields = settings.fixation_hover_fields
7620 background_color = settings.background_color
7621 line_spacing = settings.line_spacing
7622 scale_text_to_boxes = settings.scale_text_to_boxes
7623 background_image = settings.background_image
7624 background_image_size = settings.background_image_size
7625 background_image_origin = settings.background_image_origin
7626 background_image_opacity = settings.background_image_opacity
7627 fit_to_monitor = settings.fit_to_monitor
7628 show_coordinate_grid = settings.show_coordinate_grid
7629 coordinate_grid_spacing = settings.coordinate_grid_spacing
7630 if layout in {"side_by_side", "stacked"}:
7631 return _make_split_comparison_figure(
7632 words,
7633 fixations,
7634 trial_a,
7635 trial_b,
7636 settings=settings,
7637 orientation=layout,
7638 styles=(style_a, style_b),
7639 raw_gaze=raw_gaze,
7640 )
7642 # Shared metric colour range across BOTH trials (when colouring by a numeric
7643 # metric and the user didn't pin a range), so the two scanpaths use one
7644 # comparable scale.
7645 metric_range = fixation_color_range
7646 if (
7647 color_by
7648 and color_by != "line"
7649 and metric_range is None
7650 and color_by in fixations.columns
7651 and pd.api.types.is_numeric_dtype(fixations[color_by])
7652 ):
7653 both = fixations[
7654 (
7655 (fixations["participant_id"] == trial_a[0])
7656 & (fixations["trial_id"] == trial_a[1])
7657 )
7658 | (
7659 (fixations["participant_id"] == trial_b[0])
7660 & (fixations["trial_id"] == trial_b[1])
7661 )
7662 ][color_by]
7663 if len(both) and pd.notna(both.min()) and pd.notna(both.max()):
7664 metric_range = (float(both.min()), float(both.max()))
7666 fig = go.Figure()
7667 font_settings = dict(family=font_family or FONT_FAMILY, size=base_font_size)
7668 cb_style = dict(
7669 orientation=colorbar_orientation,
7670 tickangle=colorbar_tickangle,
7671 tickfont_size=colorbar_tickfont_size,
7672 )
7673 overrides = (style_a, style_b)
7674 # Stimulus-page background image (VIZ-4/23), UNDER both scanpaths.
7675 _add_background_image(
7676 fig,
7677 background_image,
7678 background_image_size,
7679 background_image_origin,
7680 background_image_opacity,
7681 )
7683 trial_specs = []
7684 for idx, trial in enumerate([trial_a, trial_b]):
7685 participant, trial_id = trial
7686 trial_words = words[
7687 (words["participant_id"] == participant) & (words["trial_id"] == trial_id)
7688 ]
7689 trial_fix = fixations[
7690 (fixations["participant_id"] == participant)
7691 & (fixations["trial_id"] == trial_id)
7692 ].sort_values("timestamp_ms")
7693 display_name = _plotly_literal(
7694 _resolve_trial_display_name(
7695 participant, trial_id, trial_words, trial_labels, idx
7696 )
7697 )
7698 style = _comparison_scanpath_style(
7699 idx, overrides[idx], default_marker_size_range=marker_size_range
7700 )
7701 trial_specs.append(
7702 dict(
7703 trial_words=trial_words,
7704 trial_fix=trial_fix,
7705 raw_gaze=_comparison_raw_gaze(
7706 raw_gaze, trial, show=settings.show_raw_gaze
7707 ),
7708 display_name=display_name,
7709 style=style,
7710 color=style["fix_color"],
7711 # The word-box outline: the scanpath's own colour unless its
7712 # style names one (`box_color`).
7713 box_color=style.get("box_color") or style["fix_color"],
7714 # Its fill: the figure's unless the style names one.
7715 box_fill_color=(
7716 style.get("box_fill_color") or settings.word_box_fill_color
7717 ),
7718 # Its raw-gaze samples: the scanpath's own colour unless its
7719 # style names one (`raw_gaze_color`).
7720 raw_gaze_color=style.get("raw_gaze_color") or style["fix_color"],
7721 # Its heatmap's colour scale: the figure's unless the style
7722 # names one (`heatmap_colorscale`). The range stays shared.
7723 heatmap_colorscale=(
7724 style.get("heatmap_colorscale") or settings.heatmap_colorscale
7725 ),
7726 )
7727 )
7729 heatmap_maps, heatmap_min, heatmap_max, heatmap_title = (
7730 _comparison_word_heatmap_data(
7731 trial_specs,
7732 metric=heatmap_metric,
7733 heatmap_range=heatmap_range,
7734 heatmap_norm=heatmap_norm,
7735 )
7736 if show_heatmap
7737 else ([], 0.0, 1.0, "")
7738 )
7740 if show_heatmap:
7741 reference_words = next(
7742 (
7743 spec["trial_words"]
7744 for spec in trial_specs
7745 if not spec["trial_words"].empty
7746 ),
7747 pd.DataFrame(),
7748 )
7749 existing = list(fig.layout.shapes) if fig.layout.shapes else []
7750 for index, half in enumerate(("left", "right")):
7751 existing.extend(
7752 _comparison_heatmap_shapes(
7753 reference_words,
7754 heatmap_maps[index],
7755 heatmap_colorscale=trial_specs[index]["heatmap_colorscale"],
7756 heatmap_norm=heatmap_norm,
7757 z_min=heatmap_min,
7758 z_max=heatmap_max,
7759 half=half,
7760 )
7761 )
7762 fig.update_layout(shapes=existing)
7763 if show_heatmap_colorbar and any(heatmap_maps):
7764 for trace in _comparison_heatmap_colorbar_traces(
7765 trial_specs,
7766 z_min=heatmap_min,
7767 z_max=heatmap_max,
7768 title=heatmap_title,
7769 heatmap_norm=heatmap_norm,
7770 colorbar_style=heat_cb_style,
7771 overlay=True,
7772 ):
7773 fig.add_trace(trace)
7775 x_range, y_range, *_ = _compute_axis_ranges(
7776 canvas_width,
7777 canvas_height,
7778 *((spec["trial_fix"], "x", "y") for spec in trial_specs),
7779 *((spec["raw_gaze"], "x", "y") for spec in trial_specs),
7780 word_frames=[
7781 spec["trial_words"] for spec in trial_specs if not spec["trial_words"].empty
7782 ],
7783 fit_to_monitor=fit_to_monitor,
7784 )
7786 # Both trials are overlaid on one shared canvas, so one display scale sizes
7787 # every word label true-to-scale (geometry is identical across the readings).
7788 fitted_w, fitted_h = _fit_display_size(
7789 canvas_width, canvas_height, x_range, y_range, spatial_axes=True
7790 )
7791 overlay_scale = _display_scale(x_range, y_range, fitted_w, fitted_h)
7793 # A categorical column or colour-by-line: one category→colour mapping for
7794 # both readings (see the split layouts).
7795 category_colors, category_legend = _shared_category_colors(
7796 [
7797 _fixation_category_labels(
7798 spec["trial_fix"], spec["trial_words"], color_by, color_by_line
7799 )
7800 for spec in trial_specs
7801 ],
7802 avoid=[spec["color"] for spec in trial_specs],
7803 )
7804 category_label = "line" if (color_by_line or color_by == "line") else color_by
7805 legend_on = show_legend or bool(category_legend)
7807 draws_stimulus = _compare_stimulus_sides(settings.compare_stimulus)
7808 # Both readings' samples before either scanpath, so neither cloud covers the
7809 # other reading's fixations.
7810 for spec in trial_specs:
7811 _add_comparison_raw_gaze_trace(
7812 fig,
7813 spec["raw_gaze"],
7814 spec["display_name"],
7815 spec["raw_gaze_color"],
7816 settings,
7817 )
7818 for _idx, spec in enumerate(trial_specs):
7819 _add_comparison_fixation_trace(
7820 fig,
7821 spec["trial_fix"],
7822 spec["display_name"],
7823 spec["style"],
7824 font_settings,
7825 show_fixations=show_fixations,
7826 show_saccades=show_saccades,
7827 show_saccade_arrows=show_saccade_arrows,
7828 show_order=show_order,
7829 order_font_size=order_font_size,
7830 show_legend=show_legend,
7831 color_by=color_by,
7832 colorscale=fixation_colorscale,
7833 color_range=metric_range,
7834 # One shared colorbar (on the first scanpath only) for the metric.
7835 show_colorbar=show_colorbars and _idx == 0,
7836 colorbar_style=cb_style,
7837 fixation_symbol=fixation_symbol,
7838 fixation_hover_fields=fixation_hover_fields,
7839 trial_words=spec["trial_words"],
7840 **_comparison_filters(spec["style"], settings),
7841 duration_scale=_settings_size_scale(settings),
7842 category_colors=category_colors[_idx],
7843 )
7844 if show_words and draws_stimulus[_idx]:
7845 existing = list(fig.layout.shapes) if fig.layout.shapes else []
7846 fig.update_layout(
7847 shapes=existing
7848 + build_word_boxes(
7849 spec["trial_words"],
7850 color=spec["box_color"],
7851 fill_color=spec["box_fill_color"],
7852 fill_opacity=settings.word_box_fill_opacity,
7853 line_opacity=settings.word_box_line_opacity,
7854 )
7855 )
7856 if show_word_labels and draws_stimulus[_idx]:
7857 _add_word_label_trace(
7858 fig,
7859 spec["trial_words"],
7860 _word_label_font_px(
7861 spec["trial_words"],
7862 scale=overlay_scale,
7863 line_spacing=line_spacing,
7864 manual_font_px=base_font_size,
7865 scale_text_to_boxes=scale_text_to_boxes,
7866 ),
7867 font_settings["family"],
7868 highlight_column=highlight_column,
7869 text_color=text_color,
7870 highlight_text_color=highlight_text_color,
7871 word_hover_measure=word_hover_measure,
7872 word_hover_fields=word_hover_fields,
7873 )
7875 _add_category_legend(fig, category_legend, category_label or "")
7877 shapes = list(fig.layout.shapes) if fig.layout.shapes else []
7878 shapes.append(
7879 dict(
7880 type="rect",
7881 x0=x_range[0],
7882 y0=y_range[1],
7883 x1=x_range[1],
7884 y1=y_range[0],
7885 line=dict(color="#000000", width=1),
7886 fillcolor="rgba(0,0,0,0)",
7887 )
7888 )
7890 # fitted_w / fitted_h were computed up front (so the label scale matched).
7891 # The title + top A/B legend get reserved space above the plot so they don't
7892 # shrink the equal-aspect plot region (same fix as make_scanpath_figure). With
7893 # the legend hidden (CMP-2 default) a slimmer band still fits the title.
7894 # The top band is now only needed for the optional A/B legend (the "Overlay
7895 # comparison" title was removed); reclaim it fully when the legend is hidden.
7896 top_px = _OVERLAY_TOP_PX if legend_on else 0
7897 # A horizontal colorbar (VIZ-23) sits below the plot, so reserve a band for
7898 # it; a vertical one keeps today's layout (it hangs off the right edge).
7899 bottom_px = (
7900 _COLORBAR_BOTTOM_PX
7901 if _colorbar_reserves(
7902 (
7903 _comparison_metric_colorbar(fixations, color_by, show_colorbars),
7904 colorbar_orientation,
7905 ),
7906 (
7907 bool(show_heatmap and show_heatmap_colorbar and any(heatmap_maps)),
7908 heat_cb_style["orientation"],
7909 ),
7910 )["colorbar_below"]
7911 else 0
7912 )
7913 grid_left = _GRID_LEFT_RESERVE_PX if show_coordinate_grid else 0
7914 grid_bottom = _GRID_BOTTOM_RESERVE_PX if show_coordinate_grid else 0
7915 xaxis = dict(
7916 showticklabels=False,
7917 showgrid=False,
7918 zeroline=False,
7919 title=None,
7920 range=x_range,
7921 constrain="domain",
7922 automargin=False,
7923 )
7924 yaxis = dict(
7925 showticklabels=False,
7926 showgrid=False,
7927 zeroline=False,
7928 title=None,
7929 range=y_range,
7930 constrain="domain",
7931 scaleanchor="x",
7932 scaleratio=1,
7933 automargin=False,
7934 )
7935 _apply_coordinate_grid_axes(
7936 xaxis,
7937 yaxis,
7938 show=show_coordinate_grid,
7939 spacing=coordinate_grid_spacing,
7940 x_range=x_range,
7941 y_range=y_range,
7942 rendered_width=fitted_w,
7943 rendered_height=fitted_h,
7944 )
7945 fig.update_layout(
7946 height=fitted_h + top_px + bottom_px + grid_bottom,
7947 width=fitted_w + grid_left,
7948 autosize=False,
7949 showlegend=legend_on,
7950 margin=dict(l=grid_left, r=0, t=top_px, b=bottom_px + grid_bottom),
7951 xaxis=xaxis,
7952 yaxis=yaxis,
7953 legend=dict(
7954 orientation="h",
7955 yanchor="bottom",
7956 y=1.02,
7957 xanchor="right",
7958 x=1,
7959 font=_compare_legend_font(base_font_size, font_family),
7960 ),
7961 template="plotly_white",
7962 plot_bgcolor=background_color,
7963 paper_bgcolor=background_color,
7964 font=font_settings,
7965 shapes=shapes,
7966 )
7967 return fig
7970# =============================================================================
7971# Line figures: metric convergence, trial-index trend
7972# =============================================================================
7975def make_metric_convergence_figure(
7976 series: dict,
7977 *,
7978 x_title: str,
7979 y_title: str,
7980 title: str,
7981 canvas_width: int,
7982 base_font_size: int,
7983 font_family: str,
7984 height: int = 340,
7985 y_range: tuple[float, float] = (0.0, 1.02),
7986 highlight_x_range: tuple[float, float] | None = None,
7987) -> go.Figure:
7988 """Line chart of a metric (one line per model) over a cumulative x axis.
7990 ``series`` maps a model name to ``(xs, ys)``. Used by the Multiple Comparison
7991 tab to show how each model's NLD vs. the real scanpath evolves as more of the
7992 reading is included — either by cumulative fixation index or by elapsed time.
7993 Optionally shades ``highlight_x_range`` (e.g. the selected fixation window).
7994 """
7995 fig = go.Figure()
7996 font_settings = dict(family=font_family or FONT_FAMILY, size=base_font_size)
7997 has_data = False
7998 for i, (name, xy) in enumerate(series.items()):
7999 xs, ys = xy
8000 if not len(xs):
8001 continue
8002 has_data = True
8003 color = _QUALITATIVE_PALETTE[i % len(_QUALITATIVE_PALETTE)]
8004 fig.add_trace(
8005 go.Scatter(
8006 x=list(xs),
8007 y=list(ys),
8008 mode="lines+markers",
8009 name=str(name),
8010 line=dict(color=color, width=2),
8011 marker=dict(size=4, color=color),
8012 hovertemplate=(
8013 f"{name}<br>{x_title}: %{{x}}<br>{y_title}: %{{y:.3f}}"
8014 "<extra></extra>"
8015 ),
8016 )
8017 )
8018 if has_data and highlight_x_range is not None:
8019 lo, hi = highlight_x_range
8020 if hi > lo:
8021 fig.add_vrect(x0=lo, x1=hi, fillcolor="#6c757d", opacity=0.10, line_width=0)
8022 fig.update_layout(
8023 height=height,
8024 width=canvas_width,
8025 autosize=False,
8026 margin=dict(l=55, r=10, t=40, b=45),
8027 template="plotly_white",
8028 font=font_settings,
8029 xaxis=dict(title=x_title),
8030 yaxis=dict(title=y_title, range=list(y_range)),
8031 legend=dict(orientation="h", yanchor="bottom", y=1.02, xanchor="right", x=1),
8032 title=title,
8033 )
8034 if not has_data:
8035 fig.add_annotation(
8036 text="No data", showarrow=False, x=0.5, y=0.5, xref="paper", yref="paper"
8037 )
8038 return fig
8041def gap_runs(xs) -> list[slice]:
8042 """Split ``xs`` (sorted) into runs with no gap: a new run starts where two
8043 whole-number ``xs`` are more than 1 apart — a trial the filters left out
8044 (#374 F35). Non-integer ``xs`` are one run."""
8045 xs = list(xs)
8046 try:
8047 whole = all(float(x).is_integer() for x in xs)
8048 except (TypeError, ValueError):
8049 whole = False
8050 if not whole:
8051 return [slice(0, len(xs))]
8052 cuts = [i for i in range(1, len(xs)) if float(xs[i]) - float(xs[i - 1]) > 1]
8053 bounds = [0, *cuts, len(xs)]
8054 return [slice(a, b) for a, b in itertools.pairwise(bounds)]
8057def break_at_gaps(xs, ys) -> tuple[list, list]:
8058 """``xs``/``ys`` with a ``None`` at every gap (see :func:`gap_runs`), so a
8059 Plotly line stops there instead of joining across it."""
8060 xs, ys = list(xs), list(ys)
8061 out_x: list = []
8062 out_y: list = []
8063 for i, run in enumerate(gap_runs(xs)):
8064 if i:
8065 out_x.append(None)
8066 out_y.append(None)
8067 out_x += xs[run]
8068 out_y += ys[run]
8069 return out_x, out_y
8072def make_trend_figure(
8073 df: pd.DataFrame,
8074 *,
8075 x_col: str,
8076 y_label: str,
8077 title: str,
8078 canvas_width: int,
8079 base_font_size: int,
8080 font_family: str,
8081 height: int = 340,
8082 x_label: str | None = None,
8083 break_gaps: bool = False,
8084) -> go.Figure:
8085 """Line+marker trend of ``value`` vs ``x_col`` with a ±SEM shaded band.
8087 ``df`` has columns ``[x_col, "value", "sem"]`` (see
8088 ``aggregation.metric_by_trial_index``). Used by the Per reader and Groups
8089 subtabs for the trial-index trend. ``x_label`` titles the x axis — the
8090 caller's name for what ``x_col`` holds (AN-9's frame calls it ``x``, which
8091 is no title); without one, ``x_col`` humanized. ``break_gaps`` stops the
8092 line and band at a missing whole-number ``x`` (a filtered-out trial).
8093 """
8094 if x_label is None:
8095 x_label = x_col.replace("_", " ").capitalize()
8096 fig = go.Figure()
8097 font_settings = dict(family=font_family or FONT_FAMILY, size=base_font_size)
8098 if df is None or df.empty:
8099 fig.update_layout(
8100 template="plotly_white",
8101 font=font_settings,
8102 title=f"{title} (no data)",
8103 height=height,
8104 )
8105 return fig
8106 xs = df[x_col].to_numpy()
8107 ys = df["value"].to_numpy()
8108 sem = df["sem"].to_numpy() if "sem" in df.columns else np.zeros(len(xs))
8109 runs = gap_runs(xs) if break_gaps else [slice(0, len(xs))]
8110 # One closed band per unbroken run, `None`-separated.
8111 band_x: list = []
8112 band_y: list = []
8113 for i, run in enumerate(runs):
8114 if i:
8115 band_x.append(None)
8116 band_y.append(None)
8117 band_x += [*xs[run], *xs[run][::-1]]
8118 band_y += [*(ys[run] + sem[run]), *(ys[run] - sem[run])[::-1]]
8119 line_x, line_y = break_at_gaps(xs, ys) if break_gaps else (xs, ys)
8120 # ±SEM band (drawn first so the line sits on top).
8121 fig.add_trace(
8122 go.Scatter(
8123 x=band_x,
8124 y=band_y,
8125 fill="toself",
8126 fillcolor="rgba(31,119,180,0.15)",
8127 line=dict(width=0),
8128 hoverinfo="skip",
8129 showlegend=False,
8130 name="±SEM",
8131 )
8132 )
8133 fig.add_trace(
8134 go.Scatter(
8135 x=line_x,
8136 y=line_y,
8137 mode="lines+markers",
8138 line=dict(color=COMPARISON_PALETTE[0], width=2),
8139 marker=dict(size=5, color=COMPARISON_PALETTE[0]),
8140 name=y_label,
8141 hovertemplate=f"{x_label}: %{{x}}<br>{y_label}: %{{y:.1f}}<extra></extra>",
8142 )
8143 )
8144 fig.update_layout(
8145 height=height,
8146 width=canvas_width,
8147 autosize=False,
8148 margin=dict(l=60, r=10, t=40, b=45),
8149 template="plotly_white",
8150 font=font_settings,
8151 xaxis=dict(title=x_label),
8152 yaxis=dict(title=y_label),
8153 title=title,
8154 showlegend=False,
8155 )
8156 return fig
8159# =============================================================================
8160# Analysis section figures (AN-1 … AN-22)
8161# =============================================================================
8162#
8163# Builders for the question-oriented Corpus Analysis subtabs. Each takes a tidy
8164# frame from ``aggregation.py`` plus the usual ``canvas_width`` / ``base_font_size``
8165# / ``font_family`` and returns a ``go.Figure``. Empty input → a "(no data)"
8166# placeholder, matching ``make_trend_figure``.
8168_DIVERGING_COLORSCALE = "RdBu"
8171def _hex_to_rgba(color: str, alpha: float) -> str:
8172 """``#rrggbb`` → ``rgba(r,g,b,alpha)`` for translucent spread bands. Passes
8173 through non-hex colours (already ``rgb(...)`` / named) by wrapping opacity in
8174 is impossible, so it returns a sensible grey fallback for those."""
8175 c = str(color).lstrip("#")
8176 if len(c) == 6:
8177 try:
8178 r, g, b = (int(c[i : i + 2], 16) for i in (0, 2, 4))
8179 return f"rgba({r},{g},{b},{alpha})"
8180 except ValueError:
8181 pass
8182 return f"rgba(120,120,120,{alpha})"
8185def _no_data_figure(title: str, *, font_family: str, base_font_size: int, height=340):
8186 fig = go.Figure()
8187 fig.update_layout(
8188 template="plotly_white",
8189 font=dict(family=font_family or FONT_FAMILY, size=base_font_size),
8190 title=f"{title} (no data)",
8191 height=height,
8192 )
8193 return fig
8196def make_small_multiples_figure(
8197 per_reader: pd.DataFrame,
8198 *,
8199 measure_label: str,
8200 canvas_width: int,
8201 base_font_size: int,
8202 font_family: str,
8203 cohort: pd.DataFrame | None = None,
8204 aggregate: str = "mean",
8205 max_panels: int = 12,
8206 panel_height: int = 110,
8207) -> go.Figure:
8208 """Stacked per-reader word profiles — one panel per participant (AN-1).
8210 ``per_reader`` is tidy ``[participant_id, word_id, value]`` (see
8211 ``aggregation.per_reader_word_measure``); panels share the X (reading order).
8212 ``cohort`` (``[word_id, value]``) draws a faint cohort-mean overlay in each
8213 panel. Caps at ``max_panels`` readers and titles the overflow (no silent cut).
8214 """
8215 from plotly.subplots import make_subplots
8217 if per_reader is None or per_reader.empty:
8218 return _no_data_figure(
8219 f"{measure_label} per participant",
8220 font_family=font_family,
8221 base_font_size=base_font_size,
8222 )
8223 readers = list(pd.unique(per_reader["participant_id"]))
8224 n_total = len(readers)
8225 readers = readers[:max_panels]
8226 n = len(readers)
8227 # Each panel's title (the reader id) sits in the gap above it, so the gap
8228 # is sized in pixels from the font rather than as a fixed fraction of the
8229 # figure: a fraction shrinks with few panels and the title lands on the
8230 # panel above's lowest tick row.
8231 title_gap_px = round(base_font_size * 2.2) + 6
8232 plot_px = panel_height * n + title_gap_px * (n - 1)
8233 fig = make_subplots(
8234 rows=n,
8235 cols=1,
8236 shared_xaxes=True,
8237 vertical_spacing=title_gap_px / plot_px if n > 1 else 0.0,
8238 subplot_titles=[str(r) for r in readers],
8239 )
8240 cohort_xy = None
8241 if cohort is not None and not cohort.empty:
8242 c = cohort.sort_values("word_id")
8243 cohort_xy = (c["word_id"].to_numpy(), c["value"].to_numpy())
8244 for i, reader in enumerate(readers, start=1):
8245 sub = per_reader[per_reader["participant_id"] == reader].sort_values("word_id")
8246 if cohort_xy is not None:
8247 fig.add_trace(
8248 go.Scatter(
8249 x=cohort_xy[0],
8250 y=cohort_xy[1],
8251 mode="lines",
8252 line=dict(color="rgba(120,120,120,0.45)", width=1.2, dash="dot"),
8253 name=f"Cohort {aggregate}",
8254 showlegend=(i == 1),
8255 hoverinfo="skip",
8256 ),
8257 row=i,
8258 col=1,
8259 )
8260 fig.add_trace(
8261 go.Scatter(
8262 x=sub["word_id"].to_numpy(),
8263 y=sub["value"].to_numpy(),
8264 mode="lines+markers",
8265 line=dict(color=COMPARISON_PALETTE[0], width=1.5),
8266 marker=dict(size=3, color=COMPARISON_PALETTE[0]),
8267 name=str(reader),
8268 showlegend=False,
8269 customdata=sub["word_text"].to_numpy() if "word_text" in sub else None,
8270 hovertemplate=(
8271 "word %{x}"
8272 + (" %{customdata}" if "word_text" in sub else "")
8273 + f"<br>{measure_label}: %{{y:.3~g}}<extra></extra>"
8274 ),
8275 ),
8276 row=i,
8277 col=1,
8278 )
8279 title = f"{measure_label} per participant (word profile)"
8280 if n_total > n:
8281 title += f" — showing {n} of {n_total} participants"
8282 margin_top = 50 + title_gap_px
8283 margin_bottom = 40
8284 fig.update_layout(
8285 height=plot_px + margin_top + margin_bottom,
8286 width=canvas_width,
8287 autosize=False,
8288 margin=dict(l=55, r=10, t=margin_top, b=margin_bottom),
8289 template="plotly_white",
8290 font=dict(family=font_family or FONT_FAMILY, size=base_font_size),
8291 title=title,
8292 legend=dict(orientation="h", yanchor="bottom", y=1.0, xanchor="right", x=1),
8293 )
8294 fig.update_xaxes(title_text="Word (reading order)", row=n, col=1)
8295 return fig
8298def make_word_matrix_heatmap(
8299 df: pd.DataFrame,
8300 *,
8301 row_col: str,
8302 measure_label: str,
8303 canvas_width: int,
8304 base_font_size: int,
8305 font_family: str,
8306 value_col: str = "value",
8307 colorscale: str = DEFAULT_HEATMAP_COLORSCALE,
8308 row_order: Iterable | None = None,
8309 height: int | None = None,
8310 row_label: str | None = None,
8311) -> go.Figure:
8312 """Word × {reader|group} heatmap (AN-2, AN-22).
8314 ``df`` is long ``[row_col, word_id, value_col]``; rows become Y, ``word_id``
8315 X, ``value_col`` the color. ``row_order`` pins the row order (e.g. Group A
8316 above Group B). Bright columns = universally hard words; bright rows = a
8317 uniformly slow reader. ``row_label`` names the rows (the dataset's own name
8318 for ``row_col``, DATA-66); without one, ``row_col`` humanized.
8319 """
8320 if row_label is None:
8321 row_label = _humanize_column(row_col)
8322 if df is None or df.empty:
8323 return _no_data_figure(
8324 f"{measure_label} by {row_label} × word",
8325 font_family=font_family,
8326 base_font_size=base_font_size,
8327 )
8328 matrix = df.pivot_table(
8329 index=row_col, columns="word_id", values=value_col, aggfunc="mean"
8330 )
8331 if row_order is not None:
8332 keep = [r for r in row_order if r in matrix.index]
8333 matrix = matrix.reindex(keep)
8334 fig = go.Figure(
8335 go.Heatmap(
8336 z=matrix.to_numpy(),
8337 x=[int(c) if float(c).is_integer() else c for c in matrix.columns],
8338 y=[str(r) for r in matrix.index],
8339 colorscale=colorscale,
8340 colorbar=dict(title=measure_label),
8341 hovertemplate="word %{x}<br>%{y}<br>"
8342 + measure_label
8343 + ": %{z:.1f}<extra></extra>",
8344 )
8345 )
8346 n_rows = max(len(matrix.index), 1)
8347 fig.update_layout(
8348 height=height or min(900, max(220, 26 * n_rows + 120)),
8349 width=canvas_width,
8350 autosize=False,
8351 margin=dict(l=120, r=10, t=50, b=45),
8352 template="plotly_white",
8353 font=dict(family=font_family or FONT_FAMILY, size=base_font_size),
8354 title=f"{measure_label} — {row_label} × word",
8355 xaxis=dict(title="Word (reading order)"),
8356 yaxis=dict(title=row_label, autorange="reversed"),
8357 )
8358 return fig
8361def make_word_profile_figure(
8362 profiles: dict,
8363 *,
8364 measure_label: str,
8365 canvas_width: int,
8366 base_font_size: int,
8367 font_family: str,
8368 spread_label: str = "SD",
8369 colors: Sequence[str] | None = None,
8370 height: int = 380,
8371 aggregate: str = "mean",
8372) -> go.Figure:
8373 """Cohort word profile(s): mean line + shaded spread band (AN-3 / AN-15).
8375 ``profiles`` maps a label → ``[word_id, value, lo, hi]`` (see
8376 ``aggregation.cohort_word_profile``). One entry draws the "average reader of
8377 this text" with uncertainty; several overlay (e.g. two groups).
8378 """
8379 entries = [
8380 (str(k), v)
8381 for k, v in (profiles or {}).items()
8382 if v is not None and not v.empty
8383 ]
8384 if not entries:
8385 return _no_data_figure(
8386 f"{measure_label} word profile",
8387 font_family=font_family,
8388 base_font_size=base_font_size,
8389 height=height,
8390 )
8391 fig = go.Figure()
8392 single = len(entries) == 1
8393 for i, (label, prof) in enumerate(entries):
8394 prof = prof.sort_values("word_id")
8395 xs = prof["word_id"].to_numpy()
8396 ys = prof["value"].to_numpy()
8397 color_choices = tuple(colors or COMPARISON_PALETTE)
8398 color = color_choices[i % len(color_choices)]
8399 rgba = _hex_to_rgba(color, 0.15)
8400 if {"lo", "hi"} <= set(prof.columns):
8401 lo = prof["lo"].to_numpy()
8402 hi = prof["hi"].to_numpy()
8403 fig.add_trace(
8404 go.Scatter(
8405 x=np.concatenate([xs, xs[::-1]]),
8406 y=np.concatenate([hi, lo[::-1]]),
8407 fill="toself",
8408 fillcolor=rgba,
8409 line=dict(width=0),
8410 hoverinfo="skip",
8411 showlegend=False,
8412 name=f"{label} {spread_label}",
8413 )
8414 )
8415 fig.add_trace(
8416 go.Scatter(
8417 x=xs,
8418 y=ys,
8419 mode="lines+markers",
8420 line=dict(color=color, width=2),
8421 marker=dict(size=4, color=color),
8422 name=label,
8423 showlegend=not single,
8424 customdata=prof["word_text"].to_numpy()
8425 if "word_text" in prof
8426 else None,
8427 hovertemplate=(
8428 "word %{x}"
8429 + (" %{customdata}" if "word_text" in prof else "")
8430 + f"<br>{measure_label}: %{{y:.3~g}}<extra></extra>"
8431 ),
8432 )
8433 )
8434 fig.update_layout(
8435 height=height,
8436 width=canvas_width,
8437 autosize=False,
8438 margin=dict(l=60, r=10, t=45, b=45),
8439 template="plotly_white",
8440 font=dict(family=font_family or FONT_FAMILY, size=base_font_size),
8441 title=f"{measure_label} by word — cohort {aggregate}, {spread_label} band",
8442 xaxis=dict(title="Word (reading order)"),
8443 yaxis=dict(title=measure_label),
8444 showlegend=not single,
8445 legend=dict(orientation="h", yanchor="bottom", y=1.02, xanchor="right", x=1),
8446 )
8447 return fig
8450def make_feature_scatter_figure(
8451 df: pd.DataFrame,
8452 *,
8453 measure_label: str,
8454 feature_label: str,
8455 categorical: bool,
8456 canvas_width: int,
8457 base_font_size: int,
8458 font_family: str,
8459 height: int = 400,
8460) -> go.Figure:
8461 """Per-word measure vs a bundled linguistic feature (AN-5).
8463 Numeric feature → scatter + OLS trend line (with Pearson r in the title);
8464 categorical feature (POS) → one box per category.
8465 """
8466 if df is None or df.empty or "feature" not in df.columns:
8467 return _no_data_figure(
8468 f"{measure_label} vs {feature_label}",
8469 font_family=font_family,
8470 base_font_size=base_font_size,
8471 height=height,
8472 )
8473 font_settings = dict(family=font_family or FONT_FAMILY, size=base_font_size)
8474 fig = go.Figure()
8475 if categorical:
8476 cats = sorted(df["feature"].dropna().astype(str).unique())
8477 for i, cat in enumerate(cats):
8478 vals = df.loc[df["feature"].astype(str) == cat, "value"].dropna().to_numpy()
8479 if vals.size:
8480 fig.add_trace(
8481 go.Box(
8482 y=vals,
8483 name=cat,
8484 boxpoints="outliers",
8485 marker_color=_QUALITATIVE_PALETTE[
8486 i % len(_QUALITATIVE_PALETTE)
8487 ],
8488 )
8489 )
8490 fig.update_layout(
8491 xaxis=dict(title=feature_label),
8492 yaxis=dict(title=measure_label),
8493 showlegend=False,
8494 )
8495 title = f"{measure_label} by {feature_label}"
8496 else:
8497 x = pd.to_numeric(df["feature"], errors="coerce").to_numpy()
8498 y = pd.to_numeric(df["value"], errors="coerce").to_numpy()
8499 ok = ~(np.isnan(x) | np.isnan(y))
8500 x, y = x[ok], y[ok]
8501 fig.add_trace(
8502 go.Scatter(
8503 x=x,
8504 y=y,
8505 mode="markers",
8506 marker=dict(size=6, color=COMPARISON_PALETTE[0], opacity=0.6),
8507 name="words",
8508 customdata=df.loc[ok, "word_text"].to_numpy()
8509 if "word_text" in df
8510 else None,
8511 hovertemplate=(
8512 f"{feature_label}: %{{x:.2f}}<br>{measure_label}: %{{y:.1f}}"
8513 + ("<br>%{customdata}" if "word_text" in df else "")
8514 + "<extra></extra>"
8515 ),
8516 )
8517 )
8518 r_txt = ""
8519 if x.size >= 2 and np.std(x) > 0:
8520 slope, intercept = np.polyfit(x, y, 1)
8521 xs = np.array([x.min(), x.max()])
8522 fig.add_trace(
8523 go.Scatter(
8524 x=xs,
8525 y=slope * xs + intercept,
8526 mode="lines",
8527 line=dict(color=TRENDLINE_COLOR, width=2, dash="dash"),
8528 name="trend",
8529 hoverinfo="skip",
8530 )
8531 )
8532 r = float(np.corrcoef(x, y)[0, 1])
8533 r_txt = f" (r = {r:.2f}, n = {x.size})"
8534 fig.update_layout(
8535 xaxis=dict(title=feature_label),
8536 yaxis=dict(title=measure_label),
8537 showlegend=False,
8538 )
8539 title = f"{measure_label} vs {feature_label}{r_txt}"
8540 fig.update_layout(
8541 height=height,
8542 width=canvas_width,
8543 autosize=False,
8544 margin=dict(l=60, r=10, t=45, b=50),
8545 template="plotly_white",
8546 font=font_settings,
8547 title=title,
8548 )
8549 return fig
8552def make_word_rate_figure(
8553 df: pd.DataFrame,
8554 *,
8555 canvas_width: int,
8556 base_font_size: int,
8557 font_family: str,
8558 height: int = 360,
8559) -> go.Figure:
8560 """Skip / regression-in rate per word — lollipop bars (AN-6)."""
8561 if df is None or df.empty:
8562 return _no_data_figure(
8563 "Skip / regression-in rate per word",
8564 font_family=font_family,
8565 base_font_size=base_font_size,
8566 height=height,
8567 )
8568 df = df.sort_values("word_id")
8569 xs = df["word_id"].to_numpy()
8570 fig = go.Figure()
8571 series = [
8572 ("Skip rate", "skip_rate", COMPARISON_PALETTE[0]),
8573 ("Regression-in rate", "regression_in_rate", COMPARISON_PALETTE[1]),
8574 ]
8575 for name, col, color in series:
8576 if col not in df.columns:
8577 continue
8578 ys = pd.to_numeric(df[col], errors="coerce").to_numpy()
8579 fig.add_trace(
8580 go.Bar(
8581 x=xs,
8582 y=ys,
8583 name=name,
8584 marker_color=color,
8585 opacity=0.8,
8586 hovertemplate="word %{x}<br>" + name + ": %{y:.0%}<extra></extra>",
8587 )
8588 )
8589 fig.update_layout(
8590 height=height,
8591 width=canvas_width,
8592 autosize=False,
8593 margin=dict(l=55, r=10, t=45, b=45),
8594 template="plotly_white",
8595 font=dict(family=font_family or FONT_FAMILY, size=base_font_size),
8596 title="Skip / regression-in rate per word",
8597 xaxis=dict(title="Word (reading order)"),
8598 yaxis=dict(title="Rate", tickformat=".0%"),
8599 barmode="group",
8600 legend=dict(orientation="h", yanchor="bottom", y=1.02, xanchor="right", x=1),
8601 )
8602 return fig
8605def make_distribution_figure(
8606 groups: dict,
8607 *,
8608 metric_label: str,
8609 canvas_width: int,
8610 base_font_size: int,
8611 font_family: str,
8612 kind: str = "violin",
8613 colors: Sequence[str] | None = None,
8614 height: int = 380,
8615) -> go.Figure:
8616 """Overlaid metric distributions — one violin/box per group (AN-7/14/18)."""
8617 arrays = [
8618 (str(name), np.asarray(arr, dtype="float64"))
8619 for name, arr in (groups or {}).items()
8620 if arr is not None and len(arr)
8621 ]
8622 if not arrays:
8623 return _no_data_figure(
8624 f"{metric_label} distribution",
8625 font_family=font_family,
8626 base_font_size=base_font_size,
8627 height=height,
8628 )
8629 fig = go.Figure()
8630 for i, (name, arr) in enumerate(arrays):
8631 color_choices = tuple(colors or _QUALITATIVE_PALETTE)
8632 color = color_choices[i % len(color_choices)]
8633 if kind == "box":
8634 fig.add_trace(
8635 go.Box(
8636 y=arr,
8637 name=name,
8638 marker_color=color,
8639 boxmean=True,
8640 boxpoints="outliers",
8641 )
8642 )
8643 else:
8644 fig.add_trace(
8645 go.Violin(
8646 y=arr,
8647 name=name,
8648 line_color=color,
8649 opacity=0.7,
8650 box_visible=True,
8651 meanline_visible=True,
8652 points=False,
8653 )
8654 )
8655 fig.update_layout(
8656 height=height,
8657 width=canvas_width,
8658 autosize=False,
8659 margin=dict(l=60, r=10, t=45, b=40),
8660 template="plotly_white",
8661 font=dict(family=font_family or FONT_FAMILY, size=base_font_size),
8662 title=f"{metric_label} distribution",
8663 yaxis=dict(title=metric_label),
8664 showlegend=False,
8665 )
8666 return fig
8669def make_density_scatter_figure(
8670 df: pd.DataFrame,
8671 *,
8672 x_col: str,
8673 y_col: str,
8674 x_label: str,
8675 y_label: str,
8676 canvas_width: int,
8677 base_font_size: int,
8678 font_family: str,
8679 height: int = 420,
8680) -> go.Figure:
8681 """2D density of two per-fixation measures — the oculomotor scatter (AN-10)."""
8682 if df is None or df.empty or not {x_col, y_col} <= set(df.columns):
8683 return _no_data_figure(
8684 f"{y_label} vs {x_label}",
8685 font_family=font_family,
8686 base_font_size=base_font_size,
8687 height=height,
8688 )
8689 x = pd.to_numeric(df[x_col], errors="coerce").to_numpy()
8690 y = pd.to_numeric(df[y_col], errors="coerce").to_numpy()
8691 ok = ~(np.isnan(x) | np.isnan(y))
8692 x, y = x[ok], y[ok]
8693 fig = go.Figure(
8694 go.Histogram2d(
8695 x=x,
8696 y=y,
8697 colorscale=DEFAULT_HEATMAP_COLORSCALE,
8698 nbinsx=40,
8699 nbinsy=40,
8700 colorbar=dict(title="Fixations"),
8701 hovertemplate=f"{x_label}: %{{x}}<br>{y_label}: %{{y}}<br>count: %{{z}}<extra></extra>",
8702 )
8703 )
8704 fig.update_layout(
8705 height=height,
8706 width=canvas_width,
8707 autosize=False,
8708 margin=dict(l=60, r=10, t=45, b=50),
8709 template="plotly_white",
8710 font=dict(family=font_family or FONT_FAMILY, size=base_font_size),
8711 title=f"{y_label} vs {x_label} (n = {x.size})",
8712 xaxis=dict(title=x_label),
8713 yaxis=dict(title=y_label),
8714 )
8715 return fig
8718def make_progression_figure(
8719 df: pd.DataFrame,
8720 *,
8721 canvas_width: int,
8722 base_font_size: int,
8723 font_family: str,
8724 height: int = 380,
8725) -> go.Figure:
8726 """Progressive vs regressive saccade counts per trial + regression share (AN-11)."""
8727 from plotly.subplots import make_subplots
8729 if df is None or df.empty:
8730 return _no_data_figure(
8731 "Progressive vs regressive saccades",
8732 font_family=font_family,
8733 base_font_size=base_font_size,
8734 height=height,
8735 )
8736 df = df.copy()
8737 labels = [str(t) for t in df["trial_id"].to_numpy()]
8738 fig = make_subplots(specs=[[{"secondary_y": True}]])
8739 fig.add_trace(
8740 go.Bar(
8741 x=labels,
8742 y=df["progressive"].to_numpy(),
8743 name="Progressive",
8744 marker_color=COMPARISON_PALETTE[0],
8745 ),
8746 secondary_y=False,
8747 )
8748 fig.add_trace(
8749 go.Bar(
8750 x=labels,
8751 y=df["regressive"].to_numpy(),
8752 name="Regressive",
8753 marker_color=COMPARISON_PALETTE[1],
8754 ),
8755 secondary_y=False,
8756 )
8757 if "regression_share" in df.columns:
8758 fig.add_trace(
8759 go.Scatter(
8760 x=labels,
8761 y=df["regression_share"].to_numpy(),
8762 name="Regression share",
8763 mode="lines+markers",
8764 line=dict(color="#555", width=2),
8765 marker=dict(size=5),
8766 ),
8767 secondary_y=True,
8768 )
8769 fig.update_layout(
8770 height=height,
8771 width=canvas_width,
8772 autosize=False,
8773 margin=dict(l=55, r=55, t=45, b=80),
8774 template="plotly_white",
8775 font=dict(family=font_family or FONT_FAMILY, size=base_font_size),
8776 title="Progressive vs regressive saccades per trial",
8777 barmode="stack",
8778 legend=dict(orientation="h", yanchor="bottom", y=1.02, xanchor="right", x=1),
8779 )
8780 fig.update_xaxes(title_text="Trial", tickangle=-40)
8781 fig.update_yaxes(title_text="Saccade count", secondary_y=False)
8782 fig.update_yaxes(
8783 title_text="Regression share", tickformat=".0%", secondary_y=True, range=[0, 1]
8784 )
8785 return fig
8788def make_paired_bars_figure(
8789 df: pd.DataFrame,
8790 *,
8791 canvas_width: int,
8792 base_font_size: int,
8793 font_family: str,
8794 height: int = 380,
8795 aggregate: str = "mean",
8796) -> go.Figure:
8797 """Side-by-side group bars per measure (AN-20), no error bars (#374).
8799 ``df`` is ``[measure, group, value, …]`` (see
8800 ``aggregation.paired_group_summary``); ``aggregate`` names what ``value``
8801 is, for the title. One subplot per measure so differing units keep their
8802 own scale.
8803 """
8804 from plotly.subplots import make_subplots
8806 if df is None or df.empty:
8807 return _no_data_figure(
8808 f"Group {aggregate}s",
8809 font_family=font_family,
8810 base_font_size=base_font_size,
8811 height=height,
8812 )
8813 measures = list(dict.fromkeys(df["measure"]))
8814 groups = list(dict.fromkeys(df["group"]))
8815 fig = make_subplots(
8816 rows=1, cols=len(measures), subplot_titles=measures, horizontal_spacing=0.08
8817 )
8818 for gi, group in enumerate(groups):
8819 color = COMPARISON_PALETTE[gi % len(COMPARISON_PALETTE)]
8820 for mi, measure in enumerate(measures, start=1):
8821 sub = df[(df["measure"] == measure) & (df["group"] == group)]
8822 if sub.empty:
8823 continue
8824 row = sub.iloc[0]
8825 fig.add_trace(
8826 go.Bar(
8827 x=[group],
8828 y=[row["value"]],
8829 name=group,
8830 marker_color=color,
8831 legendgroup=group,
8832 showlegend=(mi == 1),
8833 hovertemplate=f"{group}<br>{measure}: %{{y:.2f}}<extra></extra>",
8834 ),
8835 row=1,
8836 col=mi,
8837 )
8838 fig.update_layout(
8839 height=height,
8840 width=canvas_width,
8841 autosize=False,
8842 margin=dict(l=55, r=10, t=55, b=40),
8843 template="plotly_white",
8844 font=dict(family=font_family or FONT_FAMILY, size=base_font_size),
8845 title=f"Group {aggregate}s per measure",
8846 legend=dict(orientation="h", yanchor="bottom", y=1.04, xanchor="right", x=1),
8847 barmode="group",
8848 )
8849 return fig
8852def make_landing_curve_figure(
8853 values: np.ndarray,
8854 *,
8855 canvas_width: int,
8856 base_font_size: int,
8857 font_family: str,
8858 as_fraction: bool = True,
8859 height: int = 360,
8860) -> go.Figure:
8861 """Preferred-viewing-location curve — landing-position histogram (AN-12).
8863 ``values`` come from ``aggregation.landing_positions``: fractions of the
8864 experiment's word box, unclipped (BUG-83), so a landing assigned from beside
8865 the box shows as a bar outside 0–1 instead of a spike on the edge.
8866 """
8867 arr = np.asarray(values, dtype="float64")
8868 arr = arr[~np.isnan(arr)]
8869 if arr.size == 0:
8870 return _no_data_figure(
8871 "Landing position within words",
8872 font_family=font_family,
8873 base_font_size=base_font_size,
8874 height=height,
8875 )
8876 nbins = 20 if as_fraction else 30
8877 fig = go.Figure(
8878 go.Histogram(
8879 x=arr,
8880 nbinsx=nbins,
8881 marker_color=COMPARISON_PALETTE[0],
8882 marker_line=dict(color="white", width=0.4),
8883 hovertemplate="landing %{x}<br>count: %{y}<extra></extra>",
8884 )
8885 )
8886 x_title = (
8887 "Landing position within the word's interest area (0 = start, 1 = end)"
8888 if as_fraction
8889 else "Landing distance from word start (px)"
8890 )
8891 fig.update_layout(
8892 height=height,
8893 width=canvas_width,
8894 autosize=False,
8895 margin=dict(l=55, r=10, t=45, b=50),
8896 template="plotly_white",
8897 font=dict(family=font_family or FONT_FAMILY, size=base_font_size),
8898 title=f"Landing-position curve (n = {arr.size})",
8899 xaxis=dict(title=x_title),
8900 yaxis=dict(title="Count"),
8901 )
8902 return fig
8905def make_difference_profile_figure(
8906 df: pd.DataFrame,
8907 *,
8908 measure_label: str,
8909 label_a: str = "Group A",
8910 label_b: str = "Group B",
8911 canvas_width: int,
8912 base_font_size: int,
8913 font_family: str,
8914 colors: Sequence[str] | None = None,
8915 height: int = 380,
8916) -> go.Figure:
8917 """Per-word A−B difference profile, diverging color + zero line (AN-19)."""
8918 if df is None or df.empty or "diff" not in df.columns:
8919 return _no_data_figure(
8920 f"{measure_label} difference by word",
8921 font_family=font_family,
8922 base_font_size=base_font_size,
8923 height=height,
8924 )
8925 df = df.sort_values("word_id")
8926 xs = df["word_id"].to_numpy()
8927 diffs = pd.to_numeric(df["diff"], errors="coerce").to_numpy()
8928 vmax = np.nanmax(np.abs(diffs)) if np.isfinite(diffs).any() else 1.0
8929 vmax = vmax if vmax > 0 else 1.0
8930 difference_colors = tuple(colors or (COMPARISON_PALETTE[1], COMPARISON_PALETTE[0]))
8931 colorscale = [
8932 [0.0, difference_colors[1 % len(difference_colors)]],
8933 [0.5, "#f7f7f7"],
8934 [1.0, difference_colors[0]],
8935 ]
8936 fig = go.Figure(
8937 go.Bar(
8938 x=xs,
8939 y=diffs,
8940 marker=dict(
8941 color=diffs,
8942 colorscale=colorscale,
8943 cmin=-vmax,
8944 cmax=vmax,
8945 colorbar=dict(title=f"{label_a} − {label_b}"),
8946 ),
8947 customdata=df["word_text"].to_numpy() if "word_text" in df else None,
8948 hovertemplate=(
8949 "word %{x}"
8950 + (" %{customdata}" if "word_text" in df else "")
8951 + f"<br>Δ {measure_label}: %{{y:.3~g}}<extra></extra>"
8952 ),
8953 )
8954 )
8955 fig.add_hline(y=0, line=dict(color="#333", width=1))
8956 fig.update_layout(
8957 height=height,
8958 width=canvas_width,
8959 autosize=False,
8960 margin=dict(l=60, r=10, t=50, b=45),
8961 template="plotly_white",
8962 font=dict(family=font_family or FONT_FAMILY, size=base_font_size),
8963 title=f"{measure_label}: {label_a} − {label_b} by word",
8964 xaxis=dict(title="Word (reading order)"),
8965 yaxis=dict(title=f"Δ {measure_label}"),
8966 )
8967 return fig
8970def _setting_names(excluded: Iterable[str]) -> tuple[str, ...]:
8971 excluded = set(excluded)
8972 return tuple(
8973 field.name for field in fields(FigureSettings) if field.name not in excluded
8974 )
8977def _resolve_figure_settings(
8978 settings: FigureSettings | Mapping[str, Any] | None,
8979 overrides: Mapping[str, Any],
8980 *,
8981 legacy_defaults: Mapping[str, Any] | None = None,
8982) -> FigureSettings:
8983 """Resolve a settings object while preserving old builder-only defaults."""
8984 if settings is None and legacy_defaults:
8985 return FigureSettings.from_mapping({**legacy_defaults, **overrides})
8986 return FigureSettings.from_mapping(settings, **dict(overrides))
8989STATIC_FIGURE_OPTIONS = _setting_names(
8990 {
8991 "playback_speed",
8992 "label_a",
8993 "label_b",
8994 "show_legend",
8995 "autoplay",
8996 "anim_grid_step_ms",
8997 "anim_max_frames",
8998 "trial_labels",
8999 "layout",
9000 "style_a",
9001 "style_b",
9002 # CMP-8 §4 — B-side geometry, read only by the split comparison layouts.
9003 "canvas_b",
9004 "background_image_b",
9005 "background_image_size_b",
9006 "background_image_origin_b",
9007 # CMP-11 — the static builder draws one trial, so it has no "whose
9008 # stimulus?" question to answer. The *animation* builder does read it (a
9009 # dual co-animation takes `words_b`), so it is NOT excluded there.
9010 "compare_stimulus",
9011 # CMP-24 — B's flags in a co-animation; one trial has no B.
9012 "fixation_flags_b",
9013 # DATA-66 — names, not an option: read by `_labelled_columns`.
9014 "column_labels",
9015 }
9016)
9017#: What `make_comparison_figure` accepts (CMP-9). Only the animation-only fields
9018#: are excluded — this is a typo-catcher for `api.compare_scanpaths`, not a
9019#: semantic filter, so it still admits settings the comparison builders ignore.
9020#: Which settings actually reach a comparison figure is the table in
9021#: `scanpath_studio/CLAUDE.md` → *Which viz settings apply in which render path*.
9022COMPARISON_FIGURE_OPTIONS = _setting_names(
9023 {
9024 "playback_speed",
9025 "label_a",
9026 "label_b",
9027 "autoplay",
9028 "anim_grid_step_ms",
9029 "anim_max_frames",
9030 # CMP-24 — a comparison reads B's filters off `style_b` instead.
9031 "fixation_flags_b",
9032 # DATA-66 — names, not an option: read by `_labelled_columns`.
9033 "column_labels",
9034 }
9035)
9036ANIMATION_FIGURE_OPTIONS = _setting_names(
9037 {
9038 "x_field",
9039 "y_field",
9040 "show_fixations",
9041 "show_heatmap",
9042 "heatmap_metric",
9043 "heatmap_style",
9044 "heatmap_norm",
9045 "heatmap_sigma_px",
9046 "heatmap_range",
9047 "heatmap_colorscale",
9048 "show_raw_gaze",
9049 "critical_span_style",
9050 "saccade_color_mode",
9051 "saccade_class_colors",
9052 "saccade_type_legend",
9053 "saccade_classes",
9054 "saccade_render_mode",
9055 "fixation_snap_to_word",
9056 "span_border_color",
9057 "word_heatmap_col",
9058 "word_heatmap_title",
9059 "show_connectors",
9060 "connector_y",
9061 "illustration_reasons",
9062 "trial_labels",
9063 "layout",
9064 # CMP-8 §4 — B-side geometry, read only by the split comparison layouts.
9065 "canvas_b",
9066 "background_image_b",
9067 "background_image_size_b",
9068 "background_image_origin_b",
9069 # DATA-66 — names, not an option: read by `_labelled_columns`.
9070 "column_labels",
9071 }
9072)
9075def make_scanpath_figure(
9076 words: pd.DataFrame,
9077 fixations: pd.DataFrame,
9078 *,
9079 settings: FigureSettings | Mapping[str, Any] | None = None,
9080 raw_gaze: pd.DataFrame | None = None,
9081 **overrides: Any,
9082) -> go.Figure:
9083 """Build a static scanpath from one shared rendering-settings object.
9085 Keyword overrides remain useful for focused programmatic calls and tests;
9086 application code should pass ``settings=FigureSettings(...)`` so the same
9087 object can flow unchanged through UI, export, and headless surfaces.
9088 """
9089 resolved = _resolve_figure_settings(settings, overrides)
9090 fields_xy = (resolved.x_field, resolved.y_field)
9091 words = _finite_for_plotting(words, drop_on=_WORD_BOX_COLUMNS)
9092 fixations = _finite_for_plotting(fixations, fields_xy)
9093 raw_gaze = _finite_for_plotting(raw_gaze)
9094 with _labelled_columns(resolved.column_labels):
9095 fig = _render_scanpath_figure(
9096 words,
9097 fixations,
9098 settings=resolved,
9099 raw_gaze=raw_gaze,
9100 )
9101 _arrange_colorbars(fig)
9102 apply_legend_layout(fig, resolved.legend_layout)
9103 if resolved.show_fixations:
9104 _maybe_add_duration_key(fig, resolved, resolved.marker_size_range, fixations)
9105 return fig
9108def make_scanpath_animation(
9109 words: pd.DataFrame,
9110 fixations: pd.DataFrame,
9111 *,
9112 settings: FigureSettings | Mapping[str, Any] | None = None,
9113 fixations_b: pd.DataFrame | None = None,
9114 words_b: pd.DataFrame | None = None,
9115 **overrides: Any,
9116) -> go.Figure:
9117 """Build an animated replay from the shared rendering settings."""
9118 fig, _frame_step_ms = build_scanpath_replay(
9119 words,
9120 fixations,
9121 settings=settings,
9122 fixations_b=fixations_b,
9123 words_b=words_b,
9124 **overrides,
9125 )
9126 return fig
9129def build_scanpath_replay(
9130 words: pd.DataFrame,
9131 fixations: pd.DataFrame,
9132 *,
9133 settings: FigureSettings | Mapping[str, Any] | None = None,
9134 fixations_b: pd.DataFrame | None = None,
9135 words_b: pd.DataFrame | None = None,
9136 **overrides: Any,
9137) -> tuple[go.Figure, float]:
9138 """:func:`make_scanpath_animation`, returning the grid step with the figure.
9140 ``(figure, frame_step_ms)``: pass the step to :func:`set_replay_clock` to
9141 re-time the replay at another speed or autoplay without rebuilding a frame
9142 (PERF-15)."""
9143 resolved = _resolve_figure_settings(
9144 settings,
9145 overrides,
9146 legacy_defaults={
9147 "order_font_color": "#000000",
9148 "color_by": None,
9149 "fixation_color": None,
9150 "highlight_column": None,
9151 "word_hover_measure": None,
9152 },
9153 )
9154 words = _finite_for_plotting(words, drop_on=_WORD_BOX_COLUMNS)
9155 words_b = _finite_for_plotting(words_b, drop_on=_WORD_BOX_COLUMNS)
9156 fixations = _finite_for_plotting(fixations)
9157 fixations_b = _finite_for_plotting(fixations_b)
9158 with _labelled_columns(resolved.column_labels):
9159 fig, frame_step_ms = _render_scanpath_animation(
9160 words,
9161 fixations,
9162 settings=resolved,
9163 fixations_b=fixations_b,
9164 words_b=words_b,
9165 )
9166 apply_legend_layout(
9167 fig,
9168 resolved.legend_layout,
9169 comparing=fixations_b is not None and not fixations_b.empty,
9170 )
9171 size_range = replay_size_key_range(resolved, fixations, fixations_b)
9172 if size_range is not None:
9173 _maybe_add_duration_key(fig, resolved, size_range, fixations, fixations_b)
9174 return fig, frame_step_ms
9177def replay_size_key_range(
9178 settings: FigureSettings,
9179 fixations: pd.DataFrame | None,
9180 fixations_b: pd.DataFrame | None = None,
9181) -> tuple[int, int] | None:
9182 """The size range a replay's duration key draws, or ``None`` for no key.
9184 A lone replay's is the figure's ``marker_size_range``. A co-animation sizes
9185 each scanpath in its own style's range, and one key serves both only while
9186 those agree — as on the static comparison."""
9187 dual = all(f is not None and not f.empty for f in (fixations, fixations_b))
9188 if not dual:
9189 return tuple(settings.marker_size_range)
9190 ranges = {
9191 tuple(
9192 _comparison_scanpath_style(
9193 idx, style, default_marker_size_range=settings.marker_size_range
9194 )["marker_size_range"]
9195 )
9196 for idx, style in enumerate((settings.style_a, settings.style_b))
9197 }
9198 return ranges.pop() if len(ranges) == 1 else None
9201def _require_one_screen_per_reading(
9202 words: pd.DataFrame | None,
9203 fixations: pd.DataFrame | None,
9204 readings: Sequence[tuple[str, str]],
9205) -> None:
9206 """Refuse a comparison reading that spans several screens.
9208 Every screen of a multipart trial is its own coordinate space, so one
9209 scanpath drawn from two of them joins its last fixation on one page to the
9210 first on the next — a saccade nobody made. Callers cut each reading to one
9211 screen first (`multipart.extract_part`); this is the guard that keeps a new
9212 caller from forgetting to.
9213 """
9214 for label, (participant, trial) in zip(("A", "B"), readings, strict=False):
9215 for frame in (fixations, words):
9216 if frame is None or frame.empty or SCREEN_ID not in frame.columns:
9217 continue
9218 rows = frame[
9219 (frame["participant_id"] == participant) & (frame["trial_id"] == trial)
9220 ]
9221 screens = rows[SCREEN_ID].dropna().astype(str).unique()
9222 if len(screens) > 1:
9223 shown = ", ".join(repr(value) for value in screens[:5])
9224 raise ValueError(
9225 f"Scanpath {label} (participant={participant!r}, "
9226 f"trial={trial!r}) spans {len(screens)} screens ({shown}). "
9227 "Each screen is its own coordinate space, so a comparison "
9228 "draws one screen per scanpath: cut each trial to one "
9229 "screen first (multipart.extract_part, or "
9230 "compare_scanpaths' screen= / screen_b=)."
9231 )
9234def make_comparison_figure(
9235 words: pd.DataFrame,
9236 fixations: pd.DataFrame,
9237 trial_a: tuple[str, str],
9238 trial_b: tuple[str, str],
9239 *,
9240 settings: FigureSettings | Mapping[str, Any] | None = None,
9241 raw_gaze: pd.DataFrame | None = None,
9242 **overrides: Any,
9243) -> go.Figure:
9244 """Build a two-scanpath comparison from the shared rendering settings.
9246 ``raw_gaze`` (VIZ-48) holds either reading's samples, keyed like ``words``
9247 and ``fixations``; with ``show_raw_gaze`` each reading's are drawn under its
9248 scanpath, in its colour.
9250 Each scanpath must be one screen: a frame holding several screens of one
9251 multipart reading raises ``ValueError`` rather than pooling coordinate
9252 spaces (and drawing saccades across page boundaries)."""
9253 _require_one_screen_per_reading(words, fixations, (trial_a, trial_b))
9254 resolved = _resolve_figure_settings(
9255 settings,
9256 overrides,
9257 legacy_defaults={
9258 "show_word_labels": False,
9259 "show_order": False,
9260 "order_font_size": None,
9261 "color_by": None,
9262 "highlight_column": None,
9263 "heatmap_metric": "duration_ms",
9264 },
9265 )
9266 words = _finite_for_plotting(words, drop_on=_WORD_BOX_COLUMNS)
9267 fixations = _finite_for_plotting(fixations)
9268 raw_gaze = _finite_for_plotting(raw_gaze)
9269 with _labelled_columns(resolved.column_labels):
9270 fig = _render_comparison_figure(
9271 words,
9272 fixations,
9273 trial_a,
9274 trial_b,
9275 settings=resolved,
9276 raw_gaze=raw_gaze,
9277 )
9278 _arrange_colorbars(fig)
9279 apply_legend_layout(fig, resolved.legend_layout, comparing=True)
9280 # One key serves both scanpaths only while they share a size range; with
9281 # per-scanpath ranges (Compare's own Size) one duration is two sizes.
9282 ranges = {
9283 tuple(
9284 _comparison_scanpath_style(
9285 idx, style, default_marker_size_range=resolved.marker_size_range
9286 )["marker_size_range"]
9287 )
9288 for idx, style in enumerate((resolved.style_a, resolved.style_b))
9289 }
9290 if resolved.show_fixations and len(ranges) == 1:
9291 _maybe_add_duration_key(fig, resolved, ranges.pop(), fixations)
9292 return fig
9295def _maybe_add_duration_key(
9296 fig: go.Figure,
9297 settings: FigureSettings,
9298 size_range: tuple[int, int],
9299 *fixation_frames: pd.DataFrame | None,
9300) -> None:
9301 """Add the duration-size key when the figure draws fixation markers on a
9302 fixed scale and the caller asked for it (``duration_size_legend``)."""
9303 if not settings.duration_size_legend or settings.marker_size_scale == "relative":
9304 return
9305 if not any(f is not None and not f.empty for f in fixation_frames):
9306 return
9307 _add_duration_size_key(
9308 fig,
9309 size_range,
9310 settings.marker_size_scale,
9311 settings.marker_duration_range,
9312 font_family=settings.font_family,
9313 legend_layout=settings.legend_layout,
9314 )