Coverage for scanpath_studio/constants.py: 99%
214 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 21:10 +0000
1"""Shared constants for the Scanpath Studio app."""
3from __future__ import annotations
5import os
6import re
8PACKAGE_NAME = "scanpath_studio"
10#: On raw gaze imported with no clock: each sample's position in its trial, 1,
11#: 2, … — in place of `timestamp_ms`, which such a table does not have
12#: (`data.normalize_raw_gaze`). A column the user sees and exports, the sample
13#: number, so the plot can colour by order without calling it time.
14SAMPLE_INDEX = "sample_index"
17# --- PRE-21: features that are built but not fully integrated ----------------
18# Vertical drift correction (the PRE-3 port of Carr et al. 2021) and the NLD
19# similarity scoring under 🔬 Comparisons both work, and both are half-wired:
20# the similarity table still shows three metrics as "Not yet computed", and the
21# drift-correction subtab's follow-ups (PRE-9, PRE-10) are unfinished. Shipping
22# them visible invites users to lean on them, so ahead of publication they are
23# gated off — a **visibility gate, not a removal**: `alignment.py` and
24# `similarity.py` stay exactly where they are and stay reachable for us.
25#
26# Polarity is the opposite of `app.public_datasets_enabled`: these default
27# **off** and the env var turns them on. Read at *call time*, which is both what
28# lets a test toggle them and what stops a stale import-time read from making
29# one surface disagree with another.
30EXPERIMENTAL_ENV_VAR = "SCANPATH_EXPERIMENTAL"
33def experimental_features_enabled() -> bool:
34 """Whether the not-fully-integrated features are exposed (PRE-21).
36 Off unless ``SCANPATH_EXPERIMENTAL`` is set to something truthy.
37 """
38 return os.environ.get(EXPERIMENTAL_ENV_VAR, "").strip().lower() in (
39 "1",
40 "true",
41 "yes",
42 "on",
43 )
46def drift_correction_enabled() -> bool:
47 """Whether vertical drift correction / line assignment is exposed (PRE-21)."""
48 return experimental_features_enabled()
51def similarity_enabled() -> bool:
52 """Whether NLD scanpath-similarity scoring is exposed (PRE-21)."""
53 return experimental_features_enabled()
56def multipleye_upload_enabled() -> bool:
57 """Whether the wizard's "Dataset format" choice + MultiplEYE upload branch
58 are exposed **in the app**, this release (UX-114).
60 Held back the same way PRE-22 holds back the preprocessing panel: the code
61 (`wizard._render_multipleye_upload`, the format `segmented_control`) stays
62 in place for a later revival — most of what it did by hand is now what the
63 generalized Generic wizard can do too (UX-113's filename-derive +
64 block-aware char-AOI aggregation), so it may shrink further rather than
65 simply come back — but this release ships with no format *choice* at all:
66 every upload goes through Generic. Scope is the **app** only;
67 `datasets.multipleye_frames_from_uploads`/`load_multipleye_uploads` are
68 untouched and still callable directly.
69 """
70 return experimental_features_enabled()
73def multipleye_enabled() -> bool:
74 """Whether the MultiplEYE corpus is offered, this release (DATA-54).
76 Held back for the beta: its data is not openly available yet, and the loader
77 was built and tested on one sample (Zurich Chinese). Off, the entry leaves
78 the public-dataset registry — so the data picker, the 🗂️ Data page, Compare's
79 second dataset and share links all stop offering it — and `render`'s
80 ``--source multipleye`` / ``--export`` / ``--no-question-screens`` flags are
81 hidden from ``--help``. Like PRE-22's gate, this hides rather than breaks:
82 those flags still parse, and `datasets.load_multipleye` is untouched, so a
83 script that already uses them keeps working.
84 """
85 return experimental_features_enabled()
88def benchmark_corpora_enabled() -> bool:
89 """Whether the harmonised benchmark corpora are offered, this release
90 (DATA-54, DATA-55).
92 A bundle can only be built with the EyeGenBench pipeline, which is not public
93 yet, and the corpora are unfinished work — the picker marks each one (WIP).
94 The app no longer discovers them at all (DATA-55: a corpus is listed only
95 once someone adds it, and the flow that adds one is DATA-56). Off, this also
96 keeps an added corpus out of `app.public_dataset_registry`, and hides
97 `render`'s ``--eyegenbench`` / ``--eyegenbench-dataset`` flags from
98 ``--help``. Hidden, not removed: those flags still parse, and
99 `eyegenbench.load_eyegenbench` is untouched.
100 """
101 return experimental_features_enabled()
104def preprocessing_enabled() -> bool:
105 """Whether the soft-exclusion / merge pipeline is exposed **in the app** (PRE-22).
107 The feature is finished and tested; it is held back from this release's UI
108 and picked up in the next one, so the same flag that carries PRE-21's
109 unfinished work carries it — one env var for "not in this release", one code
110 path, and no branch to rebase.
112 `api.preprocess_data` and `scanpath-studio analyze` are held back by
113 `computed_measures_enabled` instead, which raises rather than hides.
114 """
115 return experimental_features_enabled()
118def computed_measures_enabled() -> bool:
119 """Whether the numbers Scanpath Studio works out itself are exposed, this release.
121 Held back until each is checked by hand: the per-word reading measures
122 (`api.compute_word_metrics`), the reader / trial summaries, the analysis
123 tables and `scanpath-studio analyze`, the export bundle's measure family,
124 and the Corpus Analysis views built on them (Reading summary, Progressive vs
125 regressive, Landing-position curve, Reader summary table). What stays is
126 everything that shows the dataset's own values. Unlike PRE-22's app-only
127 gate, the API raises and the CLI refuses: a script that gets no number is
128 better off than one that gets an unchecked one.
129 """
130 return experimental_features_enabled()
133def sentence_analysis_enabled() -> bool:
134 """Whether Corpus Analysis offers its **Per sentence** subtab (AN-33).
136 Held back: it works its numbers out from the fixation table, while the rest
137 of the page shows only the measures the dataset brought (AN-32), so a
138 dataset with supplied measures and no fixations read as 0 ms and skipped.
139 It comes back once it is built on the supplied measures. Off, the subtab
140 is not drawn and its table is never computed; `preprocessing.sentence_measures`
141 is untouched (`api.analysis_tables` is held back by `computed_measures_enabled`).
142 """
143 return experimental_features_enabled()
146def derived_analysis_tables_enabled() -> bool:
147 """Whether the 🗂️ Data page's "🧮 Derived analysis tables" section (Sentences
148 / Saccades / Trials / Readers / Characters) is exposed **in the app** (UX-126).
150 Same gate, same reasoning as `preprocessing_enabled()`/`drift_correction_enabled()`:
151 the feature is finished and tested, held back for a later release. Off, the
152 section doesn't render *and* its backing computation (`tabs._c_derived_tables`
153 — full reading-measure + saccade/sentence/reader aggregation) never runs, not
154 just its display — a Data page rerun with the flag off does none of that work.
155 Scope is the **app** only: `preprocessing.py`'s own table builders are
156 untouched and still directly callable.
157 """
158 return experimental_features_enabled()
161# Default text font. A single generic family that renders (monospaced) on every
162# platform including the Streamlit Cloud demo; the font field accepts any CSS
163# font name or stack if you want the exact experiment font.
164FONT_FAMILY = "monospace"
166# The demo corpus' own presentation monitor, so a figure with no declared
167# canvas still renders true-to-scale. Sourced once, in
168# `eyegenbench_geometry.DISPLAY_SPECS["onestop"]` (Berzak et al. 2025, Sci Data
169# 12:1995 — Dell U2715H, 2560 px × 1440 px over 597 mm × 336 mm).
170DEFAULT_FIGURE_SIZE = (2560, 1440)
172# Reading text is drawn true-to-scale: one line of text fills ``1/line_spacing``
173# of the line pitch (the word-box height that the data already encodes). OneStop
174# rendered each line of text with one blank line above and one below it, so the
175# line pitch is 3x the single-line height — hence a default line spacing of 3.
176DEFAULT_LINE_SPACING = 3.0
178COLORSCALES = [
179 "Blues",
180 "Greens",
181 "Oranges",
182 "Reds",
183 "Purples",
184 "Greys",
185 "Viridis",
186 "Plasma",
187 "Inferno",
188 "Magma",
189 "Cividis",
190 "Turbo",
191 "Hot",
192 "YlOrRd",
193 "YlGnBu",
194 "RdBu",
195 "Spectral",
196]
198# Both colour scales open in Blues: one hue, light to dark, colourblind-safe.
199# A keyed selectbox first-rendered inside a popover would otherwise display its
200# first option rather than a non-index-0 seeded value on first open — handled by
201# `controls._popover_selectbox` (explicit `index=`) / `_pin` + `persist_state`, so a
202# non-index-0 default here still keeps the picker and the figure in sync.
203DEFAULT_FIXATION_COLORSCALE = "Blues"
204DEFAULT_HEATMAP_COLORSCALE = "Blues"
205#: Heatmap styles that scale their smoothed density to each figure's own peak
206#: (`plots._add_interpolated_heatmap`), so a ``heatmap_range`` does nothing to
207#: them: the rail greys the range for these, and the code snippet omits it.
208#: Compare always draws word boxes, where the range applies again.
209SELF_SCALED_HEATMAP_STYLES = frozenset({"Interpolated"})
211DEFAULT_MARKER_SIZE_RANGE = (8, 24)
212# How fixation duration maps onto that size range. The three *fixed* scales map
213# one duration range (ms, below) onto it for every figure, so a 200 ms fixation
214# is the same size in any trial, comparison side, replay or export — EyeLink
215# Data Viewer's convention. "relative" is the original behaviour: each figure
216# spans its own shortest-to-longest duration, so sizes only compare within it.
217# √ is the default because a marker's size is its diameter: √duration makes the
218# marker's area grow in step with duration.
219MARKER_SIZE_SCALES = {
220 "sqrt": "√ duration (area)",
221 "linear": "Linear (diameter)",
222 "log": "Log duration",
223 "relative": "Relative to this figure",
224}
225DEFAULT_MARKER_SIZE_SCALE = "sqrt"
226#: What a figure saved before the fixed scale existed was drawn with — the
227#: migration target for old saved configs and Share links.
228LEGACY_MARKER_SIZE_SCALE = "relative"
229# 50–600 ms covers 99% of the bundled OneStop fixations (1st percentile 51 ms,
230# 99th 448 ms), sits below the 80 ms short-fixation flag, and leaves the long
231# tail distinguishable before it clamps. Durations outside it clamp to the
232# smallest / largest marker.
233DEFAULT_MARKER_DURATION_RANGE = (50, 600)
234#: The duration-bounds widget's own limits (ms).
235MARKER_DURATION_BOUNDS = (10, 3000)
236DEFAULT_ORDER_FONT_COLOR = "#111111"
238WORD_BOX_COLOR = "#6c757d"
239#: The outline's opacity; 1 draws it solid, as before the setting existed.
240WORD_BOX_LINE_OPACITY = 1.0
241#: The word boxes' fill, drawn translucent (``WORD_BOX_FILL_OPACITY``) so it
242#: tints the interest area without hiding the text, fixations or image under it.
243WORD_BOX_FILL_COLOR = "#646464"
244WORD_BOX_FILL_OPACITY = 0.05
245# VIZ-32: black, matching the colourblind-safe default palette.
246WORD_LABEL_COLOR = "#000000"
247# Default colour for highlighted ("Mark text") reading text — vermillion,
248# matching the colourblind-safe default palette. The visualization controls
249# expose a picker that overrides it per figure.
250HIGHLIGHTED_TEXT_COLOR = "#D55E00"
251# VIZ-32: reddish purple, matching the colourblind-safe default palette.
252SACCADE_COLOR = "#CC79A7"
253TRENDLINE_COLOR = "#dc3545"
254CURRENT_FIX_COLOR = "rgba(255, 127, 14, 0.6)"
255CURRENT_FIX_OUTLINE = "#ff7f0e"
256FIX_MARKER_OUTLINE = "#111"
257COMPARISON_PALETTE = ["#1f77b4", "#e45756"]
260def compare_palette_color(idx: int) -> str:
261 """Default A/B colour for comparison scanpath ``idx`` — the single source of
262 truth shared by the per-scanpath style controls (``controls._seed_compare_styles``
263 / ``_collect_compare_styles``) and the figure builders
264 (``plots._comparison_scanpath_style``), so the swatch shown in the controls can
265 never drift from what's drawn (CMP-3)."""
266 return COMPARISON_PALETTE[idx % len(COMPARISON_PALETTE)]
269#: Each comparison scanpath's default marker alpha, shared the same way: the
270#: rail seeds ``cmp{idx}_opacity`` from it (``controls._seed_compare_styles``)
271#: and the builder falls back to it (``plots._comparison_scanpath_style``).
272#: CMP-20: the builder's own 1.0 was what every headless ``compare_scanpaths`` /
273#: ``render --compare-with`` drew, so the default comparison differed from the
274#: app's. 0.7 matches the single-trial default, so overlaps show through.
275COMPARE_FIXATION_OPACITY = 0.7
278# Saccade line styles offered in the plot rail. Maps the friendly UI label to the
279# Plotly ``line.dash`` value used in the figure builders.
280SACCADE_DASH_OPTIONS = {
281 "Solid": "solid",
282 "Dashed": "dash",
283 "Dotted": "dot",
284 "Dash-dot": "dashdot",
285}
286# Saccade line width (px): default + the (min, max) the width slider allows.
287DEFAULT_SACCADE_WIDTH = 2.0
288SACCADE_WIDTH_BOUNDS = (0.5, 10.0)
289#: The Interpolated heatmap's fixed blur σ (px): the box's limits and default.
290HEATMAP_SIGMA_BOUNDS = (1.0, 500.0)
291DEFAULT_HEATMAP_SIGMA_PX = 20.0
293# VIZ-8 · colour saccades by reading type. Each saccade (the segment from one
294# fixation to the next) is classified into one of these reading-schematic
295# classes by ``measures.classify_saccades`` and — in the "By type" colour mode —
296# drawn as its own sub-trace with a small legend. ``other`` is the catch-all for
297# saccades that can't be classified (an endpoint fell outside every word box); it
298# isn't user-editable, so the palette UI exposes only the five reading classes.
299# Order controls the legend order.
300SACCADE_CLASS_ORDER = [
301 "forward",
302 "skip",
303 "refixation",
304 "return_sweep",
305 "regression",
306 "other",
307]
308SACCADE_CLASS_LABELS = {
309 "forward": "Forward",
310 "skip": "Skip",
311 "refixation": "Refixation",
312 "return_sweep": "Return sweep",
313 "regression": "Regression",
314 "other": "Other",
315}
316# VIZ-32: Okabe-Ito, matching the colourblind-safe default palette.
317SACCADE_CLASS_COLORS = {
318 "forward": "#009E73", # bluish green — normal left-to-right progression
319 "skip": "#56B4E9", # sky blue — jumps over one or more words
320 "refixation": "#CC79A7", # reddish purple — lands back on the same word
321 "return_sweep": "#E69F00", # orange — long sweep to the next line
322 "regression": "#D55E00", # vermillion — moves backward
323 "other": "#999999", # grey — unclassifiable (off-text endpoint)
324}
325# The five reading classes the palette UI lets the user recolour (``other`` is
326# fixed grey).
327SACCADE_CLASS_EDITABLE = SACCADE_CLASS_ORDER[:-1]
329# VIZ-19 · saccade colour modes. The five-way "By type" split is more than most
330# figures need, so there's a middle option between one flat colour and the full
331# reading-class breakdown: "Forward / regression", the distinction almost every
332# reading paper actually draws. It reuses the same per-class machinery — the
333# classes are just folded into two buckets before the segments are built, so the
334# colour pickers, the legend toggle and every surface stay as they are.
335SACCADE_COLOR_MODES = ("Uniform", "Forward / regression", "By type")
336SACCADE_DIRECTION_CLASSES = ("forward", "regression")
337# reading class → the bucket it is drawn in under "Forward / regression".
338# ``other`` stays ``other`` (grey catch-all) so unclassifiable saccades aren't
339# silently counted as progressive.
340SACCADE_DIRECTION_FOLD = {
341 "forward": "forward",
342 "skip": "forward",
343 "refixation": "forward",
344 "return_sweep": "forward",
345 "regression": "regression",
346 "other": "other",
347}
348SACCADE_DIRECTION_LABELS = {
349 "forward": "Forward",
350 "regression": "Regression",
351 "other": "Other",
352}
354# VIZ-15 · fixation marker shape. Plotly symbol name → the label shown in the
355# picker. Shape is a *second* encoding channel, and unlike hue it survives
356# greyscale printing — so it pairs with VIZ-17 (colour freed up once it stops
357# duplicating size) and VIZ-18 (print / colourblind palettes).
358FIXATION_SYMBOLS = {
359 "circle": "● Circle",
360 "square": "■ Square",
361 "diamond": "◆ Diamond",
362 "triangle-up": "▲ Triangle",
363 "cross": "✚ Cross",
364 "x": "✖ X",
365 "star": "★ Star",
366 "hexagon": "⬡ Hexagon",
367 "heart": "♥ Heart",
368}
369DEFAULT_FIXATION_SYMBOL = "circle"
371# Shapes Plotly's ``marker.symbol`` enum doesn't have. They're drawn as *text*
372# glyphs instead — a Scatter in text mode, sized per point via an array
373# ``textfont.size``, so duration→size still holds. Kept as a mapping so adding
374# another glyph shape needs no new branch in the figure builder.
375# Glyphs render at roughly half the visual weight of a marker of the same
376# nominal size, so the sizes are scaled up to match the other shapes.
377FIXATION_GLYPH_SYMBOLS = {"heart": "♥"}
378FIXATION_GLYPH_SIZE_SCALE = 1.8
380# VIZ-17 · the "Color fixations by" option meaning *don't* map a variable to hue.
381# Marker size already encodes fixation duration, so colouring by duration too
382# double-encodes one variable and spends the colour channel on nothing. The
383# default is therefore one flat colour, and colour-by is an explicit opt-in for a
384# *different* variable (surprisal, frequency, line, pass index).
385UNIFORM_COLOR_FIELD = "(uniform)"
386# VIZ-32: blue, matching the colourblind-safe default palette.
387DEFAULT_FIXATION_COLOR = "#0072B2"
389# Outline width (px) for hollow (outline-only) fixation markers.
390HOLLOW_OUTLINE_WIDTH = 2.0
392# Distinct mark for fixations that fall outside every word box ("out of text").
393OUT_OF_TEXT_COLOR = "#d62728" # red
395# Plot background. Default white; some analyses prefer a neutral gray.
396# A "Custom…" entry in the rail reveals a free color picker.
397DEFAULT_BACKGROUND_COLOR = "#ffffff"
398BACKGROUND_PRESETS = {
399 "White": "#ffffff",
400 "Light gray": "#e9ecef",
401 "Gray": "#bdbdbd",
402 "Black": "#000000",
403}
405CANVAS_PAD_MIN_PX = 20.0
406CANVAS_PAD_FRACTION = 0.05
409# --- BUG-101 · the Plotly config every figure is drawn with --------------------
410# plotly.js 3 defaults `showSendToCloud` to true: the modebar's "Share chart…"
411# button uploads the whole figure — words, coordinates, hover fields — to
412# cloud.plotly.com. Nothing in Scanpath Studio sends data anywhere unasked, and
413# its own Share means something else, so every figure turns the button off:
414# the app's embeds and charts, the HTML it writes, and the docs site's figures.
415# Merge it into any other config: ``{**PLOTLY_CONFIG, "responsive": False}``.
416PLOTLY_CONFIG: dict = {"showSendToCloud": False}
419# --- VIZ-18 · selectable palettes --------------------------------------------
420# These figures don't only get looked at on the screen they were made on: they go
421# into papers (printed, sometimes in black & white) and are read by colourblind
422# viewers. One palette can't serve all of that, so the colour defaults are a
423# *choice* rather than a constant.
424#
425# A palette is a preset, not a second rendering path: picking one writes the
426# ordinary per-element colour keys, so every existing picker still overrides it
427# and every surface (deep link, Save & restore, CLI, API) carries the resulting
428# colours with no new plumbing. ``palette_settings`` returns the figure-kwarg
429# form; ``controls.apply_palette`` writes the session keys.
430#
431# Rules each non-default palette follows:
432# * hues distinguishable under deuteranopia/protanopia (no red-vs-green pair
433# carrying meaning on its own), and
434# * **lightness** ordered as well as hue, so the figure still reads after a
435# greyscale conversion. Marker shape (VIZ-15) and the two-way saccade mode
436# (VIZ-19) are the redundant channels when colour alone can't carry it.
437PALETTES: dict[str, dict] = {
438 # Okabe & Ito's eight-colour set — the de-facto standard for qualitative
439 # colourblind-safe encoding — plus single-hue Blues scales, which vary in
440 # lightness only and so survive every common deficiency. VIZ-32: this is the
441 # default a fresh session opens with, not just an opt-in choice.
442 "Default (colourblind-safe)": {
443 "description": "Okabe–Ito hues + Blues scales; safe for deuteran-, "
444 "protan- and tritanopia.",
445 "fixation_color": DEFAULT_FIXATION_COLOR,
446 "fixation_colorscale": DEFAULT_FIXATION_COLORSCALE,
447 "heatmap_colorscale": DEFAULT_HEATMAP_COLORSCALE,
448 "saccade_color": SACCADE_COLOR,
449 "saccade_class_colors": dict(SACCADE_CLASS_COLORS),
450 "word_label_color": WORD_LABEL_COLOR,
451 "highlight_text_color": HIGHLIGHTED_TEXT_COLOR,
452 "background_color": DEFAULT_BACKGROUND_COLOR,
453 },
454 # Lightness-only encoding: everything survives a black & white print or
455 # photocopy, because nothing depends on hue at all.
456 "Print / greyscale": {
457 "description": "Grays only — nothing depends on hue, so it survives a "
458 "black & white print. Pair with marker shape.",
459 "fixation_color": "#1a1a1a",
460 "fixation_colorscale": "Greys",
461 "heatmap_colorscale": "Greys",
462 "saccade_color": "#7a7a7a",
463 "saccade_class_colors": {
464 # Ordered by lightness, darkest = the thing you're looking for.
465 "forward": "#a6a6a6",
466 "skip": "#8a8a8a",
467 "refixation": "#5e5e5e",
468 "return_sweep": "#c4c4c4",
469 "regression": "#000000",
470 "other": "#d9d9d9",
471 },
472 "word_label_color": "#333333",
473 "highlight_text_color": "#000000",
474 "background_color": "#ffffff",
475 },
476 # Maximum separation from the background and from each other — projectors,
477 # low-quality displays, and low-vision viewers.
478 "High contrast": {
479 "description": "Saturated, dark-on-white hues for projectors and "
480 "low-contrast displays.",
481 "fixation_color": "#0033cc",
482 "fixation_colorscale": "Cividis",
483 "heatmap_colorscale": "Cividis",
484 "saccade_color": "#cc0000",
485 "saccade_class_colors": {
486 "forward": "#006600",
487 "skip": "#0033cc",
488 "refixation": "#6600cc",
489 "return_sweep": "#cc6600",
490 "regression": "#cc0000",
491 "other": "#4d4d4d",
492 },
493 "word_label_color": "#000000",
494 "highlight_text_color": "#cc0066",
495 "background_color": "#ffffff",
496 },
497}
498DEFAULT_PALETTE = "Default (colourblind-safe)"
499#: #374: the names a person reads. The keys above are stored values (settings
500#: files, Share links, ``palette=``) and keep their spelling; every place that
501#: shows a palette shows it through `palette_label`.
502PALETTE_LABELS = {
503 "Default (colourblind-safe)": "Default (colorblind-safe)",
504 "Print / greyscale": "Print / grayscale",
505}
508def palette_label(name: str) -> str:
509 """A palette's display name (US spelling); any other value as it is."""
510 return PALETTE_LABELS.get(name, name)
513# Not a palette — the honest answer when the live colours match none of them.
514# A palette only *presets* the individual colour keys, so the moment one of those
515# pickers is changed the selector would otherwise keep naming a palette the figure
516# no longer uses. Deliberately kept OUT of ``PALETTES`` so the registry stays the
517# set of things that can actually be applied: `--palette` choices, the API's
518# expansion, and the deep link all iterate `PALETTES` and must not offer this.
519CUSTOM_PALETTE = "Custom"
522def palette_settings(name: str) -> dict:
523 """Figure-kwarg colour settings for palette ``name`` (falls back to Default).
525 Returns a fresh dict (nested ``saccade_class_colors`` copied too), so callers
526 can mutate the result without corrupting the registry.
527 """
528 entry = PALETTES.get(name) or PALETTES[DEFAULT_PALETTE]
529 settings = {k: v for k, v in entry.items() if k != "description"}
530 settings["saccade_class_colors"] = dict(settings["saccade_class_colors"])
531 return settings
534# --- App theme (BUG-6) -------------------------------------------------------
535# The branded look. Streamlit only auto-loads ``.streamlit/config.toml`` relative
536# to the *launch* directory, so ``streamlit run streamlit_app.py`` from ``app/``
537# (Streamlit Cloud) picks it up but ``python -m scanpath_studio`` from anywhere
538# else — or a ``pip``-installed console script, which never ships that file —
539# falls back to Streamlit's default red accent. ``cli.launch_app`` injects these
540# as ``--theme.*`` flags so every launch path renders the same theme regardless
541# of the working directory. Kept in sync with ``app/.streamlit/config.toml`` —
542# ``tests/test_theme.py`` asserts parity so the two can't drift.
543APP_THEME = {
544 "base": "light",
545 "primaryColor": "#1f77b4",
546 "backgroundColor": "#ffffff",
547 "secondaryBackgroundColor": "#f5f7fa",
548 "textColor": "#212529",
549 "font": "sans-serif",
550}
551# Dark-variant overrides ([theme.dark] in config.toml). Users switch via the ☰
552# menu → Settings → Appearance, or follow their OS.
553APP_THEME_DARK = {
554 "primaryColor": "#5aa9e6",
555 "backgroundColor": "#0e1117",
556 "secondaryBackgroundColor": "#1c2030",
557 "textColor": "#e8eaed",
558}
560#: Per-file upload ceiling, in MB. Streamlit's own default is **200 MB**, which
561#: is far under a real eye-tracking export — a single zipped fixation report for
562#: one OneStop regime already runs to tens of MB, and a full corpus is orders
563#: above that. Kept in sync with ``.streamlit/config.toml``'s
564#: ``server.maxUploadSize``; `cli._max_upload_cli_flags` passes it explicitly
565#: because that config file is resolved against the *launch* directory, so a
566#: pip-installed console script started from anywhere else silently fell back to
567#: the 200 MB default. This is the transport limit only — what a machine can
568#: actually parse is a separate question, which `data.UPLOAD_SIZE_WARN_BYTES`
569#: answers for the memory-capped hosted demo.
570UPLOAD_MAX_SIZE_MB = 5000
572#: ENG-68: a deployment's own, lower per-file ceiling, in MB — set on the hosted
573#: demo (its Community Cloud secrets, which Streamlit loads into the environment
574#: at startup) so a visitor cannot send a multi-GB file at a ~1 GB container,
575#: while every other install keeps :data:`UPLOAD_MAX_SIZE_MB`.
576UPLOAD_LIMIT_ENV = "SCANPATH_MAX_UPLOAD_MB"
579#: The file types every table upload box accepts (``zip`` wraps any of the
580#: others; ``txt`` is a tab-separated report; ``xls`` is a legacy workbook or
581#: EyeLink Data Viewer's text-in-an-``.xls`` export, DATA-53 / BUG-55).
582UPLOAD_FILE_TYPES = ("csv", "tsv", "txt", "parquet", "feather", "zip", "xlsx", "xls")
585def configured_upload_limit_mb() -> int | None:
586 """``SCANPATH_MAX_UPLOAD_MB`` as a positive whole number of MB, else ``None``.
588 The raw deployment setting, before it meets the server's own limit — what
589 ``scanpath-studio run`` hands the server (ENG-68).
590 """
591 raw = os.environ.get(UPLOAD_LIMIT_ENV, "").strip()
592 try:
593 limit = int(raw)
594 except ValueError:
595 return None
596 return limit if limit > 0 else None
599def upload_limit_mb() -> int | None:
600 """The per-file cap every ``st.file_uploader`` passes as ``max_upload_size``.
602 ``None`` — the server's own ``server.maxUploadSize`` — unless
603 ``SCANPATH_MAX_UPLOAD_MB`` sets one, which is held to the server's limit so
604 the browser never accepts a file the server then refuses. Read at call
605 time, like the other deployment switches, so tests can toggle it.
607 The per-widget cap is checked **in the browser**: Streamlit's upload route
608 only enforces ``server.maxUploadSize``, which cannot change once the server
609 runs. ``scanpath-studio run`` therefore passes the cap to the server too;
610 on Community Cloud, where secrets load after the server config, a scripted
611 client can still send up to the config file's limit.
612 """
613 limit = configured_upload_limit_mb()
614 if limit is None:
615 return None
616 try:
617 import streamlit as st
619 server = int(st.get_option("server.maxUploadSize"))
620 except Exception:
621 server = UPLOAD_MAX_SIZE_MB
622 return min(limit, server)
625def upload_limit_label() -> str:
626 """The per-file limit in force, as Streamlit writes it (``5GB``, ``200MB``)."""
627 mb = upload_limit_mb() or UPLOAD_MAX_SIZE_MB
628 return f"{mb // 1000}GB" if mb >= 1000 and mb % 1000 == 0 else f"{mb}MB"
631def upload_identity(uploaded) -> tuple[str | None, str]:
632 """Which upload this is: ``(file_id, sha256 of the bytes)``.
634 An import that applies a file once — a settings or setup file — compares
635 this with the identity it last applied, so an ordinary rerun is a no-op
636 while a *fresh* upload applies again. The ``file_id`` Streamlit gives each
637 upload event makes re-uploading the very same file count as fresh; the
638 content hash makes a different file count as fresh even where no
639 ``file_id`` exists. Name and size alone did neither: two files can share
640 both.
641 """
642 import hashlib
644 digest = hashlib.sha256(uploaded.getvalue()).hexdigest()
645 return getattr(uploaded, "file_id", None), digest
648CITATION = {
649 "authors": (
650 "Omer Shubi, Keren Gruteke Klein, Maya Grossman, Ella Lion, Deborah N. Jakobi, "
651 "David R. Reich, Lena Jäger, Yevgeni Berzak"
652 ),
653 "title": "Scanpath Studio",
654 # ENG-75: the Zenodo *concept* DOI — resolves to the latest archived
655 # release. Kept equal to CITATION.cff's `doi` by tests/test_citation.py.
656 "doi": "10.5281/zenodo.22933884",
657 "url": "https://github.com/lacclab/scanpath-studio",
658 "docs_url": "https://lacclab.github.io/scanpath-studio/",
659 # Where a user asks, reports, and gets the desktop app.
660 "questions_url": "https://github.com/lacclab/scanpath-studio/discussions/categories/q-a",
661 "bug_report_url": "https://github.com/lacclab/scanpath-studio/issues/new?template=bug_report.md",
662 "desktop_url": "https://github.com/lacclab/scanpath-studio/releases/latest",
663 "lab_url": "https://lacclab.github.io/",
664 "corpus_note": (
665 "Bundled demo data is a subset of OneStop Eye Movements: "
666 "Berzak, Malmaud, Shubi, Meiri, Lion, Levy (2025), "
667 '"OneStop: A 360-Participant English Eye Tracking Dataset with '
668 'Different Reading Regimes," Scientific Data. '
669 "https://doi.org/10.1038/s41597-025-06272-2"
670 ),
671}
674# --- Data-source identity + main view labels --------------------------------
675# Moved out of app.py so url_state.py / wizard.py can import them without a
676# cycle (app.py re-imports them for its own use and for tests).
677UPLOAD_CHOICE = "Upload tables"
678AUTHOR_CHOICE = "Author a scanpath"
679MANUAL_SAMPLE_CHOICE = "Synthetic sample"
680DEMO_CHOICE = "Bundled Demo"
681SYNTHETIC_CHOICE = "Synthetic test trial"
682PUBLIC_DATASETS_CHOICE = "Public datasets"
683POTEC_DEFAULT_DIR = "data/PoTeC"
684EYEGENBENCH_DEFAULT_DIR = "data/EyeGenBench"
685# DATA-27 (Task 11R): every prepared benchmark corpus is its own top-level entry
686# in the flat data-source picker, exactly like PoTeC / MultiplEYE / OneStop —
687# there is no "EyeGenBench" source fronting them. **"EyeGenBench" is provenance,
688# not a source**: it names the pipeline that harmonises the corpora and is being
689# extracted into its own repository, so it appears in descriptions and help
690# strings only — never in an entry label, which is built from the corpus'
691# manifest name (`app.py`).
692# The suffix that distinguishes a harmonised corpus from a *native* entry for the
693# same corpus (PoTeC, OneStop ship both ways). Applied by property — the
694# harmonised copy is re-derived and its geometry may be weaker — never by vendor
695# name; see consequence 1 above.
696BENCHMARK_SHORT_SUFFIX = " (harmonised benchmark)"
697# The registry-key suffix. Keys must be unique across the whole registry, and a
698# native entry's key is `"<Corpus> — <full name>"`, so this shape can't collide.
699BENCHMARK_LABEL_SUFFIX = " — harmonised benchmark corpus"
700# DATA-27 ships to main unfinished, so every harmonised corpus wears this in the
701# source picker. It is **display only** — appended by `app._entry_label` at
702# render time, never stored on the entry. Putting it on the registry key or on
703# `short` would change the picker's stored `data_source_choice` value and (for a
704# built-in) the `?corpus=` slug, so removing it later would break links and saved
705# configs written while it was up; as a formatting step it costs one line to
706# delete. Nothing derives identity from it.
707BENCHMARK_WIP_SUFFIX = " (WIP)"
709# DATA-27 R35: a benchmark manifest records `language` as an **ISO 639-1 code**
710# ('zh', 'da', 'en', …), not a name, and the picker shows it to a reader. A small
711# explicit table beats a dependency for a field this narrow; unknown codes fall
712# back to the code itself (never "Unknown" — the code is real information, and
713# inventing a placeholder for it loses that).
714LANGUAGE_NAMES = {
715 "ar": "Arabic",
716 "bg": "Bulgarian",
717 "ca": "Catalan",
718 "cs": "Czech",
719 "da": "Danish",
720 "de": "German",
721 "el": "Greek",
722 "en": "English",
723 "es": "Spanish",
724 "et": "Estonian",
725 "eu": "Basque",
726 "fa": "Persian",
727 "fi": "Finnish",
728 "fr": "French",
729 "ga": "Irish",
730 "he": "Hebrew",
731 "hi": "Hindi",
732 "hr": "Croatian",
733 "hu": "Hungarian",
734 "id": "Indonesian",
735 "is": "Icelandic",
736 "it": "Italian",
737 "ja": "Japanese",
738 "ko": "Korean",
739 "lt": "Lithuanian",
740 "lv": "Latvian",
741 "mk": "Macedonian",
742 "ml": "Malayalam",
743 "nl": "Dutch",
744 "no": "Norwegian",
745 "pl": "Polish",
746 "pt": "Portuguese",
747 "ro": "Romanian",
748 "ru": "Russian",
749 "sk": "Slovak",
750 "sl": "Slovenian",
751 "sq": "Albanian",
752 "sr": "Serbian",
753 "sv": "Swedish",
754 "ta": "Tamil",
755 "th": "Thai",
756 "tr": "Turkish",
757 "uk": "Ukrainian",
758 "ur": "Urdu",
759 "vi": "Vietnamese",
760 "zh": "Chinese",
761}
764def language_display(value) -> str:
765 """An ISO language code (or comma-joined list of them) as display names.
767 A multilingual corpus records several codes in one field (`"en,de,ru"` — MECO,
768 celer), so each part is mapped and the list re-joined. An unrecognised code
769 renders as itself: it still identifies the language to anyone who knows it,
770 which "Unknown" does not. A blank/absent value gives ``""`` so the caller can
771 drop the caption rather than print an empty label.
772 """
773 text = str(value or "").strip()
774 if not text:
775 return ""
776 parts = [part.strip() for part in text.split(",") if part.strip()]
777 return ", ".join(LANGUAGE_NAMES.get(part.lower(), part) for part in parts)
780MULTIPLEYE_DEFAULT_DIR = "data/MultiplEYE_ZH_CH_Zurich_1_2025"
781ONESTOP_CHOICE = "OneStop server bundle"
782# Public OneStop (OSF download-on-demand) — distinct from the env-var
783# ONESTOP_CHOICE server bundle. DATA-63: each reading regime is its own dataset
784# in the flat data-source picker, holding every part of that regime; these are
785# their registry labels. Named constants (not inline strings) so the deep-link /
786# Share contract in url_state.py and compare_source.py can reference them
787# without importing app.py's PUBLIC_DATASET_REGISTRY (DATA-3: shareable).
788ONESTOP_PUBLIC_DEFAULT_DIR = "data/OneStop"
789# Default folder for the lacclab OneStop variant (a lab-processed local export;
790# superset schema, no download). Blank: ENG-60 — it was one maintainer's own
791# OneDrive path, prefilled into the Data directory box for every user who picked
792# the option. Set `ONESTOP_LACCLAB_DIR`, or type the path on the 🗂️ Data page.
793ONESTOP_LACCLAB_DEFAULT_DIR = ""
794ONESTOP_REGIME_LABELS = {
795 "ordinary": "Ordinary reading",
796 "information_seeking": "Information seeking",
797 "repeated": "Repeated reading",
798 "information_seeking_repeated": "Information seeking (repeated)",
799}
800#: DATA-63 — regime → the registry label of that regime's dataset.
801ONESTOP_REGIME_CHOICES = {
802 regime: f"OneStop — {label}" for regime, label in ONESTOP_REGIME_LABELS.items()
803}
806#: DATA-63 — each regime dataset's ``?source=`` token. Its own per regime,
807#: because the token also names Compare's dataset B (``cmp_source``), which
808#: carries no regime of its own. The DATA-3 ``onestop_public`` token stays
809#: readable: with its ``onestop_regime`` it names one of these four.
810ONESTOP_REGIME_SOURCE_TOKENS = {
811 regime: f"onestop_{regime}" for regime in ONESTOP_REGIME_LABELS
812}
813ONESTOP_LEGACY_SOURCE_TOKEN = "onestop_public"
816def onestop_regime_for_choice(choice: str | None) -> str | None:
817 """The regime a OneStop dataset label names, else ``None``."""
818 return next((r for r, c in ONESTOP_REGIME_CHOICES.items() if c == choice), None)
821# OneStop trial parts (screens) → display label, in presentation order. Mirrors
822# datasets._ONESTOP_PARTS; the Parts multiselect + the parts URL/CLI contract
823# use these keys. Paragraph is the reading passage (the default).
824ONESTOP_PART_LABELS = {
825 "Title": "Title",
826 "Question_Preview": "Question preview",
827 "Paragraph": "Paragraph",
828 "Questions": "Question",
829 "Answers": "Answers",
830 "QA": "Question + answers combined (QA)",
831 "Feedback": "Feedback",
832}
833# OneStop source variants → display label. `public` downloads from OSF on demand;
834# `lacclab` reads a lab-processed local export (superset schema, no download).
835ONESTOP_VARIANT_LABELS = {
836 "public": "Public (OSF download)",
837 "lacclab": "LaCC lab (local export)",
838}
839MULTIPLEYE_BUNDLE_CHOICE = "MultiplEYE server bundle"
840# Default dir for the MultiplEYE *server bundle* (per-session parquet shards
841# under `<dir>/scanpath/`), overridable via the `MULTIPLEYE_DATA_DIR` env var.
842# Empty default so the source only appears when the env var is set (like the
843# OneStop server bundle). Distinct from MULTIPLEYE_DEFAULT_DIR above, which is
844# the public-corpus loader's tree.
845MULTIPLEYE_BUNDLE_DEFAULT_DIR = ""
846_VIEW_SCANPATH = "Scanpath Visualization"
847_VIEW_CORPUS = "Corpus Analysis"
848# DATA-26: the third top-level view — everything about the dataset *itself*
849# (source, location, column mapping, contents, recording setup, preprocessing),
850# which used to be split between the ⚙️ Configure / 🧹 Preprocessing menu
851# popovers and a 🔎 Data Inspection subtab buried in the Scanpath view.
852_VIEW_DATA = "Data"
853_MAIN_TAB_LABELS = [_VIEW_SCANPATH, _VIEW_CORPUS, _VIEW_DATA]
855#: DATA-32 — the dataset table's remembered headline counts, `{token: {...}}`.
856#: A session-state key that also travels in the recovery cache's manifest, so it
857#: lives here rather than in `app`: `persistence` writes it and `app` fills it,
858#: and `persistence` cannot import `app` (that is the cycle `app` already
859#: avoids by importing `persistence` one way).
860DATASET_COUNTS_STORE_KEY = "_dataset_counts_store"
862#: UX-174 r2 — ``{dataset token: description}``, every description the user
863#: wrote (on the add wizard or ✏️ Edit dataset), for any kind of dataset. One
864#: small dict persisted as a recovery-cache session key, *not* a field on an
865#: upload's ``_datasets`` entry: every non-frame field there is part of the
866#: cache's dataset identity, so editing one sentence would rewrite every
867#: upload's Parquet files. Here for the same import-cycle reason as above.
868DATASET_DESCRIPTIONS_KEY = "_dataset_descriptions"
870#: ``{dataset token: SetupSnapshot.to_dict()}`` — the recording setup the user
871#: saved on ✏️ Edit dataset for a **built-in or public** dataset, in place of
872#: the one the corpus declares (which is never rewritten). An upload keeps its
873#: setup on its own ``_datasets`` entry instead. A recovery-cache session key,
874#: like the descriptions, for the same reason.
875DATASET_SETUP_OVERRIDES_KEY = "_dataset_setup_overrides"
876#: Which dataset's override the ``global_*`` setup keys hold now, and what they
877#: held before it was applied — put back when that dataset is left or its
878#: override is reset (the shape of BUG-50's font snap). Both persisted, so a
879#: relaunch onto the dataset does not stash the override as its own "before".
880SETUP_OVERRIDE_FOR_KEY = "_setup_override_for"
881SETUP_OVERRIDE_RESTORE_KEY = "_setup_override_restore"
882#: The ``global_*`` keys an override writes (and its restore puts back).
883SETUP_OVERRIDE_SESSION_KEYS = (
884 "global_canvas_width",
885 "global_canvas_height",
886 "global_monitor_width_mm",
887 "global_viewing_distance_mm",
888 "global_display_dpi",
889 "global_base_font_size",
890 "global_font_family",
891 "global_line_spacing",
892 "global_scale_text_to_boxes",
893)
895#: UX-184 — the folder every public corpus downloads into, each in a subfolder
896#: (``<folder>/PoTeC``), set on the 🗂️ Data page. A recovery-cache session key
897#: so the choice survives a restart; never in a link, since it is a local path
898#: (see `session_keys.PARAM_CORPUS`). Unset, `DOWNLOAD_DIR_ENV` decides, then
899#: the checkout's ``data/`` or the per-user data home (ENG-59).
900DOWNLOAD_DIR_KEY = "download_dir"
901DOWNLOAD_DIR_ENV = "SCANPATH_STUDIO_DOWNLOAD_DIR"
903#: VIZ-45 — the raw-gaze layer's per-dataset default (`app.seed_raw_gaze_default`):
904#: the dataset it was last decided for, and the value that decision overwrote.
905#: Recovery-cache session keys, so a relaunch onto the same dataset does not
906#: decide again over the user's own choice; here for the import-cycle reason
907#: above (`persistence`, `controls` and `app` all read them).
908RAW_GAZE_SEEDED_FOR_KEY = "_raw_gaze_seeded_for"
909RAW_GAZE_SNAP_RESTORE_KEY = "_raw_gaze_snap_restore"
910#: …and the dataset an open deep link's `show_raw_gaze` belongs to — the first
911#: one decided while the link was on the URL. Session-only: a link is a visit.
912RAW_GAZE_LINK_FOR_KEY = "_raw_gaze_link_for"
914#: DATA-35 — "the ✏️ Edit dataset screen is open". The Data page is two screens
915#: now: the **overview** (the dataset table + what's in the open dataset) and the
916#: **editor** (everything that configures it — source options and location,
917#: column mapping, recording setup, trial identity, stimulus images, the two
918#: metadata tables, preprocessing). Both are built every run and switched by key,
919#: for exactly the reason the page itself is (see below): the editor is nothing
920#: but widgets that drive `prepare_data`, and Streamlit drops the key of a widget
921#: that did not render.
922DATASET_EDITOR_OPEN_KEY = "_dataset_editor_open"
923DATA_OVERVIEW_KEY = "data_overview"
924DATA_EDITOR_KEY = "data_dataset_editor"
925DATA_EDITOR_OFFSCREEN_KEY = "data_dataset_editor_offscreen"
927# DATA-26: the two keys the Data page's outer container is built with — visible
928# when that view is active, off-screen otherwise.
929#
930# The setup widgets (the loaders' directory input and ⬇ Download button, the
931# source options, the column-mapping selectboxes) *drive* `prepare_data` on
932# every rerun, and Streamlit drops the key of a widget that did not render. The
933# menu popovers they used to live in executed every run for exactly that reason;
934# a page body only executes while its view is selected. So the page is rendered
935# every run either way and simply hidden — `styles.py` gives the off-screen key
936# `display: none` — which keeps this a re-host rather than a rewrite of every
937# loader into a render/resolve pair.
938DATA_PAGE_KEY = "data_setup_page"
939DATA_PAGE_OFFSCREEN_KEY = "data_setup_page_offscreen"
941#: BUG-31 — the view the user tried to reach while the add-dataset wizard was
942#: open. `app.main` holds them on the 🗂️ Data page and the wizard asks whether to
943#: discard the setup; this remembers where they were headed so *Discard and
944#: leave* can finish the trip. Session state only, never a wire format.
945WIZARD_LEAVE_KEY = "_wizard_leave_requested"
947#: BUG-31 — the view the user answered *Keep setting up* for. The prompt is
948#: suppressed while the nav sits on exactly that view, so choosing to stay does
949#: not re-ask on every rerun; clicking a *different* view asks again. Session
950#: state only, never a wire format.
951WIZARD_STAY_KEY = "_wizard_leave_acknowledged"
953#: UX-54 — set by the dataset table's ✏️ Edit button to the dataset it opened, so
954#: the Column mapping section below can say that is why the page changed under
955#: the user. Read once and cleared; session state only, never a wire format.
956FOCUS_MAPPING_KEY = "_focus_column_mapping"
958# UX-47: ONE column grid for every control row stacked above the plot — the
959# Narrow-by row, the trial picker, the multipart screen navigator, the chip
960# strip, and (in Compare mode) the second dataset's picker and its own Narrow-by
961# row. Three tracks: **pick** (the row's own dropdown), **scrub** (the wide
962# middle — a slider, a pair of multiselects, the chips), **act** (the right-hand
963# `railbtn_*` pills, which `styles.py` packs flush right).
964#
965# It is a shared constant rather than a repeated literal because that is the
966# whole feature: the rows are built in three functions across two modules, and
967# each of them owning its own weights is exactly how they drifted apart — the
968# source dropdown ended 50 px short of the trial dropdown directly below it, and
969# the middle track started 38 px further left on one row than the next. Tracks
970# span by *summing* the weights they cover (the chip strip is 3 + 5), never by
971# inventing a second pair; a `st.columns` of merged weights differs from the
972# three-track boundary by a fraction of one gutter (~3 px at the default 1 rem),
973# which is below the threshold where an eye reads two edges as unaligned.
974#
975# Widths, not pixels: Streamlit shares the row minus its gutters out by weight,
976# so the grid holds at every window size.
977# UX-64 made it **four** tracks — dataset · trial · scrub · actions — because
978# the Narrow-by row above it is gone and its dataset picker moved down onto this
979# one. The dataset track is as wide as the trial track and deliberately does not
980# shrink: two datasets under comparison are told apart by that label. What gave
981# way is the scrubber (5.0 → 3.6) and the filters, which became one icon in the
982# actions cluster rather than a labelled **More** button of their own.
983# UX-143: the first track also holds +; borrow from the scrubber to keep the
984# dataset name readable. The same track boundaries apply to the related rows.
985# UX-181: the weights now favour the scrubber, and `styles.py` puts floors
986# under the other three tracks. The actions track is a fixed width: its pills
987# are a fixed ~11.5rem, and a proportional track grew with the window and left
988# an empty gap to the left of ◀. Its weight is nominal. The dataset and trial
989# tracks shrink with the window down to a floor that still shows a name in
990# full. So a wide window gives its extra width to the scrubber, and a narrow
991# one takes it from the scrubber first. Every row of this grid gets the same
992# floors, so the rows still line up.
993SELECTOR_ROW_GRID = [2.6, 2.3, 5.1, 1.0]
995#: UX-181: the minimum width of each `SELECTOR_ROW_GRID` track, in rem. `None`
996#: is no floor (the scrubber). The actions floor fits A's ◀ ▶ ⇅ 🔎 ✏️ — five
997#: pills, measured at ~15rem with their gaps (2026-10-07; ✏️ moved here from
998#: the chip row). The dataset floor fits the dataset
999#: dropdown beside its + menu, and the trial floor fits a OneStop trial id.
1000#: `styles.py` caps the two picker floors at a share of the row
1001#: (`SELECTOR_ROW_FLOOR_CAPS`), so a narrow window still has room for the
1002#: scrubber.
1003SELECTOR_ROW_FLOORS_REM = (15.75, 12.5, None, 15.0)
1005#: UX-181: the share of the row that caps each picker floor, as a percentage.
1006SELECTOR_ROW_FLOOR_CAPS = (25, 20, None, None)
1008#: The pre-UX-64 three-track shape, for the rows that still have three things
1009#: on them — the multipart screen navigator and compare mode's own picker rows.
1010#: Built by merging the trial and scrub tracks, so their outer boundaries still
1011#: line up with the four-track row above them (which is the whole point of
1012#: sharing one grid — see the note at the top of this block).
1013SELECTOR_ROW_TRIO = [
1014 SELECTOR_ROW_GRID[0],
1015 SELECTOR_ROW_GRID[1] + SELECTOR_ROW_GRID[2],
1016 SELECTOR_ROW_GRID[3],
1017]
1019#: The multipart screen navigator's track, appended at the right end of a trial
1020#: row (after ◀ ▶ ⇅ 🔎) when the dataset has screens: a narrow dropdown + ◀ ▶,
1021#: so the screens no longer take a row of their own under each trial row. Its
1022#: width is mostly its floor (`SELECTOR_SCREEN_FLOOR_REM`); the weight is
1023#: nominal, like the actions track's.
1024SELECTOR_SCREEN_TRACK = 1.0
1026#: The screen track's minimum width, in rem: a ~7rem dropdown beside its
1027#: ~5rem ◀ ▶ pair, then the row's ⇅ 🔎 ✏️ (~10rem, measured 160px), which follow it.
1028SELECTOR_SCREEN_FLOOR_REM = 24.5
1030#: With a screen track, the actions track keeps only ◀ ▶: its floor, in rem.
1031SELECTOR_STEPS_FLOOR_REM = 5.2
1033#: ``SELECTOR_ROW_GRID``'s three left tracks as one — for a row whose left side
1034#: is a single wide element rather than dataset + pick + scrub. Kept for
1035#: deep-link-stable layouts that still want it; UX-75 moved the chip strip off
1036#: it onto ``SELECTOR_ROW_TRIO``, so its title lands under the dataset picker
1037#: and its chips under the trial picker and scrubber.
1038SELECTOR_ROW_WIDE_GRID = [
1039 SELECTOR_ROW_GRID[0] + SELECTOR_ROW_GRID[1] + SELECTOR_ROW_GRID[2],
1040 SELECTOR_ROW_GRID[3],
1041]
1044#: VAL-7's identity check screens a sample of trials by default (PERF-6). The
1045#: 🗂️ Data page's *Check every trial* button sets this session flag to ask for
1046#: the full census instead. UI-only — it is deliberately not wire format, so it
1047#: travels in neither a share link nor a saved config.
1048TRIAL_IDENTITY_FULL_KEY = "_trial_identity_full_scan"
1050#: VAL-7's verdict used to ride a page-wide ⚠️ banner above every view, on every
1051#: run, for as long as the dataset stayed loaded — a permanent warning about a
1052#: decision that is only made twice: when a dataset is added, and when its
1053#: mapping is edited. Both moments now set this flag instead, and ``app.main``
1054#: pops it once the (already computed) report for the newly derived frames
1055#: exists — raising the verdict as a modal exactly where the Trial ID mapping
1056#: was just chosen. The value says which flow asked, so the modal's "go back"
1057#: button can name the right screen: ``"add"`` or ``"edit"``.
1058TRIAL_IDENTITY_CHECK_KEY = "_trial_identity_check_after"
1060#: #374 F30 — the name of the dataset ✅ Add dataset just stored, popped by
1061#: ``app.main`` once its frames are loaded to confirm what arrived.
1062DATASET_ADDED_KEY = "_dataset_added_name"
1065# --- UX-138 · the icon vocabulary ---------------------------------------------
1066# One Material Symbols (Rounded) icon per *concept* the app draws as chrome — a
1067# nav entry, a rail section, a subtab, a button, an alert — so the same idea
1068# looks the same everywhere and swapping one is a one-line change. Streamlit
1069# renders the shortcode in markdown and in every ``icon=`` parameter.
1070#
1071# Keyed by meaning, not by the emoji it replaced: 👁️ used to stand for the
1072# Scanpath preset, the Fixations layer *and* a data preview, and those are three
1073# entries here. Prose keeps its emoji — help text, tour bodies, docstrings,
1074# ``cli.py`` and ``docs/`` still write "the 🗂️ **Data** page" — and so do the
1075# places a shortcode cannot reach: selectbox options and dataframe cells are
1076# plain text (the trial-picker ★ 🏷️ 📝 markers, the dataset-kind tags), and
1077# Plotly text is baked into every export (▶ Play, the marker-shape previews).
1078# The typographic glyphs on buttons (◀ ▶ ⇅ ✕ ⬇ ↗) stay as they are too.
1079ICONS: dict[str, str] = {
1080 # Navigation and dialogs. `app` is the welcome tour's; the favicon itself
1081 # stays 👀 — see `app.set_page_config`'s call site.
1082 "app": ":material/visibility:",
1083 "view_scanpath": ":material/route:",
1084 "view_corpus": ":material/bar_chart:",
1085 "view_data": ":material/database:",
1086 "help": ":material/help:",
1087 "tutorials": ":material/explore:",
1088 "faq": ":material/quiz:",
1089 "about": ":material/info:",
1090 "course": ":material/school:",
1091 "question": ":material/forum:",
1092 "bug": ":material/bug_report:",
1093 "desktop": ":material/computer:",
1094 # Plot rail: design presets, layer sections and figure groups.
1095 "preset_scanpath": ":material/timeline:",
1096 "preset_custom": ":material/build:",
1097 "illustration": ":material/draw:",
1098 "fixations": ":material/blur_on:",
1099 "saccades": ":material/arrow_outward:",
1100 "stimulus": ":material/article:",
1101 "word_boxes": ":material/crop_square:",
1102 "heatmap": ":material/local_fire_department:",
1103 "raw_gaze": ":material/grain:",
1104 "plot_filter": ":material/filter_list:",
1105 "figure": ":material/aspect_ratio:",
1106 "screen": ":material/desktop_windows:",
1107 "axes": ":material/grid_on:",
1108 "labels": ":material/title:",
1109 "hover": ":material/ads_click:",
1110 "legend": ":material/legend_toggle:",
1111 "designs": ":material/palette:",
1112 "plot_controls": ":material/tune:",
1113 "animate": ":material/movie:",
1114 "compare": ":material/compare:",
1115 # Scanpath subtabs, the trial row and the welcome tour's stops.
1116 "annotations": ":material/edit_note:",
1117 "comparisons": ":material/difference:",
1118 "line_assignment": ":material/format_line_spacing:",
1119 "export": ":material/file_export:",
1120 "share": ":material/share:",
1121 "favorite": ":material/star:",
1122 "trial_filter": ":material/filter_alt:",
1123 "pick_trial": ":material/my_location:",
1124 "chips": ":material/label:",
1125 "panels": ":material/tab:",
1126 "views": ":material/dashboard:",
1127 "nav": ":material/explore:",
1128 "preview": ":material/visibility:",
1129 "python": ":material/code:",
1130 "cli": ":material/terminal:",
1131 # Generic actions.
1132 "save": ":material/save:",
1133 "download": ":material/download:",
1134 "upload": ":material/upload:",
1135 "delete": ":material/delete:",
1136 "reset": ":material/restart_alt:",
1137 "undo": ":material/undo:",
1138 "edit": ":material/edit:",
1139 "add": ":material/add:",
1140 "confirm": ":material/check:",
1141 "settings": ":material/settings:",
1142 "search": ":material/search:",
1143 "refresh": ":material/refresh:",
1144 "rename": ":material/drive_file_rename_outline:",
1145 "close": ":material/close:",
1146 "open": ":material/open_in_new:",
1147 "mute": ":material/notifications_off:",
1148 # Data → Saved on this computer, and ❓ Help → About → Debug.
1149 "recovery": ":material/history:",
1150 "debug": ":material/bug_report:",
1151 # Data page and the add-dataset wizard.
1152 "datasets": ":material/folder_open:",
1153 "folder": ":material/folder_open:",
1154 "demo": ":material/science:",
1155 "author": ":material/draw:",
1156 # UX-174 — the dataset table: the two kinds without an icon of their own,
1157 # the open dataset's badge, the row menu and the sort arrows.
1158 "private": ":material/lock:",
1159 "public": ":material/public:",
1160 "current": ":material/check_circle:",
1161 "more": ":material/more_horiz:",
1162 "sort_asc": ":material/arrow_upward:",
1163 "sort_desc": ":material/arrow_downward:",
1164 "data_mapping": ":material/assignment:",
1165 "docs": ":material/menu_book:",
1166 "auto_detected": ":material/auto_awesome:",
1167 "stats": ":material/query_stats:",
1168 "derived_tables": ":material/calculate:",
1169 # Dataset and reader metrics (ENG-36's `st.metric` rows).
1170 "participants": ":material/group:",
1171 "texts": ":material/article:",
1172 "trials": ":material/list_alt:",
1173 "words": ":material/abc:",
1174 "gaze_points": ":material/scatter_plot:",
1175 "screens": ":material/view_carousel:",
1176 "reading_speed": ":material/speed:",
1177 "fixation_duration": ":material/timer:",
1178 "regressions": ":material/keyboard_backspace:",
1179 "skip_rate": ":material/fast_forward:",
1180 "saccade_amplitude": ":material/arrow_range:",
1181 # Setup-step badges (`wizard_shell.StepStatus`).
1182 "step_done": ":material/check_circle:",
1183 "step_action": ":material/error:",
1184 "step_todo": ":material/radio_button_unchecked:",
1185 "step_optional": ":material/remove:",
1186 # Geometry provenance of a dataset's word boxes.
1187 "geometry_real": ":material/verified:",
1188 "geometry_reconstructed": ":material/build:",
1189 "geometry_synthesized": ":material/science:",
1190 # Alerts, toasts and inline status.
1191 "warning": ":material/warning:",
1192 "error": ":material/block:",
1193 "success": ":material/check_circle:",
1194 "info": ":material/info:",
1195 "tip": ":material/lightbulb:",
1196 "participant": ":material/person:",
1197 "trial_metadata": ":material/table:",
1198 "text_metadata": ":material/article:",
1199 # About dialog.
1200 "code": ":material/code:",
1201 "doi": ":material/bookmark:",
1202 "ai": ":material/smart_toy:",
1203 "update": ":material/update:",
1204 "missing_bundle": ":material/inventory_2:",
1205}
1208def plural(count: int, noun: str, plural_noun: str | None = None) -> str:
1209 """``"1 trial"`` / ``"3 trials"`` — a count with its noun agreeing.
1211 For a caption, in place of ``trial(s)``. ``plural_noun`` is for a noun that
1212 does not take an *s* (``"entry"`` → ``"entries"``).
1213 """
1214 word = noun if count == 1 else (plural_noun or f"{noun}s")
1215 return f"{count:,} {word}"
1218_MARKDOWN_SPECIALS = re.compile(r"([\\`*_{}\[\]<>()#+\-.!|~:$])")
1221def spoken(name: str) -> str:
1222 """``name`` as an icon-only button's accessible name (UX-200).
1224 A button whose face is a glyph (◀, ⇅) or an icon reads, to a screen reader,
1225 as the glyph's own name ("black left-pointing triangle") or the icon's
1226 ligature ("folder_open"). Append this to its label — ``f"◀ {spoken('Previous
1227 trial')}"``, or on its own beside ``icon=`` — and the button is named for
1228 what it does. The text is an ``<em>`` that ``styles.py`` clips to nothing,
1229 the ``.sps-sr-only`` way, so the face stays the glyph alone; markdown in
1230 ``name`` is escaped so a design called ``*draft*`` stays text.
1232 Buttons only. A popover passes its raw label to its dialog's
1233 ``aria-label`` as well, asterisks and all, so a popover is named with a
1234 plain label clipped by key instead (`styles.py`, BUG-108).
1235 """
1236 escaped = _MARKDOWN_SPECIALS.sub(r"\\\1", name)
1237 return f"*{escaped}*"
1240def icon_html(concept: str) -> str:
1241 """``ICONS[concept]`` for raw HTML, where a ``:material/…:`` shortcode is inert.
1243 A ``<span>`` in the Material Symbols font Streamlit already ships, so the
1244 ligature — the icon's snake-case name — draws as the same glyph the
1245 shortcode renders. Styled by ``.sps-icon`` in ``styles.py``.
1246 """
1247 return icons_to_html(ICONS[concept])
1250_SHORTCODE = re.compile(r":material/([a-z0-9_]+):")
1253def icons_to_html(text: str) -> str:
1254 """``text`` with every ``:material/…:`` shortcode drawn as :func:`icon_html` does.
1256 For a label that arrives as markdown (``ICONS[…]`` and all) but is written
1257 into a raw HTML block — a line starting ``<div`` — where Streamlit leaves
1258 the shortcode as literal text.
1259 """
1260 return _SHORTCODE.sub(r'<span class="sps-icon" aria-hidden="true">\1</span>', text)
1263#: The Scanpath view's subtab labels. Named because the set is no longer fixed:
1264#: PRE-21 offers Line assignment only while drift correction is exposed, so the
1265#: tabs are built as a list and mapped back by label. They live here rather than
1266#: in `tabs` because the tutorial steps in `tour` open a subtab by its label too,
1267#: and `tests/conftest.py` imports them rather than repeating the strings.
1268SUBTAB_ANNOTATIONS = f"{ICONS['annotations']} Annotations"
1269SUBTAB_STIMULUS = f"{ICONS['stimulus']} Stimulus & context"
1270SUBTAB_COMPARISONS = f"{ICONS['comparisons']} Comparisons"
1271SUBTAB_LINE_ASSIGNMENT = f"{ICONS['line_assignment']} Line assignment"
1272SUBTAB_EXPORT = f"{ICONS['export']} Export"
1273SUBTAB_SHARE = f"{ICONS['share']} Share"
1276# --- Legend layout ------------------------------------------------------------
1277# Where each legend sits (📐 Figure & canvas → Legends; `plots.apply_legend_layout`).
1278# The values are wire format — share links, saved configs, `render --legend` —
1279# so never rename one.
1281#: The legends a figure can draw, in the order the controls list them.
1282LEGEND_KINDS = ("compare", "saccades", "colors", "size_key")
1283#: What the controls call each legend.
1284LEGEND_KIND_LABELS = {
1285 "compare": "Compare (A/B)",
1286 "saccades": "Saccade types",
1287 "colors": "Fixation colours",
1288 "size_key": "Size key",
1289}
1290#: Where a legend can go: outside the plot on a side, or inside a corner.
1291LEGEND_POSITIONS = (
1292 "auto",
1293 "above",
1294 "below",
1295 "left",
1296 "right",
1297 "top-left",
1298 "top-right",
1299 "bottom-left",
1300 "bottom-right",
1301)
1302LEGEND_POSITION_LABELS = {
1303 "auto": "Auto",
1304 "above": "Above",
1305 "below": "Below",
1306 "left": "Left",
1307 "right": "Right",
1308 "top-left": "Inside top-left",
1309 "top-right": "Inside top-right",
1310 "bottom-left": "Inside bottom-left",
1311 "bottom-right": "Inside bottom-right",
1312}
1313#: How a legend's items run: one under the other, or side by side.
1314LEGEND_ARRANGEMENTS = ("auto", "stacked", "side-by-side")
1315LEGEND_ARRANGEMENT_LABELS = {
1316 "auto": "Auto",
1317 "stacked": "Stacked",
1318 "side-by-side": "Side by side",
1319}