Coverage for scanpath_studio/constants.py: 99%

214 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 21:10 +0000

1"""Shared constants for the Scanpath Studio app.""" 

2 

3from __future__ import annotations 

4 

5import os 

6import re 

7 

8PACKAGE_NAME = "scanpath_studio" 

9 

10#: On raw gaze imported with no clock: each sample's position in its trial, 1, 

11#: 2, … — in place of `timestamp_ms`, which such a table does not have 

12#: (`data.normalize_raw_gaze`). A column the user sees and exports, the sample 

13#: number, so the plot can colour by order without calling it time. 

14SAMPLE_INDEX = "sample_index" 

15 

16 

17# --- PRE-21: features that are built but not fully integrated ---------------- 

18# Vertical drift correction (the PRE-3 port of Carr et al. 2021) and the NLD 

19# similarity scoring under 🔬 Comparisons both work, and both are half-wired: 

20# the similarity table still shows three metrics as "Not yet computed", and the 

21# drift-correction subtab's follow-ups (PRE-9, PRE-10) are unfinished. Shipping 

22# them visible invites users to lean on them, so ahead of publication they are 

23# gated off — a **visibility gate, not a removal**: `alignment.py` and 

24# `similarity.py` stay exactly where they are and stay reachable for us. 

25# 

26# Polarity is the opposite of `app.public_datasets_enabled`: these default 

27# **off** and the env var turns them on. Read at *call time*, which is both what 

28# lets a test toggle them and what stops a stale import-time read from making 

29# one surface disagree with another. 

30EXPERIMENTAL_ENV_VAR = "SCANPATH_EXPERIMENTAL" 

31 

32 

33def experimental_features_enabled() -> bool: 

34 """Whether the not-fully-integrated features are exposed (PRE-21). 

35 

36 Off unless ``SCANPATH_EXPERIMENTAL`` is set to something truthy. 

37 """ 

38 return os.environ.get(EXPERIMENTAL_ENV_VAR, "").strip().lower() in ( 

39 "1", 

40 "true", 

41 "yes", 

42 "on", 

43 ) 

44 

45 

46def drift_correction_enabled() -> bool: 

47 """Whether vertical drift correction / line assignment is exposed (PRE-21).""" 

48 return experimental_features_enabled() 

49 

50 

51def similarity_enabled() -> bool: 

52 """Whether NLD scanpath-similarity scoring is exposed (PRE-21).""" 

53 return experimental_features_enabled() 

54 

55 

56def multipleye_upload_enabled() -> bool: 

57 """Whether the wizard's "Dataset format" choice + MultiplEYE upload branch 

58 are exposed **in the app**, this release (UX-114). 

59 

60 Held back the same way PRE-22 holds back the preprocessing panel: the code 

61 (`wizard._render_multipleye_upload`, the format `segmented_control`) stays 

62 in place for a later revival — most of what it did by hand is now what the 

63 generalized Generic wizard can do too (UX-113's filename-derive + 

64 block-aware char-AOI aggregation), so it may shrink further rather than 

65 simply come back — but this release ships with no format *choice* at all: 

66 every upload goes through Generic. Scope is the **app** only; 

67 `datasets.multipleye_frames_from_uploads`/`load_multipleye_uploads` are 

68 untouched and still callable directly. 

69 """ 

70 return experimental_features_enabled() 

71 

72 

73def multipleye_enabled() -> bool: 

74 """Whether the MultiplEYE corpus is offered, this release (DATA-54). 

75 

76 Held back for the beta: its data is not openly available yet, and the loader 

77 was built and tested on one sample (Zurich Chinese). Off, the entry leaves 

78 the public-dataset registry — so the data picker, the 🗂️ Data page, Compare's 

79 second dataset and share links all stop offering it — and `render`'s 

80 ``--source multipleye`` / ``--export`` / ``--no-question-screens`` flags are 

81 hidden from ``--help``. Like PRE-22's gate, this hides rather than breaks: 

82 those flags still parse, and `datasets.load_multipleye` is untouched, so a 

83 script that already uses them keeps working. 

84 """ 

85 return experimental_features_enabled() 

86 

87 

88def benchmark_corpora_enabled() -> bool: 

89 """Whether the harmonised benchmark corpora are offered, this release 

90 (DATA-54, DATA-55). 

91 

92 A bundle can only be built with the EyeGenBench pipeline, which is not public 

93 yet, and the corpora are unfinished work — the picker marks each one (WIP). 

94 The app no longer discovers them at all (DATA-55: a corpus is listed only 

95 once someone adds it, and the flow that adds one is DATA-56). Off, this also 

96 keeps an added corpus out of `app.public_dataset_registry`, and hides 

97 `render`'s ``--eyegenbench`` / ``--eyegenbench-dataset`` flags from 

98 ``--help``. Hidden, not removed: those flags still parse, and 

99 `eyegenbench.load_eyegenbench` is untouched. 

100 """ 

101 return experimental_features_enabled() 

102 

103 

104def preprocessing_enabled() -> bool: 

105 """Whether the soft-exclusion / merge pipeline is exposed **in the app** (PRE-22). 

106 

107 The feature is finished and tested; it is held back from this release's UI 

108 and picked up in the next one, so the same flag that carries PRE-21's 

109 unfinished work carries it — one env var for "not in this release", one code 

110 path, and no branch to rebase. 

111 

112 `api.preprocess_data` and `scanpath-studio analyze` are held back by 

113 `computed_measures_enabled` instead, which raises rather than hides. 

114 """ 

115 return experimental_features_enabled() 

116 

117 

118def computed_measures_enabled() -> bool: 

119 """Whether the numbers Scanpath Studio works out itself are exposed, this release. 

120 

121 Held back until each is checked by hand: the per-word reading measures 

122 (`api.compute_word_metrics`), the reader / trial summaries, the analysis 

123 tables and `scanpath-studio analyze`, the export bundle's measure family, 

124 and the Corpus Analysis views built on them (Reading summary, Progressive vs 

125 regressive, Landing-position curve, Reader summary table). What stays is 

126 everything that shows the dataset's own values. Unlike PRE-22's app-only 

127 gate, the API raises and the CLI refuses: a script that gets no number is 

128 better off than one that gets an unchecked one. 

129 """ 

130 return experimental_features_enabled() 

131 

132 

133def sentence_analysis_enabled() -> bool: 

134 """Whether Corpus Analysis offers its **Per sentence** subtab (AN-33). 

135 

136 Held back: it works its numbers out from the fixation table, while the rest 

137 of the page shows only the measures the dataset brought (AN-32), so a 

138 dataset with supplied measures and no fixations read as 0 ms and skipped. 

139 It comes back once it is built on the supplied measures. Off, the subtab 

140 is not drawn and its table is never computed; `preprocessing.sentence_measures` 

141 is untouched (`api.analysis_tables` is held back by `computed_measures_enabled`). 

142 """ 

143 return experimental_features_enabled() 

144 

145 

146def derived_analysis_tables_enabled() -> bool: 

147 """Whether the 🗂️ Data page's "🧮 Derived analysis tables" section (Sentences 

148 / Saccades / Trials / Readers / Characters) is exposed **in the app** (UX-126). 

149 

150 Same gate, same reasoning as `preprocessing_enabled()`/`drift_correction_enabled()`: 

151 the feature is finished and tested, held back for a later release. Off, the 

152 section doesn't render *and* its backing computation (`tabs._c_derived_tables` 

153 — full reading-measure + saccade/sentence/reader aggregation) never runs, not 

154 just its display — a Data page rerun with the flag off does none of that work. 

155 Scope is the **app** only: `preprocessing.py`'s own table builders are 

156 untouched and still directly callable. 

157 """ 

158 return experimental_features_enabled() 

159 

160 

161# Default text font. A single generic family that renders (monospaced) on every 

162# platform including the Streamlit Cloud demo; the font field accepts any CSS 

163# font name or stack if you want the exact experiment font. 

164FONT_FAMILY = "monospace" 

165 

166# The demo corpus' own presentation monitor, so a figure with no declared 

167# canvas still renders true-to-scale. Sourced once, in 

168# `eyegenbench_geometry.DISPLAY_SPECS["onestop"]` (Berzak et al. 2025, Sci Data 

169# 12:1995 — Dell U2715H, 2560 px × 1440 px over 597 mm × 336 mm). 

170DEFAULT_FIGURE_SIZE = (2560, 1440) 

171 

172# Reading text is drawn true-to-scale: one line of text fills ``1/line_spacing`` 

173# of the line pitch (the word-box height that the data already encodes). OneStop 

174# rendered each line of text with one blank line above and one below it, so the 

175# line pitch is 3x the single-line height — hence a default line spacing of 3. 

176DEFAULT_LINE_SPACING = 3.0 

177 

178COLORSCALES = [ 

179 "Blues", 

180 "Greens", 

181 "Oranges", 

182 "Reds", 

183 "Purples", 

184 "Greys", 

185 "Viridis", 

186 "Plasma", 

187 "Inferno", 

188 "Magma", 

189 "Cividis", 

190 "Turbo", 

191 "Hot", 

192 "YlOrRd", 

193 "YlGnBu", 

194 "RdBu", 

195 "Spectral", 

196] 

197 

198# Both colour scales open in Blues: one hue, light to dark, colourblind-safe. 

199# A keyed selectbox first-rendered inside a popover would otherwise display its 

200# first option rather than a non-index-0 seeded value on first open — handled by 

201# `controls._popover_selectbox` (explicit `index=`) / `_pin` + `persist_state`, so a 

202# non-index-0 default here still keeps the picker and the figure in sync. 

203DEFAULT_FIXATION_COLORSCALE = "Blues" 

204DEFAULT_HEATMAP_COLORSCALE = "Blues" 

205#: Heatmap styles that scale their smoothed density to each figure's own peak 

206#: (`plots._add_interpolated_heatmap`), so a ``heatmap_range`` does nothing to 

207#: them: the rail greys the range for these, and the code snippet omits it. 

208#: Compare always draws word boxes, where the range applies again. 

209SELF_SCALED_HEATMAP_STYLES = frozenset({"Interpolated"}) 

210 

211DEFAULT_MARKER_SIZE_RANGE = (8, 24) 

212# How fixation duration maps onto that size range. The three *fixed* scales map 

213# one duration range (ms, below) onto it for every figure, so a 200 ms fixation 

214# is the same size in any trial, comparison side, replay or export — EyeLink 

215# Data Viewer's convention. "relative" is the original behaviour: each figure 

216# spans its own shortest-to-longest duration, so sizes only compare within it. 

217# √ is the default because a marker's size is its diameter: √duration makes the 

218# marker's area grow in step with duration. 

219MARKER_SIZE_SCALES = { 

220 "sqrt": "√ duration (area)", 

221 "linear": "Linear (diameter)", 

222 "log": "Log duration", 

223 "relative": "Relative to this figure", 

224} 

225DEFAULT_MARKER_SIZE_SCALE = "sqrt" 

226#: What a figure saved before the fixed scale existed was drawn with — the 

227#: migration target for old saved configs and Share links. 

228LEGACY_MARKER_SIZE_SCALE = "relative" 

229# 50–600 ms covers 99% of the bundled OneStop fixations (1st percentile 51 ms, 

230# 99th 448 ms), sits below the 80 ms short-fixation flag, and leaves the long 

231# tail distinguishable before it clamps. Durations outside it clamp to the 

232# smallest / largest marker. 

233DEFAULT_MARKER_DURATION_RANGE = (50, 600) 

234#: The duration-bounds widget's own limits (ms). 

235MARKER_DURATION_BOUNDS = (10, 3000) 

236DEFAULT_ORDER_FONT_COLOR = "#111111" 

237 

238WORD_BOX_COLOR = "#6c757d" 

239#: The outline's opacity; 1 draws it solid, as before the setting existed. 

240WORD_BOX_LINE_OPACITY = 1.0 

241#: The word boxes' fill, drawn translucent (``WORD_BOX_FILL_OPACITY``) so it 

242#: tints the interest area without hiding the text, fixations or image under it. 

243WORD_BOX_FILL_COLOR = "#646464" 

244WORD_BOX_FILL_OPACITY = 0.05 

245# VIZ-32: black, matching the colourblind-safe default palette. 

246WORD_LABEL_COLOR = "#000000" 

247# Default colour for highlighted ("Mark text") reading text — vermillion, 

248# matching the colourblind-safe default palette. The visualization controls 

249# expose a picker that overrides it per figure. 

250HIGHLIGHTED_TEXT_COLOR = "#D55E00" 

251# VIZ-32: reddish purple, matching the colourblind-safe default palette. 

252SACCADE_COLOR = "#CC79A7" 

253TRENDLINE_COLOR = "#dc3545" 

254CURRENT_FIX_COLOR = "rgba(255, 127, 14, 0.6)" 

255CURRENT_FIX_OUTLINE = "#ff7f0e" 

256FIX_MARKER_OUTLINE = "#111" 

257COMPARISON_PALETTE = ["#1f77b4", "#e45756"] 

258 

259 

260def compare_palette_color(idx: int) -> str: 

261 """Default A/B colour for comparison scanpath ``idx`` — the single source of 

262 truth shared by the per-scanpath style controls (``controls._seed_compare_styles`` 

263 / ``_collect_compare_styles``) and the figure builders 

264 (``plots._comparison_scanpath_style``), so the swatch shown in the controls can 

265 never drift from what's drawn (CMP-3).""" 

266 return COMPARISON_PALETTE[idx % len(COMPARISON_PALETTE)] 

267 

268 

269#: Each comparison scanpath's default marker alpha, shared the same way: the 

270#: rail seeds ``cmp{idx}_opacity`` from it (``controls._seed_compare_styles``) 

271#: and the builder falls back to it (``plots._comparison_scanpath_style``). 

272#: CMP-20: the builder's own 1.0 was what every headless ``compare_scanpaths`` / 

273#: ``render --compare-with`` drew, so the default comparison differed from the 

274#: app's. 0.7 matches the single-trial default, so overlaps show through. 

275COMPARE_FIXATION_OPACITY = 0.7 

276 

277 

278# Saccade line styles offered in the plot rail. Maps the friendly UI label to the 

279# Plotly ``line.dash`` value used in the figure builders. 

280SACCADE_DASH_OPTIONS = { 

281 "Solid": "solid", 

282 "Dashed": "dash", 

283 "Dotted": "dot", 

284 "Dash-dot": "dashdot", 

285} 

286# Saccade line width (px): default + the (min, max) the width slider allows. 

287DEFAULT_SACCADE_WIDTH = 2.0 

288SACCADE_WIDTH_BOUNDS = (0.5, 10.0) 

289#: The Interpolated heatmap's fixed blur σ (px): the box's limits and default. 

290HEATMAP_SIGMA_BOUNDS = (1.0, 500.0) 

291DEFAULT_HEATMAP_SIGMA_PX = 20.0 

292 

293# VIZ-8 · colour saccades by reading type. Each saccade (the segment from one 

294# fixation to the next) is classified into one of these reading-schematic 

295# classes by ``measures.classify_saccades`` and — in the "By type" colour mode — 

296# drawn as its own sub-trace with a small legend. ``other`` is the catch-all for 

297# saccades that can't be classified (an endpoint fell outside every word box); it 

298# isn't user-editable, so the palette UI exposes only the five reading classes. 

299# Order controls the legend order. 

300SACCADE_CLASS_ORDER = [ 

301 "forward", 

302 "skip", 

303 "refixation", 

304 "return_sweep", 

305 "regression", 

306 "other", 

307] 

308SACCADE_CLASS_LABELS = { 

309 "forward": "Forward", 

310 "skip": "Skip", 

311 "refixation": "Refixation", 

312 "return_sweep": "Return sweep", 

313 "regression": "Regression", 

314 "other": "Other", 

315} 

316# VIZ-32: Okabe-Ito, matching the colourblind-safe default palette. 

317SACCADE_CLASS_COLORS = { 

318 "forward": "#009E73", # bluish green — normal left-to-right progression 

319 "skip": "#56B4E9", # sky blue — jumps over one or more words 

320 "refixation": "#CC79A7", # reddish purple — lands back on the same word 

321 "return_sweep": "#E69F00", # orange — long sweep to the next line 

322 "regression": "#D55E00", # vermillion — moves backward 

323 "other": "#999999", # grey — unclassifiable (off-text endpoint) 

324} 

325# The five reading classes the palette UI lets the user recolour (``other`` is 

326# fixed grey). 

327SACCADE_CLASS_EDITABLE = SACCADE_CLASS_ORDER[:-1] 

328 

329# VIZ-19 · saccade colour modes. The five-way "By type" split is more than most 

330# figures need, so there's a middle option between one flat colour and the full 

331# reading-class breakdown: "Forward / regression", the distinction almost every 

332# reading paper actually draws. It reuses the same per-class machinery — the 

333# classes are just folded into two buckets before the segments are built, so the 

334# colour pickers, the legend toggle and every surface stay as they are. 

335SACCADE_COLOR_MODES = ("Uniform", "Forward / regression", "By type") 

336SACCADE_DIRECTION_CLASSES = ("forward", "regression") 

337# reading class → the bucket it is drawn in under "Forward / regression". 

338# ``other`` stays ``other`` (grey catch-all) so unclassifiable saccades aren't 

339# silently counted as progressive. 

340SACCADE_DIRECTION_FOLD = { 

341 "forward": "forward", 

342 "skip": "forward", 

343 "refixation": "forward", 

344 "return_sweep": "forward", 

345 "regression": "regression", 

346 "other": "other", 

347} 

348SACCADE_DIRECTION_LABELS = { 

349 "forward": "Forward", 

350 "regression": "Regression", 

351 "other": "Other", 

352} 

353 

354# VIZ-15 · fixation marker shape. Plotly symbol name → the label shown in the 

355# picker. Shape is a *second* encoding channel, and unlike hue it survives 

356# greyscale printing — so it pairs with VIZ-17 (colour freed up once it stops 

357# duplicating size) and VIZ-18 (print / colourblind palettes). 

358FIXATION_SYMBOLS = { 

359 "circle": "● Circle", 

360 "square": "■ Square", 

361 "diamond": "◆ Diamond", 

362 "triangle-up": "▲ Triangle", 

363 "cross": "✚ Cross", 

364 "x": "✖ X", 

365 "star": "★ Star", 

366 "hexagon": "⬡ Hexagon", 

367 "heart": "♥ Heart", 

368} 

369DEFAULT_FIXATION_SYMBOL = "circle" 

370 

371# Shapes Plotly's ``marker.symbol`` enum doesn't have. They're drawn as *text* 

372# glyphs instead — a Scatter in text mode, sized per point via an array 

373# ``textfont.size``, so duration→size still holds. Kept as a mapping so adding 

374# another glyph shape needs no new branch in the figure builder. 

375# Glyphs render at roughly half the visual weight of a marker of the same 

376# nominal size, so the sizes are scaled up to match the other shapes. 

377FIXATION_GLYPH_SYMBOLS = {"heart": "♥"} 

378FIXATION_GLYPH_SIZE_SCALE = 1.8 

379 

380# VIZ-17 · the "Color fixations by" option meaning *don't* map a variable to hue. 

381# Marker size already encodes fixation duration, so colouring by duration too 

382# double-encodes one variable and spends the colour channel on nothing. The 

383# default is therefore one flat colour, and colour-by is an explicit opt-in for a 

384# *different* variable (surprisal, frequency, line, pass index). 

385UNIFORM_COLOR_FIELD = "(uniform)" 

386# VIZ-32: blue, matching the colourblind-safe default palette. 

387DEFAULT_FIXATION_COLOR = "#0072B2" 

388 

389# Outline width (px) for hollow (outline-only) fixation markers. 

390HOLLOW_OUTLINE_WIDTH = 2.0 

391 

392# Distinct mark for fixations that fall outside every word box ("out of text"). 

393OUT_OF_TEXT_COLOR = "#d62728" # red 

394 

395# Plot background. Default white; some analyses prefer a neutral gray. 

396# A "Custom…" entry in the rail reveals a free color picker. 

397DEFAULT_BACKGROUND_COLOR = "#ffffff" 

398BACKGROUND_PRESETS = { 

399 "White": "#ffffff", 

400 "Light gray": "#e9ecef", 

401 "Gray": "#bdbdbd", 

402 "Black": "#000000", 

403} 

404 

405CANVAS_PAD_MIN_PX = 20.0 

406CANVAS_PAD_FRACTION = 0.05 

407 

408 

409# --- BUG-101 · the Plotly config every figure is drawn with -------------------- 

410# plotly.js 3 defaults `showSendToCloud` to true: the modebar's "Share chart…" 

411# button uploads the whole figure — words, coordinates, hover fields — to 

412# cloud.plotly.com. Nothing in Scanpath Studio sends data anywhere unasked, and 

413# its own Share means something else, so every figure turns the button off: 

414# the app's embeds and charts, the HTML it writes, and the docs site's figures. 

415# Merge it into any other config: ``{**PLOTLY_CONFIG, "responsive": False}``. 

416PLOTLY_CONFIG: dict = {"showSendToCloud": False} 

417 

418 

419# --- VIZ-18 · selectable palettes -------------------------------------------- 

420# These figures don't only get looked at on the screen they were made on: they go 

421# into papers (printed, sometimes in black & white) and are read by colourblind 

422# viewers. One palette can't serve all of that, so the colour defaults are a 

423# *choice* rather than a constant. 

424# 

425# A palette is a preset, not a second rendering path: picking one writes the 

426# ordinary per-element colour keys, so every existing picker still overrides it 

427# and every surface (deep link, Save & restore, CLI, API) carries the resulting 

428# colours with no new plumbing. ``palette_settings`` returns the figure-kwarg 

429# form; ``controls.apply_palette`` writes the session keys. 

430# 

431# Rules each non-default palette follows: 

432# * hues distinguishable under deuteranopia/protanopia (no red-vs-green pair 

433# carrying meaning on its own), and 

434# * **lightness** ordered as well as hue, so the figure still reads after a 

435# greyscale conversion. Marker shape (VIZ-15) and the two-way saccade mode 

436# (VIZ-19) are the redundant channels when colour alone can't carry it. 

437PALETTES: dict[str, dict] = { 

438 # Okabe & Ito's eight-colour set — the de-facto standard for qualitative 

439 # colourblind-safe encoding — plus single-hue Blues scales, which vary in 

440 # lightness only and so survive every common deficiency. VIZ-32: this is the 

441 # default a fresh session opens with, not just an opt-in choice. 

442 "Default (colourblind-safe)": { 

443 "description": "Okabe–Ito hues + Blues scales; safe for deuteran-, " 

444 "protan- and tritanopia.", 

445 "fixation_color": DEFAULT_FIXATION_COLOR, 

446 "fixation_colorscale": DEFAULT_FIXATION_COLORSCALE, 

447 "heatmap_colorscale": DEFAULT_HEATMAP_COLORSCALE, 

448 "saccade_color": SACCADE_COLOR, 

449 "saccade_class_colors": dict(SACCADE_CLASS_COLORS), 

450 "word_label_color": WORD_LABEL_COLOR, 

451 "highlight_text_color": HIGHLIGHTED_TEXT_COLOR, 

452 "background_color": DEFAULT_BACKGROUND_COLOR, 

453 }, 

454 # Lightness-only encoding: everything survives a black & white print or 

455 # photocopy, because nothing depends on hue at all. 

456 "Print / greyscale": { 

457 "description": "Grays only — nothing depends on hue, so it survives a " 

458 "black & white print. Pair with marker shape.", 

459 "fixation_color": "#1a1a1a", 

460 "fixation_colorscale": "Greys", 

461 "heatmap_colorscale": "Greys", 

462 "saccade_color": "#7a7a7a", 

463 "saccade_class_colors": { 

464 # Ordered by lightness, darkest = the thing you're looking for. 

465 "forward": "#a6a6a6", 

466 "skip": "#8a8a8a", 

467 "refixation": "#5e5e5e", 

468 "return_sweep": "#c4c4c4", 

469 "regression": "#000000", 

470 "other": "#d9d9d9", 

471 }, 

472 "word_label_color": "#333333", 

473 "highlight_text_color": "#000000", 

474 "background_color": "#ffffff", 

475 }, 

476 # Maximum separation from the background and from each other — projectors, 

477 # low-quality displays, and low-vision viewers. 

478 "High contrast": { 

479 "description": "Saturated, dark-on-white hues for projectors and " 

480 "low-contrast displays.", 

481 "fixation_color": "#0033cc", 

482 "fixation_colorscale": "Cividis", 

483 "heatmap_colorscale": "Cividis", 

484 "saccade_color": "#cc0000", 

485 "saccade_class_colors": { 

486 "forward": "#006600", 

487 "skip": "#0033cc", 

488 "refixation": "#6600cc", 

489 "return_sweep": "#cc6600", 

490 "regression": "#cc0000", 

491 "other": "#4d4d4d", 

492 }, 

493 "word_label_color": "#000000", 

494 "highlight_text_color": "#cc0066", 

495 "background_color": "#ffffff", 

496 }, 

497} 

498DEFAULT_PALETTE = "Default (colourblind-safe)" 

499#: #374: the names a person reads. The keys above are stored values (settings 

500#: files, Share links, ``palette=``) and keep their spelling; every place that 

501#: shows a palette shows it through `palette_label`. 

502PALETTE_LABELS = { 

503 "Default (colourblind-safe)": "Default (colorblind-safe)", 

504 "Print / greyscale": "Print / grayscale", 

505} 

506 

507 

508def palette_label(name: str) -> str: 

509 """A palette's display name (US spelling); any other value as it is.""" 

510 return PALETTE_LABELS.get(name, name) 

511 

512 

513# Not a palette — the honest answer when the live colours match none of them. 

514# A palette only *presets* the individual colour keys, so the moment one of those 

515# pickers is changed the selector would otherwise keep naming a palette the figure 

516# no longer uses. Deliberately kept OUT of ``PALETTES`` so the registry stays the 

517# set of things that can actually be applied: `--palette` choices, the API's 

518# expansion, and the deep link all iterate `PALETTES` and must not offer this. 

519CUSTOM_PALETTE = "Custom" 

520 

521 

522def palette_settings(name: str) -> dict: 

523 """Figure-kwarg colour settings for palette ``name`` (falls back to Default). 

524 

525 Returns a fresh dict (nested ``saccade_class_colors`` copied too), so callers 

526 can mutate the result without corrupting the registry. 

527 """ 

528 entry = PALETTES.get(name) or PALETTES[DEFAULT_PALETTE] 

529 settings = {k: v for k, v in entry.items() if k != "description"} 

530 settings["saccade_class_colors"] = dict(settings["saccade_class_colors"]) 

531 return settings 

532 

533 

534# --- App theme (BUG-6) ------------------------------------------------------- 

535# The branded look. Streamlit only auto-loads ``.streamlit/config.toml`` relative 

536# to the *launch* directory, so ``streamlit run streamlit_app.py`` from ``app/`` 

537# (Streamlit Cloud) picks it up but ``python -m scanpath_studio`` from anywhere 

538# else — or a ``pip``-installed console script, which never ships that file — 

539# falls back to Streamlit's default red accent. ``cli.launch_app`` injects these 

540# as ``--theme.*`` flags so every launch path renders the same theme regardless 

541# of the working directory. Kept in sync with ``app/.streamlit/config.toml`` — 

542# ``tests/test_theme.py`` asserts parity so the two can't drift. 

543APP_THEME = { 

544 "base": "light", 

545 "primaryColor": "#1f77b4", 

546 "backgroundColor": "#ffffff", 

547 "secondaryBackgroundColor": "#f5f7fa", 

548 "textColor": "#212529", 

549 "font": "sans-serif", 

550} 

551# Dark-variant overrides ([theme.dark] in config.toml). Users switch via the ☰ 

552# menu → Settings → Appearance, or follow their OS. 

553APP_THEME_DARK = { 

554 "primaryColor": "#5aa9e6", 

555 "backgroundColor": "#0e1117", 

556 "secondaryBackgroundColor": "#1c2030", 

557 "textColor": "#e8eaed", 

558} 

559 

560#: Per-file upload ceiling, in MB. Streamlit's own default is **200 MB**, which 

561#: is far under a real eye-tracking export — a single zipped fixation report for 

562#: one OneStop regime already runs to tens of MB, and a full corpus is orders 

563#: above that. Kept in sync with ``.streamlit/config.toml``'s 

564#: ``server.maxUploadSize``; `cli._max_upload_cli_flags` passes it explicitly 

565#: because that config file is resolved against the *launch* directory, so a 

566#: pip-installed console script started from anywhere else silently fell back to 

567#: the 200 MB default. This is the transport limit only — what a machine can 

568#: actually parse is a separate question, which `data.UPLOAD_SIZE_WARN_BYTES` 

569#: answers for the memory-capped hosted demo. 

570UPLOAD_MAX_SIZE_MB = 5000 

571 

572#: ENG-68: a deployment's own, lower per-file ceiling, in MB — set on the hosted 

573#: demo (its Community Cloud secrets, which Streamlit loads into the environment 

574#: at startup) so a visitor cannot send a multi-GB file at a ~1 GB container, 

575#: while every other install keeps :data:`UPLOAD_MAX_SIZE_MB`. 

576UPLOAD_LIMIT_ENV = "SCANPATH_MAX_UPLOAD_MB" 

577 

578 

579#: The file types every table upload box accepts (``zip`` wraps any of the 

580#: others; ``txt`` is a tab-separated report; ``xls`` is a legacy workbook or 

581#: EyeLink Data Viewer's text-in-an-``.xls`` export, DATA-53 / BUG-55). 

582UPLOAD_FILE_TYPES = ("csv", "tsv", "txt", "parquet", "feather", "zip", "xlsx", "xls") 

583 

584 

585def configured_upload_limit_mb() -> int | None: 

586 """``SCANPATH_MAX_UPLOAD_MB`` as a positive whole number of MB, else ``None``. 

587 

588 The raw deployment setting, before it meets the server's own limit — what 

589 ``scanpath-studio run`` hands the server (ENG-68). 

590 """ 

591 raw = os.environ.get(UPLOAD_LIMIT_ENV, "").strip() 

592 try: 

593 limit = int(raw) 

594 except ValueError: 

595 return None 

596 return limit if limit > 0 else None 

597 

598 

599def upload_limit_mb() -> int | None: 

600 """The per-file cap every ``st.file_uploader`` passes as ``max_upload_size``. 

601 

602 ``None`` — the server's own ``server.maxUploadSize`` — unless 

603 ``SCANPATH_MAX_UPLOAD_MB`` sets one, which is held to the server's limit so 

604 the browser never accepts a file the server then refuses. Read at call 

605 time, like the other deployment switches, so tests can toggle it. 

606 

607 The per-widget cap is checked **in the browser**: Streamlit's upload route 

608 only enforces ``server.maxUploadSize``, which cannot change once the server 

609 runs. ``scanpath-studio run`` therefore passes the cap to the server too; 

610 on Community Cloud, where secrets load after the server config, a scripted 

611 client can still send up to the config file's limit. 

612 """ 

613 limit = configured_upload_limit_mb() 

614 if limit is None: 

615 return None 

616 try: 

617 import streamlit as st 

618 

619 server = int(st.get_option("server.maxUploadSize")) 

620 except Exception: 

621 server = UPLOAD_MAX_SIZE_MB 

622 return min(limit, server) 

623 

624 

625def upload_limit_label() -> str: 

626 """The per-file limit in force, as Streamlit writes it (``5GB``, ``200MB``).""" 

627 mb = upload_limit_mb() or UPLOAD_MAX_SIZE_MB 

628 return f"{mb // 1000}GB" if mb >= 1000 and mb % 1000 == 0 else f"{mb}MB" 

629 

630 

631def upload_identity(uploaded) -> tuple[str | None, str]: 

632 """Which upload this is: ``(file_id, sha256 of the bytes)``. 

633 

634 An import that applies a file once — a settings or setup file — compares 

635 this with the identity it last applied, so an ordinary rerun is a no-op 

636 while a *fresh* upload applies again. The ``file_id`` Streamlit gives each 

637 upload event makes re-uploading the very same file count as fresh; the 

638 content hash makes a different file count as fresh even where no 

639 ``file_id`` exists. Name and size alone did neither: two files can share 

640 both. 

641 """ 

642 import hashlib 

643 

644 digest = hashlib.sha256(uploaded.getvalue()).hexdigest() 

645 return getattr(uploaded, "file_id", None), digest 

646 

647 

648CITATION = { 

649 "authors": ( 

650 "Omer Shubi, Keren Gruteke Klein, Maya Grossman, Ella Lion, Deborah N. Jakobi, " 

651 "David R. Reich, Lena Jäger, Yevgeni Berzak" 

652 ), 

653 "title": "Scanpath Studio", 

654 # ENG-75: the Zenodo *concept* DOI — resolves to the latest archived 

655 # release. Kept equal to CITATION.cff's `doi` by tests/test_citation.py. 

656 "doi": "10.5281/zenodo.22933884", 

657 "url": "https://github.com/lacclab/scanpath-studio", 

658 "docs_url": "https://lacclab.github.io/scanpath-studio/", 

659 # Where a user asks, reports, and gets the desktop app. 

660 "questions_url": "https://github.com/lacclab/scanpath-studio/discussions/categories/q-a", 

661 "bug_report_url": "https://github.com/lacclab/scanpath-studio/issues/new?template=bug_report.md", 

662 "desktop_url": "https://github.com/lacclab/scanpath-studio/releases/latest", 

663 "lab_url": "https://lacclab.github.io/", 

664 "corpus_note": ( 

665 "Bundled demo data is a subset of OneStop Eye Movements: " 

666 "Berzak, Malmaud, Shubi, Meiri, Lion, Levy (2025), " 

667 '"OneStop: A 360-Participant English Eye Tracking Dataset with ' 

668 'Different Reading Regimes," Scientific Data. ' 

669 "https://doi.org/10.1038/s41597-025-06272-2" 

670 ), 

671} 

672 

673 

674# --- Data-source identity + main view labels -------------------------------- 

675# Moved out of app.py so url_state.py / wizard.py can import them without a 

676# cycle (app.py re-imports them for its own use and for tests). 

677UPLOAD_CHOICE = "Upload tables" 

678AUTHOR_CHOICE = "Author a scanpath" 

679MANUAL_SAMPLE_CHOICE = "Synthetic sample" 

680DEMO_CHOICE = "Bundled Demo" 

681SYNTHETIC_CHOICE = "Synthetic test trial" 

682PUBLIC_DATASETS_CHOICE = "Public datasets" 

683POTEC_DEFAULT_DIR = "data/PoTeC" 

684EYEGENBENCH_DEFAULT_DIR = "data/EyeGenBench" 

685# DATA-27 (Task 11R): every prepared benchmark corpus is its own top-level entry 

686# in the flat data-source picker, exactly like PoTeC / MultiplEYE / OneStop — 

687# there is no "EyeGenBench" source fronting them. **"EyeGenBench" is provenance, 

688# not a source**: it names the pipeline that harmonises the corpora and is being 

689# extracted into its own repository, so it appears in descriptions and help 

690# strings only — never in an entry label, which is built from the corpus' 

691# manifest name (`app.py`). 

692# The suffix that distinguishes a harmonised corpus from a *native* entry for the 

693# same corpus (PoTeC, OneStop ship both ways). Applied by property — the 

694# harmonised copy is re-derived and its geometry may be weaker — never by vendor 

695# name; see consequence 1 above. 

696BENCHMARK_SHORT_SUFFIX = " (harmonised benchmark)" 

697# The registry-key suffix. Keys must be unique across the whole registry, and a 

698# native entry's key is `"<Corpus> — <full name>"`, so this shape can't collide. 

699BENCHMARK_LABEL_SUFFIX = " — harmonised benchmark corpus" 

700# DATA-27 ships to main unfinished, so every harmonised corpus wears this in the 

701# source picker. It is **display only** — appended by `app._entry_label` at 

702# render time, never stored on the entry. Putting it on the registry key or on 

703# `short` would change the picker's stored `data_source_choice` value and (for a 

704# built-in) the `?corpus=` slug, so removing it later would break links and saved 

705# configs written while it was up; as a formatting step it costs one line to 

706# delete. Nothing derives identity from it. 

707BENCHMARK_WIP_SUFFIX = " (WIP)" 

708 

709# DATA-27 R35: a benchmark manifest records `language` as an **ISO 639-1 code** 

710# ('zh', 'da', 'en', …), not a name, and the picker shows it to a reader. A small 

711# explicit table beats a dependency for a field this narrow; unknown codes fall 

712# back to the code itself (never "Unknown" — the code is real information, and 

713# inventing a placeholder for it loses that). 

714LANGUAGE_NAMES = { 

715 "ar": "Arabic", 

716 "bg": "Bulgarian", 

717 "ca": "Catalan", 

718 "cs": "Czech", 

719 "da": "Danish", 

720 "de": "German", 

721 "el": "Greek", 

722 "en": "English", 

723 "es": "Spanish", 

724 "et": "Estonian", 

725 "eu": "Basque", 

726 "fa": "Persian", 

727 "fi": "Finnish", 

728 "fr": "French", 

729 "ga": "Irish", 

730 "he": "Hebrew", 

731 "hi": "Hindi", 

732 "hr": "Croatian", 

733 "hu": "Hungarian", 

734 "id": "Indonesian", 

735 "is": "Icelandic", 

736 "it": "Italian", 

737 "ja": "Japanese", 

738 "ko": "Korean", 

739 "lt": "Lithuanian", 

740 "lv": "Latvian", 

741 "mk": "Macedonian", 

742 "ml": "Malayalam", 

743 "nl": "Dutch", 

744 "no": "Norwegian", 

745 "pl": "Polish", 

746 "pt": "Portuguese", 

747 "ro": "Romanian", 

748 "ru": "Russian", 

749 "sk": "Slovak", 

750 "sl": "Slovenian", 

751 "sq": "Albanian", 

752 "sr": "Serbian", 

753 "sv": "Swedish", 

754 "ta": "Tamil", 

755 "th": "Thai", 

756 "tr": "Turkish", 

757 "uk": "Ukrainian", 

758 "ur": "Urdu", 

759 "vi": "Vietnamese", 

760 "zh": "Chinese", 

761} 

762 

763 

764def language_display(value) -> str: 

765 """An ISO language code (or comma-joined list of them) as display names. 

766 

767 A multilingual corpus records several codes in one field (`"en,de,ru"` — MECO, 

768 celer), so each part is mapped and the list re-joined. An unrecognised code 

769 renders as itself: it still identifies the language to anyone who knows it, 

770 which "Unknown" does not. A blank/absent value gives ``""`` so the caller can 

771 drop the caption rather than print an empty label. 

772 """ 

773 text = str(value or "").strip() 

774 if not text: 

775 return "" 

776 parts = [part.strip() for part in text.split(",") if part.strip()] 

777 return ", ".join(LANGUAGE_NAMES.get(part.lower(), part) for part in parts) 

778 

779 

780MULTIPLEYE_DEFAULT_DIR = "data/MultiplEYE_ZH_CH_Zurich_1_2025" 

781ONESTOP_CHOICE = "OneStop server bundle" 

782# Public OneStop (OSF download-on-demand) — distinct from the env-var 

783# ONESTOP_CHOICE server bundle. DATA-63: each reading regime is its own dataset 

784# in the flat data-source picker, holding every part of that regime; these are 

785# their registry labels. Named constants (not inline strings) so the deep-link / 

786# Share contract in url_state.py and compare_source.py can reference them 

787# without importing app.py's PUBLIC_DATASET_REGISTRY (DATA-3: shareable). 

788ONESTOP_PUBLIC_DEFAULT_DIR = "data/OneStop" 

789# Default folder for the lacclab OneStop variant (a lab-processed local export; 

790# superset schema, no download). Blank: ENG-60 — it was one maintainer's own 

791# OneDrive path, prefilled into the Data directory box for every user who picked 

792# the option. Set `ONESTOP_LACCLAB_DIR`, or type the path on the 🗂️ Data page. 

793ONESTOP_LACCLAB_DEFAULT_DIR = "" 

794ONESTOP_REGIME_LABELS = { 

795 "ordinary": "Ordinary reading", 

796 "information_seeking": "Information seeking", 

797 "repeated": "Repeated reading", 

798 "information_seeking_repeated": "Information seeking (repeated)", 

799} 

800#: DATA-63 — regime → the registry label of that regime's dataset. 

801ONESTOP_REGIME_CHOICES = { 

802 regime: f"OneStop — {label}" for regime, label in ONESTOP_REGIME_LABELS.items() 

803} 

804 

805 

806#: DATA-63 — each regime dataset's ``?source=`` token. Its own per regime, 

807#: because the token also names Compare's dataset B (``cmp_source``), which 

808#: carries no regime of its own. The DATA-3 ``onestop_public`` token stays 

809#: readable: with its ``onestop_regime`` it names one of these four. 

810ONESTOP_REGIME_SOURCE_TOKENS = { 

811 regime: f"onestop_{regime}" for regime in ONESTOP_REGIME_LABELS 

812} 

813ONESTOP_LEGACY_SOURCE_TOKEN = "onestop_public" 

814 

815 

816def onestop_regime_for_choice(choice: str | None) -> str | None: 

817 """The regime a OneStop dataset label names, else ``None``.""" 

818 return next((r for r, c in ONESTOP_REGIME_CHOICES.items() if c == choice), None) 

819 

820 

821# OneStop trial parts (screens) → display label, in presentation order. Mirrors 

822# datasets._ONESTOP_PARTS; the Parts multiselect + the parts URL/CLI contract 

823# use these keys. Paragraph is the reading passage (the default). 

824ONESTOP_PART_LABELS = { 

825 "Title": "Title", 

826 "Question_Preview": "Question preview", 

827 "Paragraph": "Paragraph", 

828 "Questions": "Question", 

829 "Answers": "Answers", 

830 "QA": "Question + answers combined (QA)", 

831 "Feedback": "Feedback", 

832} 

833# OneStop source variants → display label. `public` downloads from OSF on demand; 

834# `lacclab` reads a lab-processed local export (superset schema, no download). 

835ONESTOP_VARIANT_LABELS = { 

836 "public": "Public (OSF download)", 

837 "lacclab": "LaCC lab (local export)", 

838} 

839MULTIPLEYE_BUNDLE_CHOICE = "MultiplEYE server bundle" 

840# Default dir for the MultiplEYE *server bundle* (per-session parquet shards 

841# under `<dir>/scanpath/`), overridable via the `MULTIPLEYE_DATA_DIR` env var. 

842# Empty default so the source only appears when the env var is set (like the 

843# OneStop server bundle). Distinct from MULTIPLEYE_DEFAULT_DIR above, which is 

844# the public-corpus loader's tree. 

845MULTIPLEYE_BUNDLE_DEFAULT_DIR = "" 

846_VIEW_SCANPATH = "Scanpath Visualization" 

847_VIEW_CORPUS = "Corpus Analysis" 

848# DATA-26: the third top-level view — everything about the dataset *itself* 

849# (source, location, column mapping, contents, recording setup, preprocessing), 

850# which used to be split between the ⚙️ Configure / 🧹 Preprocessing menu 

851# popovers and a 🔎 Data Inspection subtab buried in the Scanpath view. 

852_VIEW_DATA = "Data" 

853_MAIN_TAB_LABELS = [_VIEW_SCANPATH, _VIEW_CORPUS, _VIEW_DATA] 

854 

855#: DATA-32 — the dataset table's remembered headline counts, `{token: {...}}`. 

856#: A session-state key that also travels in the recovery cache's manifest, so it 

857#: lives here rather than in `app`: `persistence` writes it and `app` fills it, 

858#: and `persistence` cannot import `app` (that is the cycle `app` already 

859#: avoids by importing `persistence` one way). 

860DATASET_COUNTS_STORE_KEY = "_dataset_counts_store" 

861 

862#: UX-174 r2 — ``{dataset token: description}``, every description the user 

863#: wrote (on the add wizard or ✏️ Edit dataset), for any kind of dataset. One 

864#: small dict persisted as a recovery-cache session key, *not* a field on an 

865#: upload's ``_datasets`` entry: every non-frame field there is part of the 

866#: cache's dataset identity, so editing one sentence would rewrite every 

867#: upload's Parquet files. Here for the same import-cycle reason as above. 

868DATASET_DESCRIPTIONS_KEY = "_dataset_descriptions" 

869 

870#: ``{dataset token: SetupSnapshot.to_dict()}`` — the recording setup the user 

871#: saved on ✏️ Edit dataset for a **built-in or public** dataset, in place of 

872#: the one the corpus declares (which is never rewritten). An upload keeps its 

873#: setup on its own ``_datasets`` entry instead. A recovery-cache session key, 

874#: like the descriptions, for the same reason. 

875DATASET_SETUP_OVERRIDES_KEY = "_dataset_setup_overrides" 

876#: Which dataset's override the ``global_*`` setup keys hold now, and what they 

877#: held before it was applied — put back when that dataset is left or its 

878#: override is reset (the shape of BUG-50's font snap). Both persisted, so a 

879#: relaunch onto the dataset does not stash the override as its own "before". 

880SETUP_OVERRIDE_FOR_KEY = "_setup_override_for" 

881SETUP_OVERRIDE_RESTORE_KEY = "_setup_override_restore" 

882#: The ``global_*`` keys an override writes (and its restore puts back). 

883SETUP_OVERRIDE_SESSION_KEYS = ( 

884 "global_canvas_width", 

885 "global_canvas_height", 

886 "global_monitor_width_mm", 

887 "global_viewing_distance_mm", 

888 "global_display_dpi", 

889 "global_base_font_size", 

890 "global_font_family", 

891 "global_line_spacing", 

892 "global_scale_text_to_boxes", 

893) 

894 

895#: UX-184 — the folder every public corpus downloads into, each in a subfolder 

896#: (``<folder>/PoTeC``), set on the 🗂️ Data page. A recovery-cache session key 

897#: so the choice survives a restart; never in a link, since it is a local path 

898#: (see `session_keys.PARAM_CORPUS`). Unset, `DOWNLOAD_DIR_ENV` decides, then 

899#: the checkout's ``data/`` or the per-user data home (ENG-59). 

900DOWNLOAD_DIR_KEY = "download_dir" 

901DOWNLOAD_DIR_ENV = "SCANPATH_STUDIO_DOWNLOAD_DIR" 

902 

903#: VIZ-45 — the raw-gaze layer's per-dataset default (`app.seed_raw_gaze_default`): 

904#: the dataset it was last decided for, and the value that decision overwrote. 

905#: Recovery-cache session keys, so a relaunch onto the same dataset does not 

906#: decide again over the user's own choice; here for the import-cycle reason 

907#: above (`persistence`, `controls` and `app` all read them). 

908RAW_GAZE_SEEDED_FOR_KEY = "_raw_gaze_seeded_for" 

909RAW_GAZE_SNAP_RESTORE_KEY = "_raw_gaze_snap_restore" 

910#: …and the dataset an open deep link's `show_raw_gaze` belongs to — the first 

911#: one decided while the link was on the URL. Session-only: a link is a visit. 

912RAW_GAZE_LINK_FOR_KEY = "_raw_gaze_link_for" 

913 

914#: DATA-35 — "the ✏️ Edit dataset screen is open". The Data page is two screens 

915#: now: the **overview** (the dataset table + what's in the open dataset) and the 

916#: **editor** (everything that configures it — source options and location, 

917#: column mapping, recording setup, trial identity, stimulus images, the two 

918#: metadata tables, preprocessing). Both are built every run and switched by key, 

919#: for exactly the reason the page itself is (see below): the editor is nothing 

920#: but widgets that drive `prepare_data`, and Streamlit drops the key of a widget 

921#: that did not render. 

922DATASET_EDITOR_OPEN_KEY = "_dataset_editor_open" 

923DATA_OVERVIEW_KEY = "data_overview" 

924DATA_EDITOR_KEY = "data_dataset_editor" 

925DATA_EDITOR_OFFSCREEN_KEY = "data_dataset_editor_offscreen" 

926 

927# DATA-26: the two keys the Data page's outer container is built with — visible 

928# when that view is active, off-screen otherwise. 

929# 

930# The setup widgets (the loaders' directory input and ⬇ Download button, the 

931# source options, the column-mapping selectboxes) *drive* `prepare_data` on 

932# every rerun, and Streamlit drops the key of a widget that did not render. The 

933# menu popovers they used to live in executed every run for exactly that reason; 

934# a page body only executes while its view is selected. So the page is rendered 

935# every run either way and simply hidden — `styles.py` gives the off-screen key 

936# `display: none` — which keeps this a re-host rather than a rewrite of every 

937# loader into a render/resolve pair. 

938DATA_PAGE_KEY = "data_setup_page" 

939DATA_PAGE_OFFSCREEN_KEY = "data_setup_page_offscreen" 

940 

941#: BUG-31 — the view the user tried to reach while the add-dataset wizard was 

942#: open. `app.main` holds them on the 🗂️ Data page and the wizard asks whether to 

943#: discard the setup; this remembers where they were headed so *Discard and 

944#: leave* can finish the trip. Session state only, never a wire format. 

945WIZARD_LEAVE_KEY = "_wizard_leave_requested" 

946 

947#: BUG-31 — the view the user answered *Keep setting up* for. The prompt is 

948#: suppressed while the nav sits on exactly that view, so choosing to stay does 

949#: not re-ask on every rerun; clicking a *different* view asks again. Session 

950#: state only, never a wire format. 

951WIZARD_STAY_KEY = "_wizard_leave_acknowledged" 

952 

953#: UX-54 — set by the dataset table's ✏️ Edit button to the dataset it opened, so 

954#: the Column mapping section below can say that is why the page changed under 

955#: the user. Read once and cleared; session state only, never a wire format. 

956FOCUS_MAPPING_KEY = "_focus_column_mapping" 

957 

958# UX-47: ONE column grid for every control row stacked above the plot — the 

959# Narrow-by row, the trial picker, the multipart screen navigator, the chip 

960# strip, and (in Compare mode) the second dataset's picker and its own Narrow-by 

961# row. Three tracks: **pick** (the row's own dropdown), **scrub** (the wide 

962# middle — a slider, a pair of multiselects, the chips), **act** (the right-hand 

963# `railbtn_*` pills, which `styles.py` packs flush right). 

964# 

965# It is a shared constant rather than a repeated literal because that is the 

966# whole feature: the rows are built in three functions across two modules, and 

967# each of them owning its own weights is exactly how they drifted apart — the 

968# source dropdown ended 50 px short of the trial dropdown directly below it, and 

969# the middle track started 38 px further left on one row than the next. Tracks 

970# span by *summing* the weights they cover (the chip strip is 3 + 5), never by 

971# inventing a second pair; a `st.columns` of merged weights differs from the 

972# three-track boundary by a fraction of one gutter (~3 px at the default 1 rem), 

973# which is below the threshold where an eye reads two edges as unaligned. 

974# 

975# Widths, not pixels: Streamlit shares the row minus its gutters out by weight, 

976# so the grid holds at every window size. 

977# UX-64 made it **four** tracks — dataset · trial · scrub · actions — because 

978# the Narrow-by row above it is gone and its dataset picker moved down onto this 

979# one. The dataset track is as wide as the trial track and deliberately does not 

980# shrink: two datasets under comparison are told apart by that label. What gave 

981# way is the scrubber (5.0 → 3.6) and the filters, which became one icon in the 

982# actions cluster rather than a labelled **More** button of their own. 

983# UX-143: the first track also holds +; borrow from the scrubber to keep the 

984# dataset name readable. The same track boundaries apply to the related rows. 

985# UX-181: the weights now favour the scrubber, and `styles.py` puts floors 

986# under the other three tracks. The actions track is a fixed width: its pills 

987# are a fixed ~11.5rem, and a proportional track grew with the window and left 

988# an empty gap to the left of ◀. Its weight is nominal. The dataset and trial 

989# tracks shrink with the window down to a floor that still shows a name in 

990# full. So a wide window gives its extra width to the scrubber, and a narrow 

991# one takes it from the scrubber first. Every row of this grid gets the same 

992# floors, so the rows still line up. 

993SELECTOR_ROW_GRID = [2.6, 2.3, 5.1, 1.0] 

994 

995#: UX-181: the minimum width of each `SELECTOR_ROW_GRID` track, in rem. `None` 

996#: is no floor (the scrubber). The actions floor fits A's ◀ ▶ ⇅ 🔎 ✏️ — five 

997#: pills, measured at ~15rem with their gaps (2026-10-07; ✏️ moved here from 

998#: the chip row). The dataset floor fits the dataset 

999#: dropdown beside its + menu, and the trial floor fits a OneStop trial id. 

1000#: `styles.py` caps the two picker floors at a share of the row 

1001#: (`SELECTOR_ROW_FLOOR_CAPS`), so a narrow window still has room for the 

1002#: scrubber. 

1003SELECTOR_ROW_FLOORS_REM = (15.75, 12.5, None, 15.0) 

1004 

1005#: UX-181: the share of the row that caps each picker floor, as a percentage. 

1006SELECTOR_ROW_FLOOR_CAPS = (25, 20, None, None) 

1007 

1008#: The pre-UX-64 three-track shape, for the rows that still have three things 

1009#: on them — the multipart screen navigator and compare mode's own picker rows. 

1010#: Built by merging the trial and scrub tracks, so their outer boundaries still 

1011#: line up with the four-track row above them (which is the whole point of 

1012#: sharing one grid — see the note at the top of this block). 

1013SELECTOR_ROW_TRIO = [ 

1014 SELECTOR_ROW_GRID[0], 

1015 SELECTOR_ROW_GRID[1] + SELECTOR_ROW_GRID[2], 

1016 SELECTOR_ROW_GRID[3], 

1017] 

1018 

1019#: The multipart screen navigator's track, appended at the right end of a trial 

1020#: row (after ◀ ▶ ⇅ 🔎) when the dataset has screens: a narrow dropdown + ◀ ▶, 

1021#: so the screens no longer take a row of their own under each trial row. Its 

1022#: width is mostly its floor (`SELECTOR_SCREEN_FLOOR_REM`); the weight is 

1023#: nominal, like the actions track's. 

1024SELECTOR_SCREEN_TRACK = 1.0 

1025 

1026#: The screen track's minimum width, in rem: a ~7rem dropdown beside its 

1027#: ~5rem ◀ ▶ pair, then the row's ⇅ 🔎 ✏️ (~10rem, measured 160px), which follow it. 

1028SELECTOR_SCREEN_FLOOR_REM = 24.5 

1029 

1030#: With a screen track, the actions track keeps only ◀ ▶: its floor, in rem. 

1031SELECTOR_STEPS_FLOOR_REM = 5.2 

1032 

1033#: ``SELECTOR_ROW_GRID``'s three left tracks as one — for a row whose left side 

1034#: is a single wide element rather than dataset + pick + scrub. Kept for 

1035#: deep-link-stable layouts that still want it; UX-75 moved the chip strip off 

1036#: it onto ``SELECTOR_ROW_TRIO``, so its title lands under the dataset picker 

1037#: and its chips under the trial picker and scrubber. 

1038SELECTOR_ROW_WIDE_GRID = [ 

1039 SELECTOR_ROW_GRID[0] + SELECTOR_ROW_GRID[1] + SELECTOR_ROW_GRID[2], 

1040 SELECTOR_ROW_GRID[3], 

1041] 

1042 

1043 

1044#: VAL-7's identity check screens a sample of trials by default (PERF-6). The 

1045#: 🗂️ Data page's *Check every trial* button sets this session flag to ask for 

1046#: the full census instead. UI-only — it is deliberately not wire format, so it 

1047#: travels in neither a share link nor a saved config. 

1048TRIAL_IDENTITY_FULL_KEY = "_trial_identity_full_scan" 

1049 

1050#: VAL-7's verdict used to ride a page-wide ⚠️ banner above every view, on every 

1051#: run, for as long as the dataset stayed loaded — a permanent warning about a 

1052#: decision that is only made twice: when a dataset is added, and when its 

1053#: mapping is edited. Both moments now set this flag instead, and ``app.main`` 

1054#: pops it once the (already computed) report for the newly derived frames 

1055#: exists — raising the verdict as a modal exactly where the Trial ID mapping 

1056#: was just chosen. The value says which flow asked, so the modal's "go back" 

1057#: button can name the right screen: ``"add"`` or ``"edit"``. 

1058TRIAL_IDENTITY_CHECK_KEY = "_trial_identity_check_after" 

1059 

1060#: #374 F30 — the name of the dataset ✅ Add dataset just stored, popped by 

1061#: ``app.main`` once its frames are loaded to confirm what arrived. 

1062DATASET_ADDED_KEY = "_dataset_added_name" 

1063 

1064 

1065# --- UX-138 · the icon vocabulary --------------------------------------------- 

1066# One Material Symbols (Rounded) icon per *concept* the app draws as chrome — a 

1067# nav entry, a rail section, a subtab, a button, an alert — so the same idea 

1068# looks the same everywhere and swapping one is a one-line change. Streamlit 

1069# renders the shortcode in markdown and in every ``icon=`` parameter. 

1070# 

1071# Keyed by meaning, not by the emoji it replaced: 👁️ used to stand for the 

1072# Scanpath preset, the Fixations layer *and* a data preview, and those are three 

1073# entries here. Prose keeps its emoji — help text, tour bodies, docstrings, 

1074# ``cli.py`` and ``docs/`` still write "the 🗂️ **Data** page" — and so do the 

1075# places a shortcode cannot reach: selectbox options and dataframe cells are 

1076# plain text (the trial-picker ★ 🏷️ 📝 markers, the dataset-kind tags), and 

1077# Plotly text is baked into every export (▶ Play, the marker-shape previews). 

1078# The typographic glyphs on buttons (◀ ▶ ⇅ ✕ ⬇ ↗) stay as they are too. 

1079ICONS: dict[str, str] = { 

1080 # Navigation and dialogs. `app` is the welcome tour's; the favicon itself 

1081 # stays 👀 — see `app.set_page_config`'s call site. 

1082 "app": ":material/visibility:", 

1083 "view_scanpath": ":material/route:", 

1084 "view_corpus": ":material/bar_chart:", 

1085 "view_data": ":material/database:", 

1086 "help": ":material/help:", 

1087 "tutorials": ":material/explore:", 

1088 "faq": ":material/quiz:", 

1089 "about": ":material/info:", 

1090 "course": ":material/school:", 

1091 "question": ":material/forum:", 

1092 "bug": ":material/bug_report:", 

1093 "desktop": ":material/computer:", 

1094 # Plot rail: design presets, layer sections and figure groups. 

1095 "preset_scanpath": ":material/timeline:", 

1096 "preset_custom": ":material/build:", 

1097 "illustration": ":material/draw:", 

1098 "fixations": ":material/blur_on:", 

1099 "saccades": ":material/arrow_outward:", 

1100 "stimulus": ":material/article:", 

1101 "word_boxes": ":material/crop_square:", 

1102 "heatmap": ":material/local_fire_department:", 

1103 "raw_gaze": ":material/grain:", 

1104 "plot_filter": ":material/filter_list:", 

1105 "figure": ":material/aspect_ratio:", 

1106 "screen": ":material/desktop_windows:", 

1107 "axes": ":material/grid_on:", 

1108 "labels": ":material/title:", 

1109 "hover": ":material/ads_click:", 

1110 "legend": ":material/legend_toggle:", 

1111 "designs": ":material/palette:", 

1112 "plot_controls": ":material/tune:", 

1113 "animate": ":material/movie:", 

1114 "compare": ":material/compare:", 

1115 # Scanpath subtabs, the trial row and the welcome tour's stops. 

1116 "annotations": ":material/edit_note:", 

1117 "comparisons": ":material/difference:", 

1118 "line_assignment": ":material/format_line_spacing:", 

1119 "export": ":material/file_export:", 

1120 "share": ":material/share:", 

1121 "favorite": ":material/star:", 

1122 "trial_filter": ":material/filter_alt:", 

1123 "pick_trial": ":material/my_location:", 

1124 "chips": ":material/label:", 

1125 "panels": ":material/tab:", 

1126 "views": ":material/dashboard:", 

1127 "nav": ":material/explore:", 

1128 "preview": ":material/visibility:", 

1129 "python": ":material/code:", 

1130 "cli": ":material/terminal:", 

1131 # Generic actions. 

1132 "save": ":material/save:", 

1133 "download": ":material/download:", 

1134 "upload": ":material/upload:", 

1135 "delete": ":material/delete:", 

1136 "reset": ":material/restart_alt:", 

1137 "undo": ":material/undo:", 

1138 "edit": ":material/edit:", 

1139 "add": ":material/add:", 

1140 "confirm": ":material/check:", 

1141 "settings": ":material/settings:", 

1142 "search": ":material/search:", 

1143 "refresh": ":material/refresh:", 

1144 "rename": ":material/drive_file_rename_outline:", 

1145 "close": ":material/close:", 

1146 "open": ":material/open_in_new:", 

1147 "mute": ":material/notifications_off:", 

1148 # Data → Saved on this computer, and ❓ Help → About → Debug. 

1149 "recovery": ":material/history:", 

1150 "debug": ":material/bug_report:", 

1151 # Data page and the add-dataset wizard. 

1152 "datasets": ":material/folder_open:", 

1153 "folder": ":material/folder_open:", 

1154 "demo": ":material/science:", 

1155 "author": ":material/draw:", 

1156 # UX-174 — the dataset table: the two kinds without an icon of their own, 

1157 # the open dataset's badge, the row menu and the sort arrows. 

1158 "private": ":material/lock:", 

1159 "public": ":material/public:", 

1160 "current": ":material/check_circle:", 

1161 "more": ":material/more_horiz:", 

1162 "sort_asc": ":material/arrow_upward:", 

1163 "sort_desc": ":material/arrow_downward:", 

1164 "data_mapping": ":material/assignment:", 

1165 "docs": ":material/menu_book:", 

1166 "auto_detected": ":material/auto_awesome:", 

1167 "stats": ":material/query_stats:", 

1168 "derived_tables": ":material/calculate:", 

1169 # Dataset and reader metrics (ENG-36's `st.metric` rows). 

1170 "participants": ":material/group:", 

1171 "texts": ":material/article:", 

1172 "trials": ":material/list_alt:", 

1173 "words": ":material/abc:", 

1174 "gaze_points": ":material/scatter_plot:", 

1175 "screens": ":material/view_carousel:", 

1176 "reading_speed": ":material/speed:", 

1177 "fixation_duration": ":material/timer:", 

1178 "regressions": ":material/keyboard_backspace:", 

1179 "skip_rate": ":material/fast_forward:", 

1180 "saccade_amplitude": ":material/arrow_range:", 

1181 # Setup-step badges (`wizard_shell.StepStatus`). 

1182 "step_done": ":material/check_circle:", 

1183 "step_action": ":material/error:", 

1184 "step_todo": ":material/radio_button_unchecked:", 

1185 "step_optional": ":material/remove:", 

1186 # Geometry provenance of a dataset's word boxes. 

1187 "geometry_real": ":material/verified:", 

1188 "geometry_reconstructed": ":material/build:", 

1189 "geometry_synthesized": ":material/science:", 

1190 # Alerts, toasts and inline status. 

1191 "warning": ":material/warning:", 

1192 "error": ":material/block:", 

1193 "success": ":material/check_circle:", 

1194 "info": ":material/info:", 

1195 "tip": ":material/lightbulb:", 

1196 "participant": ":material/person:", 

1197 "trial_metadata": ":material/table:", 

1198 "text_metadata": ":material/article:", 

1199 # About dialog. 

1200 "code": ":material/code:", 

1201 "doi": ":material/bookmark:", 

1202 "ai": ":material/smart_toy:", 

1203 "update": ":material/update:", 

1204 "missing_bundle": ":material/inventory_2:", 

1205} 

1206 

1207 

1208def plural(count: int, noun: str, plural_noun: str | None = None) -> str: 

1209 """``"1 trial"`` / ``"3 trials"`` — a count with its noun agreeing. 

1210 

1211 For a caption, in place of ``trial(s)``. ``plural_noun`` is for a noun that 

1212 does not take an *s* (``"entry"`` → ``"entries"``). 

1213 """ 

1214 word = noun if count == 1 else (plural_noun or f"{noun}s") 

1215 return f"{count:,} {word}" 

1216 

1217 

1218_MARKDOWN_SPECIALS = re.compile(r"([\\`*_{}\[\]<>()#+\-.!|~:$])") 

1219 

1220 

1221def spoken(name: str) -> str: 

1222 """``name`` as an icon-only button's accessible name (UX-200). 

1223 

1224 A button whose face is a glyph (◀, ⇅) or an icon reads, to a screen reader, 

1225 as the glyph's own name ("black left-pointing triangle") or the icon's 

1226 ligature ("folder_open"). Append this to its label — ``f"◀ {spoken('Previous 

1227 trial')}"``, or on its own beside ``icon=`` — and the button is named for 

1228 what it does. The text is an ``<em>`` that ``styles.py`` clips to nothing, 

1229 the ``.sps-sr-only`` way, so the face stays the glyph alone; markdown in 

1230 ``name`` is escaped so a design called ``*draft*`` stays text. 

1231 

1232 Buttons only. A popover passes its raw label to its dialog's 

1233 ``aria-label`` as well, asterisks and all, so a popover is named with a 

1234 plain label clipped by key instead (`styles.py`, BUG-108). 

1235 """ 

1236 escaped = _MARKDOWN_SPECIALS.sub(r"\\\1", name) 

1237 return f"*{escaped}*" 

1238 

1239 

1240def icon_html(concept: str) -> str: 

1241 """``ICONS[concept]`` for raw HTML, where a ``:material/…:`` shortcode is inert. 

1242 

1243 A ``<span>`` in the Material Symbols font Streamlit already ships, so the 

1244 ligature — the icon's snake-case name — draws as the same glyph the 

1245 shortcode renders. Styled by ``.sps-icon`` in ``styles.py``. 

1246 """ 

1247 return icons_to_html(ICONS[concept]) 

1248 

1249 

1250_SHORTCODE = re.compile(r":material/([a-z0-9_]+):") 

1251 

1252 

1253def icons_to_html(text: str) -> str: 

1254 """``text`` with every ``:material/…:`` shortcode drawn as :func:`icon_html` does. 

1255 

1256 For a label that arrives as markdown (``ICONS[…]`` and all) but is written 

1257 into a raw HTML block — a line starting ``<div`` — where Streamlit leaves 

1258 the shortcode as literal text. 

1259 """ 

1260 return _SHORTCODE.sub(r'<span class="sps-icon" aria-hidden="true">\1</span>', text) 

1261 

1262 

1263#: The Scanpath view's subtab labels. Named because the set is no longer fixed: 

1264#: PRE-21 offers Line assignment only while drift correction is exposed, so the 

1265#: tabs are built as a list and mapped back by label. They live here rather than 

1266#: in `tabs` because the tutorial steps in `tour` open a subtab by its label too, 

1267#: and `tests/conftest.py` imports them rather than repeating the strings. 

1268SUBTAB_ANNOTATIONS = f"{ICONS['annotations']} Annotations" 

1269SUBTAB_STIMULUS = f"{ICONS['stimulus']} Stimulus & context" 

1270SUBTAB_COMPARISONS = f"{ICONS['comparisons']} Comparisons" 

1271SUBTAB_LINE_ASSIGNMENT = f"{ICONS['line_assignment']} Line assignment" 

1272SUBTAB_EXPORT = f"{ICONS['export']} Export" 

1273SUBTAB_SHARE = f"{ICONS['share']} Share" 

1274 

1275 

1276# --- Legend layout ------------------------------------------------------------ 

1277# Where each legend sits (📐 Figure & canvas → Legends; `plots.apply_legend_layout`). 

1278# The values are wire format — share links, saved configs, `render --legend` — 

1279# so never rename one. 

1280 

1281#: The legends a figure can draw, in the order the controls list them. 

1282LEGEND_KINDS = ("compare", "saccades", "colors", "size_key") 

1283#: What the controls call each legend. 

1284LEGEND_KIND_LABELS = { 

1285 "compare": "Compare (A/B)", 

1286 "saccades": "Saccade types", 

1287 "colors": "Fixation colours", 

1288 "size_key": "Size key", 

1289} 

1290#: Where a legend can go: outside the plot on a side, or inside a corner. 

1291LEGEND_POSITIONS = ( 

1292 "auto", 

1293 "above", 

1294 "below", 

1295 "left", 

1296 "right", 

1297 "top-left", 

1298 "top-right", 

1299 "bottom-left", 

1300 "bottom-right", 

1301) 

1302LEGEND_POSITION_LABELS = { 

1303 "auto": "Auto", 

1304 "above": "Above", 

1305 "below": "Below", 

1306 "left": "Left", 

1307 "right": "Right", 

1308 "top-left": "Inside top-left", 

1309 "top-right": "Inside top-right", 

1310 "bottom-left": "Inside bottom-left", 

1311 "bottom-right": "Inside bottom-right", 

1312} 

1313#: How a legend's items run: one under the other, or side by side. 

1314LEGEND_ARRANGEMENTS = ("auto", "stacked", "side-by-side") 

1315LEGEND_ARRANGEMENT_LABELS = { 

1316 "auto": "Auto", 

1317 "stacked": "Stacked", 

1318 "side-by-side": "Side by side", 

1319}